diff --git a/.github/docs/deploy/step-03-github-config.md b/.github/docs/deploy/step-03-github-config.md index 0a568a3c7..6b08d83cc 100644 --- a/.github/docs/deploy/step-03-github-config.md +++ b/.github/docs/deploy/step-03-github-config.md @@ -110,6 +110,26 @@ The per-origin cert vars below are **optional overrides** — set one only if yo | `CDK_MCP_SANDBOX_EXTRA_FRAME_ANCESTORS` | — | Comma-separated extra origins (beyond `https://{CDK_DOMAIN_NAME}`) allowed to embed the MCP Apps sandbox proxy via CSP `frame-ancestors`. Set to `http://localhost:4200` to point a local SPA at this deployment. **Leave unset in production.** | | `CDK_FINE_TUNING_CORS_ORIGINS` | — | Comma-separated extra CORS origins for the SageMaker fine-tuning data bucket, beyond `https://{CDK_DOMAIN_NAME}`. Optional — fine-tuning itself is always provisioned. | +### Managed Knowledge Bases + +Every variable below is **optional**. Leave them all unset for the shipped state: the managed knowledge-base backend is deployed but **dormant** — no knowledge base is created managed, no migration runs, and the daily reconciler reports what it *would* delete without deleting anything. + +The three flags are independent opt-ins that each default to **off**. An unset GitHub Variable arrives at the deploy as an empty string, which is read as off — so forgetting one never silently arms it. + +| Variable Name | Default | Description | +|---------------|---------|-------------| +| `CDK_MANAGED_KB_NEW_DEFAULT` | `false` | Set to `true` so newly created knowledge bases are provisioned on the managed backend instead of the legacy one. Existing knowledge bases are untouched. | +| `CDK_MANAGED_KB_MIGRATION_ENABLED` | `false` | Set to `true` to let the background migration worker run at all. While unset, the worker performs no work and its schedule stays disabled. | +| `CDK_MANAGED_KB_RECONCILER_ARMED` | `false` | Set to `true` to let the daily reconciler **delete** orphaned knowledge bases. While unset the reconciler still runs and still logs every deletion it intends to make — review those logs before arming it. | +| `CDK_MANAGED_KB_PER_OWNER_BYTES` | `104857600` (100 MB) | Per-owner stored-bytes cap for the standard role tier, **in bytes**. Deliberately below the 1 GB user-files precedent: at 30,000 users a 1 GB cap permits 30 TB. | +| `CDK_MANAGED_KB_PER_OWNER_ELEVATED_BYTES` | `1073741824` (1 GB) | Per-owner cap for the elevated, admin-granted tier, **in bytes**. | +| `CDK_MANAGED_KB_PER_KB_CEILING_BYTES` | `524288000` (500 MB) | Ceiling for any single knowledge base, **in bytes**, bounding one runaway corpus inside an owner's allowance. | +| `CDK_MANAGED_KB_RETENTION_WINDOW_DAYS` | `30` | How long legacy vector data is kept after a knowledge base is promoted to the managed backend, **in days**, so a rollback stays possible. Do not set below `30`. | +| `CDK_MANAGED_KB_STORAGE_ALARM_GB` | `500` | CloudWatch alarm threshold for **fleet-wide** managed knowledge base storage, **in GB**. The per-owner caps above bound one user; this is the only thing that bounds the whole account. | +| `CDK_MANAGED_KB_DAILY_COST_ALARM_USD` | `100` | CloudWatch alarm threshold for the rolled-up daily Knowledge-Base cost, **in USD**. Set alongside the storage alarm — per-owner caps alone permit roughly two orders of magnitude more spend than expected usage. | + +> Accepted values for the three flags are `true`, `false`, `1`, `0`, or empty (empty means off). Anything else fails fast at deploy time with a message naming the variable. + --- ## 3c. Authentication diff --git a/.github/workflows/platform.yml b/.github/workflows/platform.yml index 839f7917f..da38f5201 100644 --- a/.github/workflows/platform.yml +++ b/.github/workflows/platform.yml @@ -146,6 +146,43 @@ jobs: # cdk.context.json stay inert. CDK_MCP_TOKEN_ENRICHMENT_ENABLED: ${{ vars.CDK_MCP_TOKEN_ENRICHMENT_ENABLED }} CDK_MCP_TOKEN_ENRICHMENT_CLAIMS: ${{ vars.CDK_MCP_TOKEN_ENRICHMENT_CLAIMS }} + # Managed knowledge bases (.kiro/specs/managed-kb-migration). THREE + # INDEPENDENT OPT-IN flags, all defaulting to OFF — the inverse of the + # kill-switch flags above, and the difference matters here. An unset + # GitHub Actions variable renders as an EMPTY STRING, not as absent, so + # a `!== 'false'` reading of an unset variable would resolve to TRUE and + # arm the feature on every fork. config.ts reads these with + # parseBooleanEnv, which maps both unset and empty to undefined and falls + # through to `false` (Requirement 19.8). Leave all three unset to deploy + # the managed backend without starting a fleet migration. + # + # CDK_MANAGED_KB_NEW_DEFAULT new KBs are created managed + # CDK_MANAGED_KB_MIGRATION_ENABLED the background migrator runs at all + # CDK_MANAGED_KB_RECONCILER_ARMED the daily reconciler DELETES orphans + # rather than only reporting them + # + # reconcilerArmed is the inverted one: the Reconciler is deployed and + # running from day one but DISARMED, so its judgement can be reviewed + # against real data before it deletes anything (Requirements 14.7, 19.7). + CDK_MANAGED_KB_NEW_DEFAULT: ${{ vars.CDK_MANAGED_KB_NEW_DEFAULT }} + CDK_MANAGED_KB_MIGRATION_ENABLED: ${{ vars.CDK_MANAGED_KB_MIGRATION_ENABLED }} + CDK_MANAGED_KB_RECONCILER_ARMED: ${{ vars.CDK_MANAGED_KB_RECONCILER_ARMED }} + # Storage cost controls. Byte_Caps are in BYTES (Requirement 12.2), + # defaulting to 100 MB standard / 1 GB elevated / 500 MB per knowledge + # base; the retention window is in DAYS and must stay >= 30 + # (Requirement 15.11). Leave unset to take those defaults — these exist + # so an environment can tune them without a code change. + CDK_MANAGED_KB_PER_OWNER_BYTES: ${{ vars.CDK_MANAGED_KB_PER_OWNER_BYTES }} + CDK_MANAGED_KB_PER_OWNER_ELEVATED_BYTES: ${{ vars.CDK_MANAGED_KB_PER_OWNER_ELEVATED_BYTES }} + CDK_MANAGED_KB_PER_KB_CEILING_BYTES: ${{ vars.CDK_MANAGED_KB_PER_KB_CEILING_BYTES }} + CDK_MANAGED_KB_RETENTION_WINDOW_DAYS: ${{ vars.CDK_MANAGED_KB_RETENTION_WINDOW_DAYS }} + # Fleet-level alarm thresholds (Requirement 12.13). The Byte_Caps above + # bound ONE owner; these two bound the whole account, which is the gap + # between ~$169/month expected and ~$15,000/month that per-owner caps + # alone permit. Storage is in GB (default 500), daily cost in USD + # (default 100). Leave unset to take those defaults. + CDK_MANAGED_KB_STORAGE_ALARM_GB: ${{ vars.CDK_MANAGED_KB_STORAGE_ALARM_GB }} + CDK_MANAGED_KB_DAILY_COST_ALARM_USD: ${{ vars.CDK_MANAGED_KB_DAILY_COST_ALARM_USD }} # Secrets AWS_ROLE_ARN: ${{ secrets.AWS_ROLE_ARN }} AWS_ACCESS_KEY_ID: ${{ secrets.AWS_ACCESS_KEY_ID }} diff --git a/.kiro/specs/managed-kb-migration/.config.kiro b/.kiro/specs/managed-kb-migration/.config.kiro new file mode 100644 index 000000000..0b0157564 --- /dev/null +++ b/.kiro/specs/managed-kb-migration/.config.kiro @@ -0,0 +1 @@ +{"specId": "612a1431-367c-427e-8fe0-08872d03a9b9", "workflowType": "design-first", "specType": "feature"} diff --git a/.kiro/specs/managed-kb-migration/HANDOFF.md b/.kiro/specs/managed-kb-migration/HANDOFF.md new file mode 100644 index 000000000..ca16cfbcd --- /dev/null +++ b/.kiro/specs/managed-kb-migration/HANDOFF.md @@ -0,0 +1,522 @@ +# Managed KB Migration — Handoff + +**Last updated:** 2026-08-26 (groups 11–13, group 14 backend half, tag contract, **14.3 upgrade UX + enrolment surface**) · **Branch:** `feature/kb-migration` · **Nothing deployed** + +Working state for this feature so a fresh session can pick it up without re-deriving +anything. Read this, then `tasks.md`. + +--- + +## 1. Status + +| | | +|---|---| +| Spec | Complete, audited 3× to clean. 25 requirements, 201 criteria, 0 dangling refs | +| Implementation | Groups **1–13** done, plus group 14 except 14.5. 14.4's one-click document retry is deferred. 4 subtasks left: 14.4's retry, 14.5, and group 15 | +| Tests | 617 infra (jest) · **6,603** backend (pytest, 6 m 20 s) · **1,886** frontend (vitest, 7 s) · 5 pre-existing unrelated failures | +| Deployed | **Nothing.** No `cdk deploy`, no AWS mutation, at any point | +| Feature flags | `migrationEnabled` **on in development**, off in production. `newDefault` and `reconcilerArmed` off in both (explicit `false`, set as GitHub Environment variables) | + +### Commits (16 on the branch, all pushed) + +``` +45239838 one source of truth for the managed KB tag contract +4acaa8f2 handoff reflects group 14 backend half and three more defects +e59f771c register the managed backend, fleet metrics, tagged teardown (group 14 backend) +d5e56f31 handoff reflects group 13 and four more defects +ee091971 migration dispatcher and the shadow/verify/promote/retain worker (group 13) +53476544 handoff reflects groups 11-12 and two new defects +a361fdd4 opt-in dual-read pilot that legacy always wins (group 12) +a43d80bf app-side authorization, IAM-enforced sharing, publication (group 11) +58f0c6b6 handoff document and accurate task-list state +8079f7e2 tombstone deletion sagas and the report-only reconciler (group 10) +e6936b0b ingestion consumer with exclusive engine routing (group 9) +620fa49c managed KB provisioning, retrieval and direct ingestion (group 8) +d433d6f1 per-owner byte cap with atomic reserve/commit/release (group 7) +f2e86afe clamp retrieval queries and fail closed on status (groups 5, 6) +24689de1 backend abstraction seam behind the retrieval entry point (group 4) +ffa7a408 KB_Record data layer with conditional state transitions (group 3) +5f2c98b1 spec, schema and worker platform (groups 1, 2) +``` + +**Uncommitted working tree:** the 14.3 upgrade surface — `apis/app_api/kb_upgrade/`, +two transitions appended to `kb_backend/records.py`, the Angular card and service, +three test files. See §7 for the file map and §2 for how to run it. + +### Is the feature reachable yet? + +**Yes, end to end, once the flag is on — except that nothing performs the work.** +Group 14.3 closed the last gap in the *control* path: a user can now enrol a +knowledge base, which writes a `KB#` record in `shadow` with the GSI7 work keys. +Before it, nothing wrote either, so every group could have been finished with the +feature unreachable (§5 defect 21). + +What is still missing is the **worker's deployment**, not its code. The dispatcher +and worker are Lambdas behind an undeployed image, so an enrolled record sits in +`shadow` indefinitely and the card shows perpetual progress. That is the correct +local behaviour, not a bug. + +**Three behaviour changes ARE live on the existing path** and are the only things +worth testing by hand right now: +1. The document-status filter now **fails closed** (group 6). +2. Retrieval queries are **clamped to 10,000 chars** (group 5). +3. Retrieval requires a resolved access grant (group 11). Both production callers + pass one; the parameter is required and keyword-only, so a third caller added + without one fails at the call site rather than silently serving nothing. + +--- + +## 2. Environment + +macOS host, tooling installed locally. **There is no devcontainer.** +`.kiro/steering/dev-environment.md` describes a different machine (WSL2/nspawn, +`/home/colin/...` paths) — ignore it here. + +```bash +# infrastructure +cd infrastructure && npm run build # tsc +cd infrastructure && npx jest # 611 passing + +# backend +cd backend && uv run python -m pytest tests/ -q # 6 m 20 s, 6,603 passing +cd backend && uv run ruff check + +# frontend +cd frontend/ai.client && npx ng test --watch=false # 1,886 passing, ~7 s +cd frontend/ai.client && npx ng test --watch=false --include="**/kb-upgrade*" +cd frontend/ai.client && npx tsc -p tsconfig.app.json --noEmit +``` + +There is **no eslint config** in the repo despite the steering docs mentioning +ESLint; `npx eslint` fails with "couldn't find an eslint.config.*". Type-check +with `tsc --noEmit` and build with `ng build` instead. + +### Running the upgrade UI locally + +```bash +# 1. turn the offer on (LOCAL ONLY — backend/src/.env is gitignored) +echo 'MANAGED_KB_MIGRATION_ENABLED=true' >> backend/src/.env + +# 2. app_api on :8000, reading the dev account's DynamoDB via backend/src/.env +cd backend/src/apis/app_api && uv run python main.py + +# 3. SPA on :4200 — environment.ts already points at localhost:8000 +cd frontend/ai.client && npm run start +``` + +Then edit an assistant that has documents. Without the flag the card renders +nothing at all, which is correct rather than broken. + +⚠️ **The local API writes to the real dev tables.** Enrolling writes a genuine +`KB#{id}` item under `AST#{id}`. Use a throwaway assistant; undo by deleting that +item. The upgrade will sit at "Upgrading…" forever because the worker Lambda is +not deployed — expected, not a bug. + +**Baselines that are NOT your fault:** +- 5 backend failures in `tests/agents/main_agent/{session/test_async_persistence.py,streaming/test_cancellation_state.py}` — Strands SDK contract tests looking for `cancel_signal`/`async_mode` that the installed SDK lacks. Pre-existing, unrelated. +- `ruff check src/ tests/` repo-wide reports **369** pre-existing errors in untouched files. Scope ruff to your own files. + +--- + +## 3. Constraints that will bite you + +### DynamoDB one-GSI-per-deploy limit ⚠️ RELEASE-BLOCKING + +`UpdateTable` permits exactly **one** GSI create/delete per call, and CloudFormation +issues one per changed table. Two indexes on an existing table = failed deploy + full +stack rollback. **This took production down on 2026-08-01 in release 1.12.0.** + +This feature's `GSI7` (`KbWorkIndex`) consumes the **entire** `rag-assistants` GSI +budget for whatever release ships it. If another branch adds a GSI to that table, the +two cannot ship together. + +Guards: `infrastructure/test/gsi-update-limit.test.ts` (generation) and +`scripts/release/check-gsi-update-limit.mjs` (CI, vs `origin/main`). +Regenerate: `cd infrastructure && UPDATE_GSI_INVENTORY=1 npx jest gsi-update-limit` + +### DynamoDB cannot do arithmetic in a ConditionExpression + +`storedBytes + reservedBytes + :n <= :cap` is **rejected** (`Cannot parse condition +starting at:+ reserved <= :cap`). The byte cap therefore keeps a single `totalBytes` +accumulator and compares it against a **client-computed literal** (`cap - n`). One +atomic conditional `ADD`, so concurrent reservations cannot collectively overshoot. +See `byte_cap.py`. + +### DynamoDB reserved keywords bit this feature twice + +`total` and `ttl` are both reserved. Alias via `ExpressionAttributeNames`. The failure +is a `ValidationException`, which is loud — but it also masquerades as a caught +mutation (see §4). + +### Import boundary — the reason `kb_backend` exists as its own package + +`apis.shared.assistants.__init__` imports `rag_service`, which imports the embeddings +stack at module scope. Pulling that into a Lambda image blows the size budget. + +- `kb_backend/__init__.py` is **empty**, deliberately. +- Module-level imports are **stdlib only**; `boto3` and anything heavy is + function-local. +- Enforced by `backend/tests/architecture/test_kb_backend_boundary.py`. +- `apis.shared.embeddings` is a *separate* package and is fine to use. + +### Module constants must be read at call time + +Never `def f(timeout=MODULE_CONSTANT)`. Python binds default arguments once at import, +so the constant becomes unpatchable. This cost a 33-second test that silently ignored +its own override. Use `timeout: Optional[float] = None` and resolve inside. + +### The knowledge-base component spec needs every collaborator stubbed + +`KnowledgeBaseSectionComponent` loads documents, crawls, sync policies and +connectors on hydration. Leave any of those services real and their HTTP requests +stay pending, so `fixture.whenStable()` never settles and **every test in the file +times out at 5 s** with "Test timed out in 5000ms" and no hint as to why. 30 of 31 +failed this way before the stubs went in. + +`quietCollaborators()` in `knowledge-base-section.component.spec.ts` provides all +six (`DocumentService`, `FileSourceService`, `WebSourceService`, +`SyncPolicyService`, `UserConnectorsService`, `OAuthConsentService`). Note +`OAuthConsentService`'s `completion` and `inFlightProviders` must be **signals**, +not plain values — the component calls them. + +--- + +## 4. Mutation testing — the discipline that has repeatedly paid + +Every security- or correctness-relevant assertion in this feature has been verified by +breaking the guard and watching a **specific, correct** test fail. This has caught +**seven** tests that passed with their guard removed. Do not skip it. + +**Four ways a mutation lies to you.** All four have happened here: + +| Trap | Symptom | Fix | +|---|---|---| +| Anchor never matched | reported "caught", file unchanged | `diff` the file; assert the anchor matches exactly once | +| Orphaned expression values | removing a `ConditionExpression` leaves its values unused → `ValidationException` → the **happy-path** test goes red | strip the orphaned values too, so the write genuinely succeeds | +| Syntax error | collection error mistaken for a detection | `ast.parse`/`py_compile` the mutant | +| Wrong test failed | something failed, but not the guard's test | always check **which** test failed by name | + +**Also:** never assert a constant against itself. `assert CAP == module.CAP` is a +tautology that follows the constant wherever it moves. Pin the literal, with a comment +saying why that number is a property of AWS rather than a knob. + +--- + +## 5. Defects found and fixed (do not reintroduce) + +### In my own spec + +1. **`PutMetricData` on a reserved namespace.** Req 20.10 originally scoped it to + `AWS/Bedrock/KnowledgeBases`. AWS reserves every namespace beginning with `AWS` and + rejects writes. The grant would have deployed cleanly and published **nothing**, + forever. Root cause: conflating *reading* Bedrock's own metrics (genuinely in that + namespace) with *writing* ours. Now `{projectPrefix}/ManagedKb`; Req 20.13 appended + for the read grant. +2. **Dead grant on the service role.** Same metric grant was also on the Bedrock + service role, which Bedrock assumes and which never publishes our metrics. Removed; + Req 20.10 now says "calling identities only". +3. **`managedKnowledgeBaseConfiguration={}`.** The shape has no *required* members, + but its only members are the embedding pin and encryption — so a literal `{}` makes + Req 8.5's pin unsatisfiable. "No required members" ≠ "must be empty". +4. **`float32`** → the enum value is **`FLOAT32`**; lowercase is rejected. +5. **Missing gate §14.3.** Authorization/publication was absent entirely while the + test matrix demanded tests for it. Added as Requirement 25 → group 11. +6. **Ordering issue:** tasks 4.5/4.7 reference the managed adapter, which task 8.1 + builds. Resolved with a fake backend conforming to the protocol — legitimate, since + the score conversion is adapter-local and the parity rules belong to the facade. + +### In the code + +7. **Reconciler arming bypass (MAJOR).** `lambda_handler` forwarded an `armed` field + from the invocation event, so an EventBridge target with constant + `{"armed": true}` — or anyone with `lambda:InvokeFunction` — would delete user + knowledge bases while all reviewable config said report-only. The pre-existing test + was named `test_an_event_cannot_arm_by_accident` but only covered the *string* + `"true"`; the boolean that actually armed was untested. +8. **Dispatcher over-grant undetectable.** A test asserted only one statement's shape, + so Bedrock permissions added in a *separate* statement went unnoticed. Now a + whole-role whitelist scan. +9. **Inline metadata unbounded** against a 50-attribute limit, and truncation was + alphabetical — which would have dropped `document_id`, the status filter's join + key. Reserved keys now go first. +10. **Latent config bug, twice.** `--context managedKb.x=…` sets a **flat dotted** + key; a nested-only `tryGetContext('managedKb')?.x` read silently ignores it. Hit + the byte caps and then the alarm thresholds. +11. **`{}` treated as an unreadable record (group 11).** `is_reclaim_exempt` used + `if not kb_record`, which conflated "absent, so fail closed" with "read, no + holds set". Every unheld knowledge base would have been exempt and the whole + predicate vacuous. `None` and `{}` are now distinct. Found by writing the test + first and believing it over the implementation. +12. **Requirement 25.6 had no IAM behind it (group 11).** There was no + resource-policy grant anywhere in the construct, so the sharing code would have + deployed as inert. Same category as defect 1: correct-looking, clean-deploying, + authorizes nothing. Now `grantManagedKbResourcePolicyAdmin`, on its own role, + with a test asserting no retrieval identity ever receives it. +13. **A resumed migration re-ingested everything (group 13).** The + completed-document set lived inside `migrationProgress`, which a later write + replaces wholesale, so a crash near the end of a 25-document corpus re-parsed + all 25 — 37–264 s each. Now a separate `migratedDocIds` string set updated with + `ADD` per batch. Found by the convergence property test counting a document + ingested twice. +14. **`promote_engine` permitted a second promotion (group 13).** Every guard it + had stayed true *after* a successful promotion, so two genuinely concurrent + workers would both succeed — exactly what Req 15.10 forbids. Now guarded on + `attribute_not_exists(retrievalEngine)`; rollback `REMOVE`s it, so a deliberate + re-promotion still works. +15. **Fixing 14 then broke resumption (group 13).** A resume after a successful + promotion had its write refused and marked the migration `failed` — a promoted + knowledge base with no retention window. `run_promote` now treats "already + promoted" as success, re-reading before deciding so a genuine guard failure + still raises. +16. **Four mutation-test lies, in one sitting (group 13).** A limit assertion the + final `[:limit]` trim masked; a derivation whose test was vacuous because the + priority list happened to be complete; a `match=` pattern loose enough that the + *other* check satisfied it; and an `except LeaseLost: raise` that was dead code + because the lease was taken outside the `try`. Each was fixed rather than + annotated. + +--- + +17. **Nothing registered the managed backend (group 14).** `register_backend` + was defined in task 4.2 and called by nothing. All 15 groups could have been + finished with the feature unreachable — a promoted record raises + `BackendUnavailable`, a correct fail-safe and a useless signal. Registration is + now at import, so there is no startup sequence to forget. +18. **A three-defect shell script (group 14).** `scripts/teardown/managed-kb.sh`, + all three found by *running* it: an infinite spin at a zero poll interval that + burned sixteen hours of a test run; `list | cut | grep -q` reporting false + absence when SIGPIPE became the pipeline's status under `pipefail`; and a + swallowed `list-knowledge-bases` failure reporting a clean teardown having + deleted nothing. `set -e` is suspended inside a function called in a condition, + which is why the last one was silent. +19. **Requirement 20.13 existed only as a comment (group 14).** The metrics *read* + grant was described in a comment explaining the write grant and never + implemented, so the reconciler could not have read Bedrock's own `Invocations`. + +--- + +20. **The tag contract had drifted three ways (post-group-14).** The Python wrote + keys `prefix`/`env` from variables the provisioning Lambda never receives; the + reconciler's filter was a documented *mirror* of that writer; the construct + declared different key names and exported the correct values as env vars + **nothing read**; and the teardown script read a third pair. Writer and + reconciler agreed only because both fell back to the same hardcoded defaults, + so the sole symptom was a teardown that matched nothing and reported success. + Now `kb_backend/tags.py` owns the keys and one fallback chain, and + `tests/supply_chain/test_kb_tag_contract.py` parses the TypeScript and the + shell script to assert agreement across all three languages. + + ⚠️ **Tag keys are namespaced** (`ManagedKbPrefix`, not `prefix`) because many + accounts carry an org-wide cost-allocation tag literally called `env`. Note the + KB_Record *attribute* `appKbId` is a different thing from the AWS *tag* + `ManagedKbAppKbId`; only the latter belongs to this contract. + +--- + +21. **Nothing enrolled a knowledge base (group 14.3).** The mirror image of + defect 17, and missed by it. `register_backend` made a promoted record + *servable*; this is about a record ever reaching `shadow` in the first place. + The worker only picks up records already in a migration state and the + dispatcher only sweeps GSI7, so with no enrolment surface both were correct + and inert. Task 14.3 was written as frontend-only, which is how it hid: the + missing piece was an **HTTP surface** nobody had scoped. Now + `apis/app_api/kb_upgrade/`. + +22. **A one-put enrolment would have stranded every knowledge base (group 14.3).** + `KbRecord.to_item` does not write `GSI7_PK`/`GSI7_SK` — only + `set_migration_state` maintains them. So the obvious enrolment (one + `put_item` with `migrationState="shadow"`) yields a record that reports an + upgrade in progress to every surface while being invisible to the dispatcher's + sweep **forever**: a spinner with nothing behind it, and no error anywhere. + Enrolment is therefore `create_provisioning` *then* `set_migration_state`, + both conditional. Two tests pin it, including one asserting the created record + does **not** carry `migrationState`. + +23. **An unrecognised failure reason leaked the operator's string (group 14.3).** + Found by mutation, not review. `_failure_reason` maps known tokens to + plain-language copy, and the test proved that for `ByteCapExceeded` — so + mutating the fallback to `return stored or _FAILURE_FALLBACK` **survived**, and + a user would have read `ClientError: An error occurred + (AccessDeniedException)…` in the card. Testing the mapped path proved nothing + about the unmapped one; that is the whole lesson. + `test_an_unrecognised_failure_does_not_leak_the_operator_string` pins it. + +24. **The upgrade flag never reached the service that reads it (pre-merge).** + Third instance of this feature's signature failure, and the most nearly + shipped. `kb-migration-construct.ts` sets all three `MANAGED_KB_*` booleans on + the four migration Lambdas; `app-api-environment.ts` set the byte caps and the + metric namespace but **not** `MANAGED_KB_MIGRATION_ENABLED` — which + `apis/app_api/kb_upgrade/service.py` reads to decide whether to offer the + upgrade at all. + + Setting the environment variable in GitHub would therefore have changed + nothing: the card would render `phase: "none"` for every user in every + environment, forever, with a clean deploy and no log line. Found only by + tracing where the flag is actually consumed before setting it. + + Now wired, shipped as an explicit `'false'` rather than omitted so the state + is readable in the task definition, and guarded by three tests in + `app-api-environment.test.ts`. The mutation — deleting the line, which is + precisely what the defect was — is caught. + + ⚠️ `managedKb.newDefault` has **no reader anywhere in `backend/src`**. It is + set on the Lambdas' environment and consumed by nothing, because + "new knowledge bases are created managed" is a follow-up spec (design §14.7 + steps 5–8), not this phase. Leave it off; turning it on is a no-op that reads + like a behaviour change. + +--- + +## 6. Remaining work + +| Group | Subtasks | Notes | +|---|---|---| +| **14** Surfaces | 1½ | **14.5** admin surface (filter by engine, stored bytes + document counts, bulk migrate, per-KB retry) — not started. **14.4** is surfaced but its one-click document retry is deferred; see the deferral below. 14.0–14.3, 14.6, 14.7 are **done**. | +| **15** Pre-promotion verification | 3 | The gate before any real traffic moves. | + +### Known deferrals (correct, not oversights) + +- **One-click document reprocess (Req 21.2).** Ingestion is S3-event-triggered + (`documents/ingestion/handler.py`) and there is **no reprocess endpoint** — the + only document writes are upload-url, import, upload-failed and delete. A retry + control therefore needs new backend that re-fires the pipeline against bytes + already in S3, which is a change to a live ingestion path. Deliberately not + improvised. The card directs the user to re-upload via "Add files", a retry path + that works today. **Close by building the endpoint or by amending Req 21.2 to + accept re-upload** — do not leave it ambiguous. +- **`backend/Dockerfile.kb-migration`** does not exist yet, on purpose. The real image + needs five artefacts that do not exist: the handler modules, their + `requirements.txt`, a case in `scripts/build/build-one.sh`, `backend.yml` jobs, and + entries in the **hand-maintained** lists in + `backend/tests/supply_chain/test_dockerfile_pinning.py` and + `test_lambda_image_imports.py`. Per platform-as-bootstrap, CDK ships the bootstrap + stub and the **workflow** ships the real image. +- **Reconciler EventBridge wiring** (Reqs 14.1, 14.7) — `infrastructure/`, platform + group. Backend code never deploys before the IAM and resources it requires. +- **Group 7's snapshot reservation now has its caller** (`run_shadow`), reserving + the whole corpus before anything is provisioned. + +--- + +## 7. File map + +``` +.kiro/specs/managed-kb-migration/ requirements.md · design.md · tasks.md · HANDOFF.md +docs/specs/bedrock-managed-kb-evaluation.md the measured source of truth + +backend/src/apis/shared/kb_backend/ + __init__.py EMPTY, deliberately + records.py KB_Record + conditional transitions + protocol.py KnowledgeBaseBackend + frozen Chunk (score = relevance) + resolver.py engine → backend registry; absence ⇒ legacy; load_record + s3vectors_backend.py legacy adapter; converts distance → relevance HERE + managed_backend.py ManagedKbBackend: retrieval + direct ingestion + provisioning.py create saga + CUSTOM connector data source + byte_cap.py reserve / commit / release + tombstones.py delete sagas + resource_policy.py IAM-enforced sharing; staleness is state, not an event + dual_read.py pilot: start early, detach, compare, serve legacy + idleness.py activity = max(retrieval, bound agents' use) + tags.py THE tag contract — keys + value resolution, one place + query_guard.py 10,000-char clamp + metrics.py namespace + best-effort emit_count / emit_value + +backend/src/apis/shared/assistants/ + rag_service.py the FACADE — access gate, dual read, status filter, caps + kb_access.py KbAccess grant; reuses resolve_assistant_permission + kb_publication.py engine swap ≠ corpus change; reclaim exemption + +backend/src/apis/app_api/kb_migration/ + ingestion_consumer.py routes by engine; legacy ⇒ do nothing + reconciler.py daily join, report-only + dispatcher.py sparse-index sweep, bounded, no-ops when the flag is off + worker.py ONE step per invocation, leased, resumable + +backend/src/apis/app_api/kb_upgrade/ the OWNER-FACING surface (HTTP only) + models.py camelCase wire models; UpgradePhase + DocumentIssueKind + service.py phase derivation, enrolment, retry, notice, doc triage + routes.py 4 endpoints; read is any permission, writes are edit-only + NOT in kb_migration/: that package's modules share one + size-constrained Lambda image and this one imports the + embeddings-pulling assistants package + +frontend/ai.client/src/app/knowledge-base/ + kb-upgrade.service.ts fails soft; getStatus resolves to phase 'none' + knowledge-base-section.component.* the card: offer / progress / notice / failure + + the stranded-document disclosure + +scripts/teardown/ + managed-kb.sh delete tag-matched KBs BEFORE any stack + +docs/specs/ + managed-kb-cost-attribution.md filter on usagetype, never service code alone + +infrastructure/lib/constructs/managed-kb/ + managed-kb-role-construct.ts Bedrock service role + grant methods + kb-migration-construct.ts 4 Lambdas sharing ONE image + alarms +``` + +### Where authorization lives, and why not in `kb_backend` + +`kb_access` and `kb_publication` sit in `apis.shared.assistants` because they reuse +`resolve_assistant_permission` and `listing.is_on_shelf`, and `kb_backend` may not +import that package. Authorization is above the seam by nature anyway: the answer is +the same whichever engine serves the query, so implementing it once above both +adapters is the only way it cannot differ between them. + +The facade's `access` parameter is **required and keyword-only**. Forgetting it is a +`TypeError` at the call site; a genuine denial passes `None` and fails closed. A +`KbAccess` cannot be built with a permission outside the read set, so holding one is +evidence the permission model was consulted — holding a string is not. + +Reclaim exemption keys on `listing.is_on_shelf`, **never** `is_listed`: an admin +requesting changes on a live listing leaves it serving but moves its state out of +`LISTED_STATES`, so by state name alone a reclaim pass would delete the corpus behind +an agent users can still see in the store. + +### Score direction — the highest-silent-risk detail + +S3 Vectors returns cosine **distance** (lower better). Managed returns **relevance** +(higher better). The protocol canonicalizes on `relevance`; `s3vectors_backend` +converts by **exact negation** (order-preserving and losslessly reversible, unlike +`1-d`); the managed adapter applies **no** conversion. The facade still emits a +derived `distance` key so no caller changed. + +Invert it and nothing raises — retrieval keeps returning five chunks and the answers +quietly get worse. `tests/property/test_pbt_kb_score_direction.py` is the only guard. + +--- + +## 8. Data model + +``` +PK = AST#{assistant_id} +SK = METADATA # the assistant row (pre-existing) +SK = KB#{app_kb_id} # app_kb_id == assistant_id THIS PHASE +SK = KBTOMB#{app_kb_id} # whole-KB tombstone, NO TTL +SK = KBTOMB#{app_kb_id}#DOC#{document_id} # document tombstone, NO TTL + +GSI7 "KbWorkIndex" (projection ALL) — sparse + GSI7_PK = KBWORK#{state} GSI7_SK = {dueAt ISO-8601} +``` + +Keys are written **only** while a record is work-eligible and `REMOVE`d on reaching a +terminal state, so ineligible knowledge bases are invisible to the dispatcher **by +physics** rather than by filter. Third use of this convention on this table +(`DueSyncIndex`, `AgentDirectoryIndex`, `AgentReportsIndex`). + +**Absence means legacy.** `retrievalEngine` is only ever written as `"managed"`. +Nothing writes `"s3vectors"` onto a record that lacked it — that is what makes the +migration zero-backfill across 1,692 existing records and makes rollback a single +attribute `REMOVE` rather than a data rewrite. + +Same convention for two more attributes: + +- `dualReadPilot` — read as `is True`, never truthiness. Absence is off. +- `policyAwsKbId` — the `awsKbId` the resource policy was last applied to. + `policy_is_stale` compares it against the live one, so re-application after a + replacement identifier is a comparison nothing can bypass by omission. + +Source bytes already live at +`assistants/{assistant_id}/documents/{document_id}/{filename}`. Migration is a +**re-ingest**, never a re-upload. diff --git a/.kiro/specs/managed-kb-migration/design.md b/.kiro/specs/managed-kb-migration/design.md new file mode 100644 index 000000000..b6e72a35c --- /dev/null +++ b/.kiro/specs/managed-kb-migration/design.md @@ -0,0 +1,963 @@ +# Design Document: Managed Knowledge Base Migration + +## Overview + +This design replaces the custom RAG retrieval backend (Docling → Titan → +Amazon S3 Vectors) with **Amazon Bedrock Managed Knowledge Base**, one knowledge +base at a time, behind a single abstraction seam, with rollback available at every +step. + +The shape of the change is a **strangler fig**. There are exactly two retrieval +call sites today, both routed through +`search_assistant_knowledgebase_with_formatting`. That function becomes a thin +facade over a `KnowledgeBaseBackend` protocol with two implementations. A +per-knowledge-base discriminator selects which one runs. Nothing above the seam +learns which backend it received. + +Three properties are load-bearing and everything else follows from them: + +1. **Absence is the default.** A knowledge base with no `retrievalEngine` + attribute is a legacy knowledge base. No backfill write is ever required, so a + half-finished rollout cannot half-break the fleet. +2. **The expensive resource is created late and deleted through a tombstone.** + Provisioning is lazy and idempotent; deletion writes a durable marker before it + calls AWS. +3. **Promotion is a single conditional write, and legacy data survives it.** + That makes rollback a pointer flip rather than a data restoration. + +### What is deliberately not here + +Phases 5–8 of the evaluation's §14.7 (managed-by-default, stopping legacy writes, +reclaiming legacy vectors, removing the old pipeline), agentic retrieval, any +change to the 2,000-character context cap, and the 1:1 → 0..N binding change (F4). +See `requirements.md` § "Scope boundary" and § "Non-goals". + +--- + +## Guiding measured constraints + +Every number here is measured in the evaluation, not assumed. They are collected +in one place because they are the reason the design has the shape it does. + +| Constraint | Measured value | Design consequence | +|---|---|---| +| `CreateKnowledgeBase` → ACTIVE | 47–124 s (n=7, median ≈73 s) | Never on an interactive path; lazy provisioning with generous timeouts | +| Per-KB cold first ingest | ~68 s, remarkably constant (68.296/68.232/68.334 s) | A fixed cost of the *knowledge base*, not the document; pay it once, in background | +| Warm ingest, small text | ~2.5 s | Comparable to today; bulk migration is feasible | +| Warm ingest, 50 KiB PDF | 68–264 s | Long tail; ingestion timeouts ≥300 s, treated as background work | +| INDEXED → actually retrievable | 0.75–1.03 s | Two distinct timestamps; poll for retrievable, not indexed | +| `Retrieve` p50 / p95 | 662–695 ms / 762–800 ms | +405 ms p50, +538 ms p95 vs today; acceptable but real TTFT cost | +| `StartIngestionJob` | 0.1 RPS, account-wide, **not adjustable** | Direct ingestion only; never per-document sync jobs | +| `IngestKnowledgeBaseDocuments` | **10 documents max**, server-enforced | Batch at 10, not the 25 the user guide claims | +| Concurrent Ingest+Delete document ops | 10 per account | Fleet migration throughput ceiling ~2 docs/s | +| `Retrieve` query input | 10,000 chars, **not adjustable** | Hard clamp at the seam | +| `Retrieve` RPM per KB | 600 + 25 RPS burst | Safe; per-KB isolation is the main quota win | +| `AgenticRetrieveStream` RPM | **60 per account** | Agentic retrieval cannot be a default path — out of scope | +| Managed storage | $5.00/GB-month | 35× today; byte caps are mandatory, not optional | +| Retrieval | $0.001/query | 2.3% of a $0.044 turn | +| Empty/idle KB | $0.00000203 measured for the month | No per-KB floor; count pressure is near zero | +| KB deletion | 2–6 minutes, async | Poll `ListKnowledgeBases`; "accepted" ≠ "gone" | +| Filter operators | fail **closed** (measured 0 results) | A mistyped filter yields nothing rather than leaking — but see the isolation note below | +| Managed reranking | separates scores 0.89/0.38/0.25/0.21/0.19 vs flat 1.00/0.84/0.78/0.77/0.77 | The reranker is what makes a 2,000-char cap defensible | + +--- + +## Architecture + +### The seam + +``` + inference_api/chat/routes.py app_api/assistants/routes.py + │ │ + └──────────────┬───────────────┘ + ▼ + search_assistant_knowledgebase_with_formatting() + (facade — unchanged public signature) + │ + ┌──────────────────┴──────────────────┐ + │ resolve_backend(app_kb_id) │ + │ reads KB_Record.retrievalEngine │ + │ absent ⇒ "s3vectors" │ + └──────────────────┬──────────────────┘ + ▼ + KnowledgeBaseBackend (Protocol) + │ + ┌───────────────────────┴───────────────────────┐ + ▼ ▼ + S3VectorsBackend ManagedKbBackend + (today's code, moved verbatim, (bedrock-agent + agent-runtime, + distance → relevance conversion) managedSearchConfiguration) + │ │ + S3 Vectors index Managed KB +``` + +New Python package: `backend/src/apis/shared/kb_backend/` + +| Module | Responsibility | +|---|---| +| `protocol.py` | `KnowledgeBaseBackend` Protocol, `Chunk` dataclass | +| `resolver.py` | `retrievalEngine` → backend instance; absence defaults to legacy | +| `s3vectors_backend.py` | Legacy adapter; owns distance → relevance conversion | +| `managed_backend.py` | Managed adapter; owns `managedSearchConfiguration` | +| `query_guard.py` | 10,000-character clamp + truncation metric | +| `records.py` | KB_Record read/write, conditional transitions | +| `provisioning.py` | The provisioning saga | +| `byte_cap.py` | reserve / commit / release | +| `tombstones.py` | Durable delete markers | + +> **Why a top-level package under `shared/`, not under `shared/assistants/`.** +> `kb_sync/records.py` documents that importing `apis.shared.assistants` "drags in +> the embeddings stack", which is why the kb-sync Lambdas use raw table access +> instead. The migration and ingestion Lambdas have that same constraint. Nesting +> the seam inside `assistants/` would force them to trip the very import the +> existing code goes out of its way to avoid. A sibling package keeps +> `apis.shared.kb_backend` importable by both the APIs and the Lambdas. +> +> Two rules make that hold rather than merely intend it: +> 1. `kb_backend/__init__.py` stays **empty** — no re-exports. +> 2. Heavy dependencies (`boto3` clients, the embeddings module) are imported +> **inside functions**, matching the existing convention in `kb_sync/records.py`. +> +> An architecture test asserts that `apis.shared.kb_backend` does not transitively +> import `apis.shared.assistants`, alongside the existing boundary tests in +> `backend/tests/architecture/`. + +> `apis/shared/assistants/vector_search.py` currently exists as a zero-byte +> placeholder. It is unused and unimported; leave it alone rather than repurposing +> it, so the new package's boundaries are unambiguous. + +### Component inventory + +| Component | Type | New or changed | Notes | +|---|---|---|---| +| `kb_backend/` package | library | **new** | The seam | +| `search_assistant_knowledgebase_with_formatting` | function | changed | Becomes a facade; signature preserved | +| `_filter_vectors_by_document_status` | function | changed | Fail closed (Req 5) | +| Ingestion consumer | Lambda | **new** | Replaces orchestration role of the Docling Lambda for managed KBs | +| Existing Docling ingestion Lambda | Lambda | unchanged | Still authoritative for legacy KBs | +| Migration dispatcher | Lambda | **new** | Copies `kb-sync` dispatcher shape | +| Migration worker | Lambda | **new** | Shares one image with the dispatcher | +| Reconciler | Lambda | **new** | Daily, report-only initially | +| KB service role | IAM role | **new** | One role serves many KBs | +| KB_Record | DynamoDB items | **new** | In the existing assistants table | + +### Why reuse the `kb-sync` topology + +`infrastructure/lib/constructs/kb-sync/kb-sync-construct.ts` already implements +exactly the shape this feature needs, and `scheduled-runs-construct.ts` documents +itself as following it closely — so this is the third use of an established +in-repo pattern, not a new invention: + +- two Docker Lambdas sharing **one** image (`backend/Dockerfile.kb-sync`); +- the platform-as-bootstrap pattern — CDK ships a byte-stable stub from + `bootstrap-assets/`, the workflow ships the real image via + `update-function-code`; +- SSM parameters publishing the generated function names so the deploy script can + find them; +- an EventBridge `rate()` schedule into the dispatcher; +- a bounded per-tick dispatch limit (`KB_SYNC_DISPATCH_LIMIT`, default 20). + +The migration Lambdas must also follow `kb_sync/records.py`'s **raw table access** +convention. That file exists for a documented reason: importing +`apis.shared.assistants` drags in the whole embeddings stack, and keeping the +Lambda image small is a deliberate constraint. The migration worker has the same +constraint and takes the same approach. + +--- + +## Data model + +All new items live in the **existing** assistants table +(`boisestateai-v2-rag-assistants`), preserving the adjacency-list convention. + +### KB_Record + +For this phase `App_KB_Id == assistant_id`, so the record is a sibling of +`METADATA` under the assistant's partition. This is exactly the compatible +phase-1 option §14.2 proposes, and it is what `compat.py` already anticipates: +its docstring states that when F4 lands `ref` "becomes a real KB id with no shape +change here". + +``` +PK = AST#{assistant_id} +SK = KB#{app_kb_id} # app_kb_id == assistant_id in this phase +``` + +| Attribute | Type | Notes | +|---|---|---| +| `appKbId` | S | Stable identity. What bindings reference | +| `ownerUserId` | S | For byte accounting and cost attribution. Opaque id, never email/PII | +| `visibility` | S | Mirrors the assistant's visibility in this phase | +| `retrievalEngine` | S | `"managed"`. **Never written as `"s3vectors"`** | +| `provisioningState` | S | `provisioning` / `active` / `failed` / `deleting` | +| `awsKbId` | S | AWS `knowledgeBaseId`. Replaceable. Never in a binding | +| `awsDataSourceId` | S | The `CUSTOM` connector id | +| `embeddingModelId` | S | `amazon.titan-embed-text-v2:0`. **Immutable** | +| `embeddingDimensions` | N | 1024. **Immutable** | +| `parserConfig` | M | Managed-parser settings captured at creation, including `imageExtraction`. Recorded because §14.2 requires immutable choices be persisted, and because a corpus indexed without image extraction is not comparable to one indexed with it | +| `imageExtraction` | BOOL | Convenience mirror of `parserConfig.imageExtraction` for queries | +| `storedBytes` | N | Committed bytes, from S3 `HEAD` | +| `reservedBytes` | N | In-flight reservations | +| `lastRetrievedAt` | S | Throttled write, one winner per 24 h | +| `migrationState` | S | `shadow` / `verify` / `promote` / `retain` / `failed`, plus `reclaim` reserved but never entered in this phase | +| `migrationGeneration` | N | Increments per attempt; guards stale workers | +| `migrationLeaseUntil` | S | Worker lease expiry | +| `migrationProgress` | M | `{migrated, total, lastDocumentId}` | +| `migrationError` | S | Plain-language reason for the UI | +| `promotedAt` / `rolledBackAt` | S | Rollback observation window anchors | +| `retainUntil` | S | Earliest eligible reclaim time | +| `pinned` / `exemptFromReclaim` | BOOL | Lifecycle exemptions | +| `clientToken` | S | Persisted so a retry reuses it | + +### Tombstone + +``` +PK = AST#{assistant_id} +SK = KBTOMB#{app_kb_id} # whole-KB delete +SK = KBTOMB#{app_kb_id}#DOC#{document_id} # document delete +``` + +Carries `intent`, `awsKbId`, `awsDataSourceId`, `createdAt`, `attempts`, +`lastError`. **No TTL** — a tombstone is cleared by confirmed deletion or it stays +as a work item. Letting TTL remove it would recreate the exact silent-leak class +this design exists to close. + +### Sparse GSI for work discovery + +Migration work is discovered through a **sparse** GSI: the key attributes are +written *only while the record is eligible*, so ineligible and pinned knowledge +bases are invisible to the scan **by physics** rather than by filter. + +This is an established convention on this exact table, not a new idea. Three +existing indexes already work this way and say so in their own comments: +`DueSyncIndex` (GSI4, written only while a sync policy is `active`), +`AgentDirectoryIndex` (GSI5, written only while a listing is `published`), and +`AgentReportsIndex` (GSI6, written only while a report is `open`). + +The table currently has **six** GSIs, named `GSI_PK`/`GSI_SK` for the first and +`GSI2_PK`/`GSI2_SK` through `GSI6_PK`/`GSI6_SK` thereafter. The new index is +therefore **GSI7**: + +``` +GSI: KbWorkIndex (partition GSI7_PK, sort GSI7_SK, projection ALL) + GSI7_PK = KBWORK#{state} # e.g. KBWORK#shadow + GSI7_SK = {dueAt ISO-8601} +``` + +When a knowledge base reaches a terminal state, the worker **removes** `GSI7_PK` +and `GSI7_SK`. A bug that fails to remove them causes repeated no-op work bounded +by the per-tick dispatch limit, not a runaway. + +Following the `AgentDirectoryIndex` precedent, the generic assistant-update path +must list `GSI7_*` as immutable, so a routine edit can never resurrect a work key +on a knowledge base that has left the queue. + +--- + +## Backend protocol + +```python +# apis/shared/kb_backend/protocol.py +from dataclasses import dataclass +from typing import Any, Protocol + +@dataclass(frozen=True) +class Chunk: + text: str + relevance: float # canonical: HIGHER IS MORE RELEVANT + document_id: str + metadata: dict[str, Any] + key: str + +class KnowledgeBaseBackend(Protocol): + async def search(self, kb_ref: str, query: str, top_k: int) -> list[Chunk]: ... + async def ingest(self, kb_ref: str, document_id: str, source: "DocumentSource") -> None: ... + async def delete_document(self, kb_ref: str, document_id: str) -> None: ... +``` + +### Score direction — the silent-failure risk + +This is the single most dangerous detail in the migration, because getting it +wrong produces **no error, just worse answers**. + +- S3 Vectors returns cosine **distance**: lower is better. The current formatted + result dict literally has a `"distance"` key, and its docstring says + *"lower = more similar"*. +- Managed KB returns **relevance**: higher is better. The probe measured + `score: 1.0` on an exact hit. + +The protocol canonicalizes on **relevance**. `S3VectorsBackend` performs the +conversion in its adapter, and `ManagedKbBackend` passes through. The facade keeps +emitting a `distance` key for any existing consumer during the transition, derived +from relevance, so no caller breaks on the field rename. + +A test asserts that for the same ordered input both backends rank the known-best +chunk first (Req 2.4). Without it, an inversion is undetectable by any other test +in the suite. + +### Query guard + +```python +MAX_QUERY_CHARS = 10_000 # Managed KB Retrieve cap; NOT adjustable +``` + +Applied in the facade, before backend dispatch, so both backends are protected +identically. Truncation emits a metric and never raises. This replaces the +existing inline comment in `bedrock_embeddings.py` asserting that the query is a +"short string, no token validation needed" — which is true only because Titan v2 +tolerates ~32,000 characters. + +### Retrieval configuration + +`ManagedKbBackend` sends `managedSearchConfiguration`, never +`vectorSearchConfiguration` — the latter is rejected outright for managed +knowledge bases: + +```python +retrievalConfiguration = { + "managedSearchConfiguration": { + "numberOfResults": top_k, # 5, parity + "rerankingModelType": "MANAGED", # NOT "NONE" + # "filter": {...} equals/in only for isolation-critical filters + } +} +``` + +Hybrid search is not configurable for managed knowledge bases and is simply how +managed retrieval works; there is no toggle to set and none is attempted. + +### The document-status filter runs on both backends + +Requirement 3.3 keeps the `status == "complete"` post-filter on the managed path +too, even though managed ingestion makes it largely redundant — because removing it +in the same change that swaps the engine would confound the comparison. Parity +means parity, including the parts that look unnecessary. + +This works on the managed path only because `customDocumentIdentifier` is set to +the platform's `document_id` (Requirement 9.4). The filter needs a `document_id` +per returned chunk; the 1:1 identifier mapping is what supplies it. Without that +mapping there would be nothing to join on, which is a second reason the `CUSTOM` +connector beats pointing a native S3 connector at the prefix. + +The filter is applied in the facade, above the seam, so there is exactly one +implementation and it **fails closed** on both backends (Requirement 5). Its +removal from the managed path is a follow-up-spec decision, made only once managed +is the sole engine. + +--- + +## Authorization, isolation, and publication + +This section closes evaluation gate §14.3. It is the gate most easily mistaken for +already-solved, because Managed KB ships two features whose names suggest they do +more than they do. + +### Three isolation levels, correctly ranked + +| Level | Mechanism | What it actually guarantees | +|---|---|---| +| **Weakest** | Metadata filter (`equals`/`in`) | *Logical* separation only. AWS's own multi-tenant guidance calls this "filter-level (logical) isolation, **not** IAM-enforced (infrastructure) isolation" | +| **Middle** | ACL-aware retrieval | Fails closed, which is better than today's document-status filter — but AWS states plainly that it "is not authorization" and does not authenticate users. Identity is **email only, with no alias resolution, and mismatches fail silently** | +| **Strongest** | One knowledge base per boundary, plus a resource policy | Genuine IAM-enforced `bedrock:Retrieve` / `bedrock:GetDocumentContent` | + +**Design consequence: the app remains the authorization authority.** Neither +metadata filters nor ACL-aware retrieval may be the sole thing standing between one +user's documents and another's. Because this phase keeps `App_KB_Id == +assistant_id`, the per-assistant boundary *is* a per-knowledge-base boundary, which +is the strongest of the three by construction. Filters are used for sub-scoping +within a knowledge base, never as the tenant boundary. + +The email-only identity limitation is why ACL-aware retrieval is **not** adopted in +this phase: this platform authenticates via OIDC with claim mappings, and a +silently-failing email match is a worse primitive than an explicit app-side check. + +### Invocation-time access resolution + +The runtime resolves the invoking user's access to a knowledge base **before** +retrieval, reusing the existing assistant permission model rather than inventing a +parallel one: + +- **owner / editor** — may read, may upload, may trigger an upgrade. +- **viewer** — may read through the agent; never sees the upgrade control. +- **no access** — retrieval is not attempted. + +Because this phase is 1:1, an agent's knowledge base is exactly the agent's own, so +"can this user invoke this agent" already answers "may this user's turn retrieve +from this knowledge base". A turn is never failed because of a knowledge base the +user cannot reach — there is no such case while the relationship stays 1:1. That +changes with F4, which is precisely why F4 is a separate spec: the "one +inaccessible knowledge base among N blocks the whole turn?" question only becomes +real then, and it is recorded here as inherited-open rather than answered +prematurely. + +### Published agents and corpus drift + +A marketplace listing freezes a knowledge base **reference**, not its contents, so +a published agent's answers can change after review without any re-review. This +phase does not solve that, and must not pretend to. It takes the one position that +is safe and reversible: + +- Migration **does not change** what a published agent retrieves — parity is the + whole contract, so an engine swap is not a corpus change and needs no re-review. +- A published agent is **exempt from lifecycle reclaim while listed**, and + `taken_down` requires an explicit transition rather than falling through to + reclaim. +- Whether published agents should pin a corpus revision, require re-review after + content changes, or bind only publisher-managed knowledge bases is an **open + question owned by the marketplace spec**, recorded in "Open questions carried + forward". Exemption from cleanup alone does not close that review bypass, and this + design does not claim it does. + +### Resource policies + +Resource policies are MANAGED-only and are the only mechanism here offering real +infrastructure isolation. This phase creates them only where a knowledge base is +shared beyond its owner. Because they attach to the **AWS knowledge base ARN**, any +cycle producing a new `awsKbId` silently drops sharing — so re-application after +rehydration is a tested invariant, not a runbook note. + +--- + +## Dual-read pilot + +The pilot exists so the rollout rests on evidence from *our* corpus and *our* +users, not solely on a 3-document benchmark. + +```mermaid +sequenceDiagram + participant F as Facade + participant L as S3VectorsBackend + participant M as ManagedKbBackend + participant U as User + + F->>L: search(query) + F->>M: search(query) %% concurrent + L-->>F: chunks (authoritative) + M-->>F: chunks (observation only) + F->>F: log overlap, rank correlation, per-backend latency + F-->>U: LEGACY results +``` + +Rules that make it safe to leave on: + +- **Legacy is always what is served.** The managed result is observation only. +- **The managed call is fire-and-forget with respect to correctness.** A managed + failure or timeout is logged and discarded; it can never fail the turn. +- **It must not add user-visible latency.** The two calls are concurrent and the + response is returned as soon as legacy resolves, so the managed call's 662–695 ms + p50 is not additive. +- **Opt-in per knowledge base, default off**, so pilot cost is bounded and + deliberate. + +Recorded per read: overlap in returned `document_id` values, rank correlation, and +per-backend latency. That is the same measure-first pattern used for the +prompt-cache and document-offload work. + +--- + +## Provisioning saga + +Ordering exists to guarantee that a crash leaves a **retry anchor**, never an +invisible paying resource. + +```mermaid +sequenceDiagram + participant IC as Ingestion Consumer + participant DDB as Assistants Table + participant BA as bedrock-agent + + IC->>DDB: conditional PutItem KB_Record
provisioningState=provisioning
attribute_not_exists(SK) + alt another worker already won + DDB-->>IC: ConditionalCheckFailed + IC->>IC: poll existing record until active + else this worker owns provisioning + DDB-->>IC: ok (clientToken persisted) + IC->>BA: CreateKnowledgeBase(type=MANAGED,
managedKnowledgeBaseConfiguration=embedding pin,
clientToken) + Note over IC,BA: 47-124 s to ACTIVE.
"Unable to verify embedding model"
is IAM eventual consistency -> RETRY + BA-->>IC: knowledgeBaseId + IC->>BA: CreateDataSource(MANAGED_KNOWLEDGE_BASE_CONNECTOR
connectorParameters={type:CUSTOM}
dataDeletionPolicy=RETAIN
imageExtractionStatus=ENABLED) + BA-->>IC: dataSourceId + IC->>DDB: conditional update -> active
attach awsKbId, awsDataSourceId + end +``` + +Five details that are each a defect if omitted: + +1. **DDB before AWS.** A crash after `CreateKnowledgeBase` leaves a + `provisioning` record the Reconciler can match against the orphan, so the + resource is adoptable rather than stranded. +2. **`clientToken` is built, not interpolated.** Minimum length is **33 + characters**; the natural `{id}-{variant}-kb` token is 31 and fails client-side + validation. It is persisted on the record so a retry reuses the same token and + AWS deduplicates. +3. **`dataDeletionPolicy: RETAIN` at creation.** This is the documented remedy for + the `DELETE_UNSUCCESSFUL` state, and the dev account already contains a + knowledge base stuck in it since 2025-11-24. Set it deliberately up front, not + as incident response. +4. **`imageExtractionStatus: ENABLED`.** Opt-in. Left default, chart and image + content is never described and never indexed — a silent loss of the capability + being paid for. +5. **The embedding-model verification failure is retryable.** It was observed + against a model confirmed `ACTIVE` and directly invokable. Treated as fatal, lazy + provisioning fails intermittently while pointing at the wrong cause. + +--- + +## Ingestion control plane + +The browser creates an `uploading` `DOC#` row and receives a presigned S3 PUT. +**There is no upload-complete API call**, so the bucket's `ObjectCreated` +notification remains the only trigger. A durable consumer is therefore required — +not an in-process `asyncio.ensure_future` task. + +```mermaid +sequenceDiagram + participant S3 as Documents Bucket + participant IC as Ingestion Consumer + participant DDB as Assistants Table + participant Old as Docling Pipeline + participant BA as bedrock-agent + + S3->>IC: ObjectCreated + IC->>DDB: read DOC# + KB_Record + alt retrievalEngine absent (legacy) + IC->>Old: existing pipeline (unchanged) + else retrievalEngine == managed + IC->>DDB: reserve bytes (S3 HEAD size) + IC->>IC: provisioning saga if needed + IC->>BA: IngestKnowledgeBaseDocuments
(<=10 docs, customDocumentIdentifier=document_id) + loop until retrievable + IC->>BA: GetKnowledgeBaseDocuments + end + IC->>BA: canary Retrieve (indexed != retrievable) + IC->>DDB: DOC# -> complete, commit bytes + end +``` + +- **Routing is exclusive.** A document is indexed on exactly one backend outside a + deliberate migration or dual-read pilot, so no double-indexing. +- **Two timestamps, not one.** `indexedAt` and `retrievableAt` are recorded + separately; the gap measured 0.75–1.03 s and is a real, distinct event. +- **Timeouts ≥300 s.** A 50 KiB PDF has been observed at 264 s. +- **No chunk-key bookkeeping.** `customDocumentIdentifier = document_id` gives a + 1:1 mapping, which retires the whole `{doc_id}#{chunk_index}` scheme including + `delete_vector_tail` and the chunk-shrinkage stash on the managed path. + +--- + +## Migration state machine + +```mermaid +stateDiagram-v2 + [*] --> legacy: no retrievalEngine + legacy --> shadow: owner opts in + shadow --> verify: all complete docs ingested + verify --> shadow: catch-up found new docs + verify --> promote: manifest match + canary pass + converged + promote --> retain: conditional write succeeded + retain --> reclaim: OUT OF SCOPE (follow-up spec) + shadow --> failed: unrecoverable + verify --> failed: manifest mismatch + failed --> legacy: stays usable, retry offered + retain --> legacy: rollback (pointer flip) +``` + +`retain` is the terminal state this spec reaches. `reclaim` is present in the enum +so the follow-up spec adds a transition rather than a schema change, but nothing +here enters it. + +| Phase | Work | Serving | User sees | +|---|---|---|---| +| `shadow` | Provision KB, re-ingest every `complete` doc from existing S3 keys | **legacy** | "Upgrading — 12 of 40 documents", fully usable | +| `verify` | Exact source manifest compare + canary retrieve | **legacy** | same | +| `promote` | Single conditional write `retrievalEngine="managed"` | managed | one-time success note | +| `retain` | Legacy vectors preserved ≥30 days | managed | nothing | +| `reclaim` | **Out of scope — follow-up spec.** The state exists in the enum and the machine reaches `retain` and stops | managed | nothing | + +### Timing, recomputed from the revised measurements + +The evaluation's §10.3 quoted "a 20-doc assistant ≈ 4 min; 100 docs ≈ 9.5 min", +but those totals were computed from the **superseded** §5 figures (85 s create, +~65 s first ingest, ~5 s each thereafter). Recomputed from §5.1's revised numbers: + +| Corpus | Arithmetic | Total | +|---|---|---| +| 20 small text documents | 73 s + 68 s + 19 × 2.5 s | **~3 min** | +| 100 small text documents | 73 s + 68 s + 99 × 2.5 s | **~6.5 min** | +| 20 native layout PDFs (50 KiB class) | 73 s + 68 s + 19 × (68–264 s) | **~24–86 min** | +| 20 scanned PDFs (260 KiB class) | 73 s + 68 s + 19 × (37–58 s) | **~14–21 min** | + +The two PDF rows are kept separate because the measurements come from two different +document classes and averaging them would invent a number: the 50 KiB *native* +PDF measured 68–264 s, while the 260 KiB *scanned* PDF measured 37–58 s. The larger +file was consistently faster, so size is not the predictor — content structure is. + +⚠️ **The PDF rows are the ones to plan around, and they are absent from the +evaluation's own estimate.** Per-document parse time dominates everything else for a +PDF-heavy corpus — the same 50 KiB PDF took 68 s, 89 s, 99 s and 264 s across four +runs. Progress reporting must therefore be per-document rather than +time-estimated, because a credible ETA cannot be computed up front. + +Fleet ceiling is ~2 documents/second given the 10-concurrent-document-operation +account limit — roughly 85 minutes for 10,000 documents, and that is a floor, not a +forecast, for the same reason. + +### Verification is a manifest, not a count + +Document-count parity would pass while content silently diverged. `verify` +compares an exact manifest of `document_id` + content hash or generation, then +performs at least one canary retrieval proving expected content comes back from +the managed side. Count parity alone is explicitly insufficient. + +### Writes and deletes during migration + +Coexistence is **converge-on-quiet**, not dual-write, so exactly one write path +stays authoritative until promotion: + +1. Uploads keep flowing to legacy as today. +2. The worker snapshots the doc-id set and migrates it. +3. A catch-up pass picks up anything created since the snapshot. +4. Repeat until a pass finds nothing new — the same shape as the crawler's + consecutive-miss rule. Warm ingest is ~2.5 s, so convergence is fast. +5. Every document's `DOC#` record is re-read **immediately before** ingesting it + and skipped if it is gone or no longer `complete`. Without this re-read, a + document deleted mid-migration resurrects in the new knowledge base. +6. Promotion is conditional on a converged pass, so two workers cannot both + promote. + +--- + +## Reconciler + +Runs daily. Joins a paginated, tag-filtered `ListKnowledgeBases` against +KB_Records. + +| Case | Action | +|---|---| +| AWS only | Orphan. Delete **only if the AWS-reported `createdAt` is >24 h old** | +| Record only | Stale pointer. Mark `vectorState: missing`, re-create on next ingest. **Never delete the record** — the documents are still valid | +| Both | Refresh `storedBytes` for quota accounting | +| Tombstone present | Retry the delete; escalate `DELETE_UNSUCCESSFUL` as an operator state | + +**Age-gate on `createdAt`, not on discovery time.** A reconciler that was down for +a week would otherwise wake up and delete every in-flight create. + +**Ships in report-only mode.** It logs what it would have deleted and deletes +nothing. It runs that way for weeks before being armed — the inverted flag +convention the evaluation calls for. The arming flag treats an **empty string as +off**, because the repo has been bitten by empty workflow variables before. + +--- + +## Byte cap accounting + +Storage is 35× more expensive per gigabyte than today. The existing 1 GB-per-user +file precedent, applied here at 30,000 users, is a **$150,000/month** exposure. +This is the only part of the design that can cause real financial damage. + +``` +reserve(owner, bytes) → conditional update, fails if committed + reserved + bytes > cap +commit(owner, bytes) → reserved -= bytes; stored += bytes +release(owner, bytes) → reserved -= bytes (on ingestion failure) +``` + +- Size comes from an **S3 `HEAD`** on the stored object, never from a + client-reported value. +- Reserve is a **conditional** update, so two uploads racing the same remaining + allowance cannot both win. +- The default per-owner cap is **lower** than 1 GB and resolves by role tier. +- `RawDataSize` is **not** used for enforcement: it returned 0 datapoints for a + directly-ingested document over a 60-minute lookback, and the cause is + unconfirmed. It may be used for reporting only. +- Cost-allocation tags are delayed reporting, not enforcement. + +### Concrete defaults + +The evaluation requires "a lower role-tier default" without naming one. Proposed, +and flagged as **requiring product sign-off before implementation**: + +| Tier | Per-owner cap | Worst-case at 30,000 users | +|---|---|---| +| Standard user | **100 MB** | 3 TB → ~$15,000/mo | +| Elevated (opt-in, admin-granted) | **1 GB** | — | +| Per-knowledge-base ceiling | **500 MB** | bounds a single runaway corpus | +| The 1 GB precedent, for contrast | 1 GB for everyone | 30 TB → **~$150,000/mo** | + +100 MB is ~88× the measured average of 1.13 MB per active user, so it is generous +in practice while cutting worst-case exposure 10×. Expected spend at full adoption +on measured behaviour remains ~$169/month; the gap between $169 expected and +$15,000 permitted is exactly why the alarms below are not optional. + +### Enforcement points + +The cap is checked at **every** path that can add bytes to a managed knowledge +base, not just interactive upload: + +1. **Upload** — reserve before ingest, commit on success, release on failure. +2. **Migration re-ingest** — the migration worker reserves for the whole snapshot + before entering `shadow`, and **fails the migration up front** rather than + part-migrating a corpus that will not fit. A knowledge base that exceeds its + owner's cap is surfaced as a plain-language failure with the option to request an + elevated tier. +3. **Rehydration** (follow-up spec) — same reserve path. + +Migration is the easy one to miss and the worst one to miss: it is the single +largest byte-adding operation in the system, and it is the one that runs +unattended. + +### Account-level alarms + +Per-owner caps bound one user. They do not bound the fleet, so gate §14.6 also +requires account-wide guards: + +| Alarm | Threshold | Why | +|---|---|---| +| Total managed KB storage | configurable GB | The only thing standing between expected and permitted spend | +| Managed KB count | 80% of the 10,000 default quota | The quota is adjustable, but capacity requests take lead time | +| `AmazonBedrockAgentCore` Knowledge-Base usagetype daily cost | configurable USD | Catches a cost shape no per-owner cap anticipated | +| `KbOrphansFound` sustained non-zero | any | The delete saga is leaking | + +Alarms use `TreatMissingData.NOT_BREACHING`, matching the posture of the existing +kb-sync, scheduled-runs, and prompt-cache observability constructs. + +### Who consumes retrieval quota + +Requirement 12.10 asks whether the knowledge base **owner** or the **invoking +user** consumes retrieval quota. Half the answer is a fact about AWS rather than a +choice, and it inverts the current model: + +| | Today (S3 Vectors) | Managed KB | +|---|---|---| +| `Retrieve` throughput | 20 rps **account-wide** | 600/min + 25 rps burst, **per knowledge base** | + +So the quota is consumed **per knowledge base**, which means the *owner's* knowledge +base absorbs the throughput of everyone who invokes their agent. The invoking user +does not carry a retrieval allowance of their own. + +**Decision: the owner is the payer, and this is an improvement, not a compromise.** +Today a single hot assistant can exhaust a 20 rps account-wide ceiling and degrade +retrieval for every other user on the platform. Per-KB quotas make that blast radius +one agent instead of the fleet — noisy-neighbour containment we do not currently +have. + +Consequences worth stating, because they follow from the decision rather than from +the implementation: + +* A **published** agent is the case to watch. Its knowledge base is one partition + serving an unbounded audience, so it is the only realistic way to approach + 600/min. Measured headroom is comfortable — ~26 requests/min average on a hot + shared knowledge base against an allowance of 600, about 4% — but the ceiling is + now per-agent and therefore reachable by a single popular agent in a way the + account-wide limit never made obvious. +* Retrieval is billed at **$0.001/query** and that cost attaches to the account, not + to a tenant. Attributing it per invoking user is a cost-reporting question, not a + quota question, and is out of scope here. +* Byte caps are per **owner**, consistent with this: the owner controls the corpus, + so the owner carries both its storage cost and its throughput ceiling. + +No code enforces a per-user retrieval allowance, deliberately. Adding one would +invent a limit AWS does not impose and that the existing per-agent permission model +already bounds. + +--- + +## IAM and encryption + +One Bedrock service role serves many knowledge bases — verified: a second +knowledge base created against the first one's role reached ACTIVE normally. +10,000 knowledge bases do not require 10,000 roles. + +| Control | Shape | +|---|---| +| Confused-deputy guard | `aws:SourceAccount` + `ArnLike` on `AWS:SourceArn` scoped to `knowledge-base/*` | +| PassRole | Caller's `iam:PassRole` conditioned on `iam:PassedToService` | +| S3 | Conditioned on `aws:ResourceAccount` | +| KMS | `serverSideEncryptionConfiguration.kmsKeyArn` where customer-managed keys are required | +| Separation | Provisioner/migrator CRUD, direct-ingestion, and inference `bedrock:Retrieve` scoped independently | +| Metrics (write) | `cloudwatch:PutMetricData` scoped to the non-reserved `{projectPrefix}/ManagedKb` namespace on the **calling identities only** — not the service role, which Bedrock assumes and which never publishes our metrics | +| Metrics (read) | `cloudwatch:GetMetricData` / `GetMetricStatistics` for Bedrock's own `AWS/Bedrock/KnowledgeBases` metrics | +| Async safety | Synchronous boto3 calls from async request paths run off the event loop | + +Three notes worth encoding rather than rediscovering: + +- **Metric publishing is best-effort and permission-gated.** Omit the + `PutMetricData` grant and metrics silently vanish while requests keep + succeeding. CDK assertions cover it. +- **The publish namespace must not begin with `AWS`.** CloudWatch reserves those + for its own services — "You cannot specify a namespace that begins with AWS" — + so `PutMetricData` scoped to `AWS/Bedrock/KnowledgeBases` authorizes nothing + that can ever succeed: a grant that reads as correct and silently does nothing. + Our own metrics (the table under Observability below) go to + `{projectPrefix}/ManagedKb`; the prefix keeps two environments in one account + from blending. Bedrock's `AWS/Bedrock/KnowledgeBases` metrics remain a **read** + source via `GetMetricData` / `GetMetricStatistics` — reading a reserved + namespace is fine, only writing is not. Do not "simplify" the two back into one + namespace. +- **Managed embedding and managed reranking need no Bedrock model access at all.** + Only `CUSTOM` does — and this design pins `CUSTOM` Titan v2 embeddings for + continuity across an immutable choice, so the grant is required. + +### Resource policies and rehydration + +Resource policies are MANAGED-only and give genuine IAM-enforced sharing for +`bedrock:Retrieve` and `bedrock:GetDocumentContent`. They attach to the **AWS +knowledge base ARN**, so any cycle producing a new `awsKbId` silently drops +sharing. Re-application after rehydration is a tested invariant (Req 24.12), not a +runbook step. + +### Teardown + +Managed knowledge bases are runtime-created and are **not** CloudFormation +children. `scripts/teardown/destroy.sh` must list and delete only resources tagged +for the project and environment, **before** deleting their service role and the +platform stack. Ordering is not cosmetic: deleting the role while a knowledge base +is still `DELETING` is a plausible route into `DELETE_UNSUCCESSFUL`, and a role +cannot be deleted until its inline policies are removed. + +--- + +## Observability + +EMF metrics alongside the existing PromptCache metrics. All of the following are +**our own** metrics and publish to `{projectPrefix}/ManagedKb` — never to +`AWS/Bedrock/KnowledgeBases`, which is reserved and rejects writes: + +| Metric | Why | +|---|---| +| `KbCount`, `KbStorageGB` | Leading indicators for the adjustable 10,000 cap and the storage curve; feed the alarms above | +| `KbIdleGB` | **Emitted for baseline only in this phase.** Nothing reclaims yet, but the follow-up spec needs historical idleness data to choose its eviction threshold, and that data cannot be backfilled | +| `KbOrphansFound` | **Sustained non-zero is the only signal the delete saga is leaking** | +| `KbQueryClamped` | Req 4 truncation rate | +| `KbStatusFilterFailClosed` | Req 5 — distinguishes a confirmed-empty result from an unconfirmable one | +| `KbMigration{Started,Promoted,Failed,RolledBack}` | Rollout health | +| `KbDualRead{Overlap,RankCorrelation,Latency}` | The pilot's whole output: how much the two engines agree, and what the managed one costs. Values rather than counts, and `KbDualReadLatency` is dimensioned per backend so the 662–695 ms vs 257 ms gap is measured on our own traffic rather than assumed from the benchmark | +| `KbDualReadFailed` | Managed-side failures during the pilot. Never user-facing — the turn was served from legacy before the comparison ran — but sustained non-zero says the engine is not ready | +| `KbByteCapRejected` | Whether the proposed 100 MB default is actually workable, before it hardens into policy | + +`KbReclaimedGBPerDay` is deliberately **not** emitted: nothing reclaims in this +phase, and a metric that is structurally always zero trains operators to ignore it. +It arrives with the reclaim tier. + +**Idleness** is `max(own lastRetrievedAt, max(lastUsedAt) over bound agents)` — +never retrieval alone, or an actively used agent's knowledge base is evicted +because its queries did not match. `lastRetrievedAt` uses the throttled +conditional write pattern (one winner per 24 h), never a write per retrieval. +Per-knowledge-base `Invocations` from `AWS/Bedrock/KnowledgeBases` is a cheaper +idleness signal and is preferred where it is sufficient. That is a **read** of +Bedrock's own namespace via `cloudwatch:GetMetricData` (Req 20.13), not a publish. + +**Cost attribution filters on `usagetype`.** Managed KB bills under +`AmazonBedrockAgentCore`, so anything keyed on `AmazonBedrock` misses it entirely, +and anything keyed on service code alone blends it into the AgentCore Runtime +memory line that is already 73% of that bill. + +--- + +## Flags and deployment choreography + +**Three** independent flags, all defaulting to off, all treating an empty string as +off: + +| Flag | Controls | This spec | +|---|---|---| +| `MANAGED_KB_NEW_DEFAULT` | new knowledge bases are created managed | ships **off** (phase 5) | +| `MANAGED_KB_MIGRATION_ENABLED` | the background migrator runs at all | ships **off**, enabled per-pilot | +| `MANAGED_KB_RECONCILER_ARMED` | the reconciler deletes, rather than only reporting | ships **off** — report-only for weeks first | + +The third is the inverted-convention flag: the reconciler is *deployed* from day +one but *disarmed*, so its judgement can be reviewed against real data before it is +allowed to delete anything. + +Deployment order is fixed by a hard rule: **backend code must never deploy before +the IAM and resources it requires.** + +1. **Platform** — additive schema, sparse GSI, service role, IAM, Lambda shells, + SSM parameters, teardown support. No behaviour change. +2. **Backend** — seam, both adapters, fail-closed filter, query clamp. All three + flags off, so managed code is dark. +3. **Pilot** — opt-in dual read on selected knowledge bases, still serving legacy. +4. **Opt-in migration** — owner-initiated, with the rollback observation window. + +Steps 5–8 of §14.7 are a follow-up spec. Because all three flags default off, +reaching +them is a configuration change rather than a code change. + +--- + +## UX surfaces + +| State | Surface | +|---|---| +| legacy, no action needed | **nothing** — no badge, no nag. A knowledge base that works needs no UI | +| upgrade available | Inline opt-in card: only benefits the §13 benchmark proved, expected duration, and "your knowledge base keeps working during the upgrade" | +| `shadow` / `verify` | Non-blocking progress ("Upgrading — 12 of 40 documents"); safe to navigate away | +| `promote` succeeded | One-time dismissible note. No permanent badge | +| failed | Plain-language reason + Retry. Stays on legacy, which keeps working. Never a dead end | + +Gated on the existing `_require_edit_permission`; viewers never see the control. +The word "vector" never appears in user-facing copy. No silent auto-migration in +this phase. + +**Admin surface:** knowledge bases filterable by engine, with stored bytes and +document counts, bulk migrate, and per-knowledge-base retry. + +### Surfacing failed and stuck documents + +Migration carries only `complete` documents. Measured against production, 200 of +1,692 `DOC#` records (11.8%) are not `complete`: 101 stuck `deleting`, 95 +`failed`, 4 `uploading`. Silently dropping the 95 failures is correct for the +index and wrong for the user — those people believe their uploads worked. The +upgrade flow surfaces them and offers retry. + +Two related messaging defects are in scope only to the extent of Req 21.4 +(distinguishing an unsupported format from a processing failure). The underlying +`.txt` ingestion bug — the deployed Docling build has no plain-text input format +despite the repo and frontend both advertising support, so a user waits 56 s for a +generic failure — is a **separate pre-existing bug**, not fixed here. + +--- + +## Testing strategy + +Mirrors the house pattern in `reliable-document-deletion`: unit tests plus +`hypothesis` property tests for invariants, with AWS stubbed. + +| Area | Approach | +|---|---| +| Adapter parity + score direction | Same input through both backends; assert identical ranking of a known-best chunk | +| Query clamp | Property: for any query length, output ≤10,000 chars and never raises | +| Fail-closed filter | Simulate table-level failure and missing table name; assert zero chunks | +| Byte cap races | Property: concurrent reserves never let committed total exceed the cap | +| Provisioning idempotency | Two concurrent first-ingests create exactly one KB | +| Crash after AWS create | Record left as a retry anchor; Reconciler adopts rather than duplicating | +| Reconciliation | Record-only and AWS-only cases; age-gate honours AWS `createdAt` | +| Migration interference | Upload and delete during migration; deleted doc never resurrects | +| Mixed deployment | Old and new code serving simultaneously; absent discriminator still resolves legacy | +| Resource policy rehydration | New `awsKbId` re-applies the policy | +| CDK assertions | IAM conditions from Req 20, including `PutMetricData` | +| Teardown | Only tagged resources deleted, and before the role | + +Managed AWS APIs are **stubbed**, never called live, so the suite stays +hermetic and free. + +--- + +## Open questions carried forward + +These remain genuinely open and are recorded so they are not mistaken for +settled: + +1. **Does a knowledge base go cold again after idleness?** In progress in the + evaluation. If a cold penalty exists, owners must be warned before eviction, + because rehydration pays the ~68 s cold-ingest cost. Affects the follow-up + spec's reclaim tier more than this one. +2. **Is there an account-level ingestion-concurrency limit?** The quota page lists + none. Probe with a many-knowledge-base backfill during the pilot before sizing a + wide migration. +3. **Does `RawDataSize` ever publish for directly-ingested documents?** Unconfirmed. + Until it does, byte accounting uses S3 `HEAD` (already the design). +4. **Native Google Drive connector vs the current AgentCore-Identity adapter.** + Never investigated. May sidestep the vault principal-binding dead-end at the + cost of moving token custody into Secrets Manager. +5. **`bedrock:GetDocumentContent` shape, size limits, and cost.** Relevant to + whole-document tasks that chunk retrieval structurally cannot serve. Unverified. diff --git a/.kiro/specs/managed-kb-migration/requirements.md b/.kiro/specs/managed-kb-migration/requirements.md new file mode 100644 index 000000000..a543fd7ab --- /dev/null +++ b/.kiro/specs/managed-kb-migration/requirements.md @@ -0,0 +1,858 @@ +# Requirements Document + +## Introduction + +This document specifies the requirements for replacing the platform's custom RAG +pipeline (Docling parse → Titan embed → Amazon S3 Vectors) with **Amazon Bedrock +Managed Knowledge Base** as the retrieval backend for assistant knowledge bases. + +The decision to proceed is grounded in `docs/specs/bedrock-managed-kb-evaluation.md`, +whose §13.4 decision gate was **cleared on 2026-08-14**: on a 9-question benchmark +with every variable held constant, the current pipeline answered 4/9 and managed +answered 9/9. Two document classes moved from unusable to working — native +layout-heavy PDFs (1/3 → 3/3) and scanned/OCR PDFs (0/3 → 3/3). + +The governing principle is **parity first, improvements later**. The user must +perceive nothing from the plumbing swap except the parser quality gain. Every +deliberate quality change that the evaluation identified as available — agentic +retrieval, raising the context cap, 0..N agent-to-KB bindings — is explicitly out +of scope here so that its effect remains attributable to itself. + +Migration is **additive and reversible at every step**. Legacy resources remain in +place for dual reads, rollback, and retention; no legacy resource is removed by +this spec. + +### Scope boundary + +This spec covers phases 1–4 of the evaluation's §14.7 choreography: + +1. additive schema, service role, IAM, worker resources, cleanup support; +2. dual backends dark, with mixed-version compatibility; +3. opted-in dual-read pilot, serving legacy; +4. opt-in migration with a rollback observation window. + +Phases 5–8 (managed-by-default for new KBs, stopping legacy writes, reclaiming +legacy vectors, and final target-state cleanup) are **deliberately deferred to a +follow-up spec**. The flags in Requirement 19 exist so that those phases are +config changes rather than code changes. + +### Non-goals + +The following are explicitly **not** in scope, each for a stated reason: + +- **Agentic retrieval.** Gated on the `AgenticRetrieveStream` account quota of + 60 requests/minute being raised (evaluation §6.4, §13.5 requirement 2). The + user-triggered escalation design in §6.5 is a separate future feature. +- **Raising the 2,000-character context cap.** The §13.6 experiment measured no + correctness change from 2,000 to 20,000 characters on either backend. Holding it + constant is required to keep the swap attributable (§9, §13.5 requirement 3). +- **0..N agent-to-KB bindings (F4).** §10.6 requires that the engine swap and the + binding-cardinality change not be coupled, because a joint failure is + unattributable. This spec lands the `KnowledgeBase` entity record while + preserving 1:1 binding semantics. +- **Routing conversation attachments through Managed KB.** §6.3 rejects this on + four grounds, including that a chat attachment ingested into a shared agent KB + becomes retrievable by every other user of that agent. Attachments remain + session-scoped inline blocks. +- **Native Google Drive connector evaluation.** §11 question 4, never + investigated; remains open. +- **Cleanup of the 101 stuck `deleting` and 95 `failed` legacy documents as a + standalone production migration.** Requirement 21 folds this into the migration + path instead. + +## Glossary + +- **Managed_KB**: An Amazon Bedrock Knowledge Base created with + `type: "MANAGED"`, which provisions no customer-visible vector store. Distinct + SKU from the classic `VECTOR` knowledge base, GA 2026-06-17. +- **Legacy_Backend**: The existing retrieval implementation over Amazon S3 + Vectors, as it exists today in `apis/shared/assistants/rag_service.py` and + `apis/shared/embeddings/bedrock_embeddings.py`. +- **Managed_Backend**: The new retrieval implementation over a Managed_KB. +- **KB_Backend_Protocol**: The Python `Protocol` defining `search`, `ingest`, and + `delete_document`, which both backends satisfy and behind which all callers sit. +- **Retrieval_Engine**: The per-knowledge-base discriminator selecting a backend. + Values are `"s3vectors"` and `"managed"`; **absence means `"s3vectors"`**. +- **KB_Record**: The new DynamoDB entity representing a knowledge base as a + first-class object, keyed by App_KB_Id. +- **App_KB_Id**: The stable, application-owned knowledge base identifier that + agent bindings reference. Never the AWS `knowledgeBaseId`. +- **AWS_KB_Id**: The AWS-assigned `knowledgeBaseId`, which is replaceable across a + dormancy/rehydration cycle and therefore never referenced by a binding. +- **Custom_Connector**: A Managed_KB data source of connector type `CUSTOM`, + nested inside the `MANAGED_KNOWLEDGE_BASE_CONNECTOR` envelope, which accepts + direct document ingestion. +- **Direct_Ingestion**: `IngestKnowledgeBaseDocuments`, which writes documents + into a Custom_Connector without a sync job, bypassing the + `StartIngestionJob` quota. +- **Ingestion_Consumer**: The durable S3 `ObjectCreated` event consumer that + replaces the current Docling ingestion Lambda's orchestration role. +- **Migration_Worker**: The background worker that moves one knowledge base from + Legacy_Backend to Managed_Backend through the Migration_State machine. +- **Migration_State**: The per-knowledge-base lifecycle + `shadow → verify → promote → retain`, plus a terminal `failed` state that returns + the knowledge base to Legacy_Backend, and a `reclaim` state reserved in the enum + but never entered in this phase. +- **Reconciler**: The daily job that joins `ListKnowledgeBases` against KB_Records + to detect orphaned AWS resources and stale pointers. +- **Doc_Status_Filter**: The query-time filter in + `rag_service._filter_vectors_by_document_status` that drops chunks whose parent + document is not `status == "complete"`. +- **Byte_Cap**: The enforced per-owner and per-knowledge-base limit on stored + source bytes. +- **Tombstone**: A durable DynamoDB marker written before an AWS delete call and + cleared only after AWS confirms deletion, so that a crashed delete is a + retryable work item rather than a silent leak. +- **Parity_Contract**: The set of retrieval properties held identical across both + backends so that the swap is perceptually invisible (Requirement 3). +- **Assistants_Table**: The existing DynamoDB table storing assistant and + document records (`PK=AST#{assistant_id}`, `SK=DOC#{document_id}`). + +## Requirements + +### Requirement 1: Backend Abstraction Seam + +**User Story:** As a developer, I want exactly one seam through which all +knowledge base retrieval and ingestion flows, so that the backend can be swapped +per knowledge base without any caller knowing which implementation it received. + +#### Acceptance Criteria + +1. THE system SHALL define a KB_Backend_Protocol in + `backend/src/apis/shared/kb_backend/` exposing `search`, `ingest`, and + `delete_document` operations. + +> Placement note: a top-level package under `shared/`, **not** under +> `shared/assistants/`. `apis/shared/assistants/__init__.py` imports +> `rag_service`, which imports `apis.shared.embeddings.bedrock_embeddings` at +> module level — so importing the assistants package drags in the embeddings stack. +> `kb_sync/records.py` uses raw table access specifically to avoid that, and the +> new Lambdas have the same constraint. Requirement 24.15 enforces the boundary by +> test. +2. THE system SHALL provide two implementations of KB_Backend_Protocol: + Legacy_Backend and Managed_Backend. +3. THE Legacy_Backend SHALL preserve the existing S3 Vectors behaviour, moved + without functional change. +4. WHEN a caller resolves a backend, THE system SHALL select it solely from the + knowledge base's Retrieval_Engine value. +5. THE two existing retrieval call sites (`inference_api/chat/routes.py` and + `app_api/assistants/routes.py`, both via + `search_assistant_knowledgebase_with_formatting`) SHALL be the only callers, + and SHALL NOT branch on backend identity. +6. WHEN a KB_Record has no Retrieval_Engine attribute, THE system SHALL resolve + the backend to Legacy_Backend. +7. THE system SHALL NOT write the value `"s3vectors"` to any record that does not + already carry it, so that backwards compatibility is achieved by absence and + requires zero backfill writes. + +### Requirement 2: Score Direction Canonicalization + +**User Story:** As a user, I want retrieved chunks ranked correctly regardless of +backend, so that answer quality does not silently invert when my knowledge base is +migrated. + +#### Acceptance Criteria + +1. THE KB_Backend_Protocol SHALL define chunk scores as **relevance**, where a + higher value is more relevant. +2. WHEN the Legacy_Backend returns S3 Vectors cosine **distance** values, THE + Legacy_Backend SHALL convert them to relevance before returning them across + the seam. +3. THE Managed_Backend SHALL pass Managed_KB relevance scores through unchanged. +4. THE system SHALL include a test asserting that, for the same ordered input, both + backends rank a known-best chunk first. + +### Requirement 3: Parity Contract + +**User Story:** As a user, I want a migrated knowledge base to behave exactly as it +did before except for parser quality, so that I cannot attribute any regression to +the upgrade. + +#### Acceptance Criteria + +1. THE system SHALL request `top_k = 5` on both backends. +2. THE system SHALL apply a context cap of **2,000 characters** on both backends, + unchanged from today's `max_context_length` default. +3. THE system SHALL retain the Doc_Status_Filter on **both** backends during + parity, even though Managed_Backend makes it redundant. +4. THE system SHALL build citations from the same `context_chunks` structure on + both backends, with the excerpt clip held at 500 characters. +5. THE system SHALL NOT enable agentic retrieval on any path. +6. THE system SHALL NOT alter the answer model, system prompt, or `top_k` as part + of this change. + +### Requirement 4: Query Length Clamp + +**User Story:** As a user, I want a long pasted message to still search my +knowledge base, so that I do not receive a hard failure for asking a long question. + +#### Acceptance Criteria + +1. WHEN a retrieval query is issued, THE system SHALL clamp the query string to at + most **10,000 characters** before it reaches the backend. +2. THE clamp SHALL be applied at the KB_Backend_Protocol seam so that it protects + both backends identically. +3. WHEN a query is clamped, THE system SHALL emit a metric or log record + identifying that truncation occurred. +4. THE clamp SHALL NOT raise an error or fail the turn. +5. THE system SHALL remove the inline assertion in + `apis/shared/embeddings/bedrock_embeddings.py` that no token validation is + needed for the query string. + +> Rationale: Managed_KB caps `Retrieve` query input at 10,000 characters and the +> quota is **not adjustable** (evaluation §6.4). Titan v2's ~32,000-character +> tolerance is the only reason nothing fails today. This is the single finding in +> the evaluation that produces a hard API failure rather than a cost or quality +> effect (§13.5 requirement 4). + +### Requirement 5: Fail-Closed Document Status Filter + +**User Story:** As a user who deleted a document, I want that document's content to +never be retrievable, so that a database problem cannot expose content I removed. + +#### Acceptance Criteria + +1. WHEN the Doc_Status_Filter cannot confirm a document's status because of a + table-level lookup failure, THE system SHALL drop that document's chunks. +2. WHEN the Doc_Status_Filter cannot confirm a document's status because the + documents table name is not configured, THE system SHALL drop all chunks. +3. WHEN the Doc_Status_Filter drops chunks because status could not be confirmed, + THE system SHALL emit a distinct error-level signal separating this case from an + ordinary empty-result case. +4. THE per-document lookup path SHALL continue to fail closed, as it does today. +5. **This requirement supersedes Requirement 3.4 of the + `reliable-document-deletion` spec**, which specified that a DynamoDB error + SHALL fall back to returning unfiltered results. +6. THE change SHALL ship as part of this feature's deployment, not as a standalone + production change. + +> Rationale: evaluation §7.4 documents this as a live fail-open path. §14.4 +> requires the filter fail closed before migration. The prior behaviour was a +> deliberate availability-over-privacy choice; retiring it is therefore a +> supersession and must be recorded as one. + +### Requirement 6: Knowledge Base as a First-Class Entity + +**User Story:** As a developer, I want a knowledge base to be its own record with a +stable identifier, so that the AWS resource behind it can be replaced without +breaking any agent binding. + +#### Acceptance Criteria + +1. THE system SHALL introduce a KB_Record persisted in DynamoDB. +2. THE KB_Record SHALL carry at minimum: App_KB_Id; owner identity; visibility or + ACL state; Retrieval_Engine; provisioning/lifecycle state; AWS_KB_Id; + data-source id; embedding and parser configuration including immutable choices; + stored-byte accounting; `lastRetrievedAt`; Migration_State with generation, + progress, lease, error and rollback timestamps; and pin/retention/exemption + flags. +3. Agent bindings SHALL reference App_KB_Id only. +4. THE system SHALL NOT persist AWS_KB_Id in any binding. +5. FOR this phase, THE system SHALL set `App_KB_Id == assistant_id`, preserving + the existing 1:1 relationship. +6. WHEN no KB_Record exists for an assistant, THE system SHALL treat it as a + virtual legacy S3 Vectors knowledge base and SHALL NOT create a record as a + side effect of a read. +7. THE system SHALL NOT change the cardinality of the agent-to-knowledge-base + relationship, and the existing rejections in `bindable_catalog.py` and + `binding_validation.py` SHALL remain in force. +8. THE test suite SHALL assert that an explicit `knowledge_base` binding is still + rejected and that `bindable_catalog` still returns an empty list for it, so the + 1:1 freeze is enforced by test rather than by intention. + +### Requirement 7: Lazy Provisioning Saga + +**User Story:** As a system operator, I want a knowledge base created in AWS only +when it is first needed and never duplicated, so that we do not pay for empty +resources or strand orphans. + +#### Acceptance Criteria + +1. THE system SHALL NOT call `CreateKnowledgeBase` when an assistant or knowledge + base is created. +2. WHEN the first document for a knowledge base is successfully ready to ingest, + THE system SHALL provision the Managed_KB. +3. THE system SHALL write the KB_Record in a `provisioning` state **before** + calling AWS, and SHALL attach returned identifiers with a conditional write. +4. WHEN two ingestions race to provision the same knowledge base, THE system SHALL + create at most one Managed_KB. +5. THE system SHALL pass a `clientToken` that satisfies the API's **33-character + minimum**, 256-character maximum, and + `[a-zA-Z0-9](-*[a-zA-Z0-9]){0,256}` pattern. +6. THE system SHALL construct the `clientToken` programmatically rather than by + interpolating a template that may fall below the minimum length. +7. WHEN `CreateKnowledgeBase` fails with a message indicating the embedding model + could not be verified, THE system SHALL treat the failure as retryable. +8. WHEN provisioning is interrupted after the AWS call but before the conditional + write, THE KB_Record SHALL remain a durable retry anchor discoverable by the + Reconciler. + +> Rationale: §5.1 measured `CreateKnowledgeBase` → ACTIVE at 47–124 s (n=7), so +> this must never sit on an interactive path. The "embedding model could not be +> verified" failure was observed to be pure IAM eventual consistency against a +> model confirmed ACTIVE and invokable. + +### Requirement 8: Managed Knowledge Base Configuration + +**User Story:** As a system operator, I want each Managed_KB created with the exact +configuration the evaluation validated, so that we do not silently lose a +capability we are paying for. + +#### Acceptance Criteria + +1. THE system SHALL call `CreateKnowledgeBase` with `type: "MANAGED"`, a + `roleArn`, and `managedKnowledgeBaseConfiguration`. + +> Shape note, verified against the packaged botocore service model: +> `managedKnowledgeBaseConfiguration` has **no required members**, but its only +> members are `embeddingModelType`, `embeddingModelArn`, +> `embeddingModelConfiguration` and `serverSideEncryptionConfiguration`. So the +> embedding pin required by criterion 5 below has nowhere else to live, and sending +> a literal `{}` would make that criterion unsatisfiable. "No required members" is +> not the same as "must be empty" — earlier drafts of this spec said `{}`, which is +> why this note exists. +2. THE system SHALL omit `storageConfiguration` entirely. +3. THE system SHALL create its data source with + `dataSourceConfiguration.type = "MANAGED_KNOWLEDGE_BASE_CONNECTOR"` and the + real connector type in + `managedKnowledgeBaseConnectorConfiguration.connectorParameters`. +4. THE system SHALL use connector type `CUSTOM`. +5. THE system SHALL set `embeddingModelType: CUSTOM` pinned to + `amazon.titan-embed-text-v2:0` at `FLOAT32` (the service-model enum value; lowercase is rejected) and 1024 dimensions. +6. THE system SHALL enable + `mediaExtractionConfiguration.imageExtractionConfiguration.imageExtractionStatus + = ENABLED` on the data source. +7. THE system SHALL set the data source's `dataDeletionPolicy` to `RETAIN` at + creation time. +8. THE system SHALL treat embedding configuration as immutable after creation and + SHALL NOT attempt to change it. +9. A single Bedrock service role SHALL be reusable across many Managed_KBs. + +> Rationale: §11.1 — image extraction is opt-in and silently indexes nothing if +> left default; custom Titan v2 embeddings measured identical cold-ingest time and +> identical 9/9 quality, and preserve continuity with today's embedding across an +> immutable choice; `dataDeletionPolicy: RETAIN` is the documented remedy for the +> `DELETE_UNSUCCESSFUL` state already observed in the dev account. + +### Requirement 9: Direct Document Ingestion + +**User Story:** As a user uploading documents, I want ingestion to keep up with +bulk uploads, so that a large batch is not serialized behind an API quota. + +#### Acceptance Criteria + +1. THE system SHALL ingest documents using Direct_Ingestion into the + Custom_Connector. +2. THE system SHALL NOT use `StartIngestionJob` for per-document ingestion. +3. THE system SHALL send at most **10 documents** per + `IngestKnowledgeBaseDocuments` call. +4. THE system SHALL set `customDocumentIdentifier` to the platform's + `document_id`. +5. THE system SHALL treat concurrent `Ingest` and `Delete` document operations as + limited to **10 per account** and SHALL bound its own concurrency accordingly. +6. THE system SHALL NOT carry forward the `{doc_id}#{chunk_index}` vector-key + bookkeeping, including `delete_vector_tail` and the chunk-shrinkage stash, on + the Managed_Backend path. + +> Rationale: `StartIngestionJob` is 0.1 RPS account-wide and not adjustable +> (§9). The API reference caps the document array at 10; AWS's user guide claim of +> 25 was disproven server-side for managed knowledge bases (§11.1). + +### Requirement 10: Durable Ingestion Control Plane + +**User Story:** As a user, I want an upload to reliably become searchable even if a +worker crashes, so that documents do not silently fail to index. + +#### Acceptance Criteria + +1. THE Ingestion_Consumer SHALL be a durable, retryable compute resource triggered + by the documents bucket's `ObjectCreated` notification. +2. THE Ingestion_Consumer SHALL resolve each document's knowledge base and + Retrieval_Engine before doing any work. +3. WHEN a document belongs to a legacy knowledge base, THE Ingestion_Consumer + SHALL route it to the existing pipeline. +4. WHEN a document belongs to a managed knowledge base, THE Ingestion_Consumer + SHALL route it to Direct_Ingestion. +5. THE system SHALL NOT index the same document on both backends outside of a + deliberate migration or dual-read pilot. +6. THE Ingestion_Consumer SHALL poll until the document is not merely reported + indexed but **actually retrievable**, and SHALL record those as two distinct + timestamps. +7. THE Ingestion_Consumer SHALL update the `DOC#` record to a terminal + complete or failed state with bounded retries and a durable retry anchor. +8. THE system SHALL NOT perform ingestion orchestration in an in-process + `asyncio.ensure_future` task. +9. THE Ingestion_Consumer SHALL tolerate ingestion latency of at least 300 + seconds for a single document. + +> Rationale: §14.1 — the browser creates an `uploading` row and receives a +> presigned PUT; there is no upload-complete API call, so the S3 event remains the +> only trigger. §5.1 measured a fixed per-knowledge-base warm-up of ~68 s and a +> long tail to 264 s on a 50 KiB PDF, so timeouts must be generous. + +### Requirement 11: Managed Retrieval Configuration + +**User Story:** As a user, I want retrieval against a managed knowledge base to use +the correct API shape and managed reranking, so that results are well ordered. + +#### Acceptance Criteria + +1. THE Managed_Backend SHALL use `managedSearchConfiguration` and SHALL NOT send + `vectorSearchConfiguration`. +2. THE Managed_Backend SHALL request managed reranking rather than + `rerankingModelType: NONE`. +3. THE Managed_Backend SHALL NOT attempt to configure or toggle hybrid search. +4. WHEN a metadata filter is applied, THE system SHALL rely on filters failing + **closed**, as measured. +5. THE Managed_Backend SHALL constrain any isolation-critical filter to `equals` + or `in`. + +> Rationale: §5.1 — `vectorSearchConfiguration` is rejected outright for managed +> knowledge bases. §11 question 3 measured `equals`, `startsWith` and +> `stringContains` on an impossible key all returning 0 results, disproving the +> silent-ignore/fail-open claim. §11.1 — managed reranking measurably separates +> scores (0.89/0.38/0.25/0.21/0.19 versus a nearly flat 1.00/0.84/0.78/0.77/0.77 +> without it), and **the reranker is what makes a 2,000-character cap defensible**. + +### Requirement 12: Enforceable Storage Cost Controls + +**User Story:** As a platform owner, I want stored bytes capped per owner before any +managed knowledge base holds production data, so that storage cost cannot grow into +a six-figure monthly exposure. + +#### Acceptance Criteria + +1. THE system SHALL enforce a per-owner Byte_Cap and a per-knowledge-base + Byte_Cap. +2. THE per-owner default SHALL be **100 MB**, an elevated admin-granted tier SHALL + be **1 GB**, and the per-knowledge-base ceiling SHALL be **500 MB**. All three + SHALL be configurable and resolvable by role tier. These values require product + sign-off before implementation. +3. THE system SHALL determine a document's contribution to the Byte_Cap from an S3 + `HEAD` on the stored object, NOT from a client-reported size. +4. THE system SHALL apply byte accounting as an atomic reserve → commit → release + flow. +5. WHEN two uploads race against the same remaining allowance, THE system SHALL + NOT allow the combined committed total to exceed the Byte_Cap. +6. WHEN an ingestion fails, THE system SHALL release the reservation. +7. THE system SHALL NOT depend on the `RawDataSize` CloudWatch metric for + enforcement. +8. THE system SHALL NOT depend on cost-allocation tags for enforcement. +9. THE Byte_Cap SHALL be enforced before any production traffic is promoted to + Managed_Backend. +10. THE system SHALL define and document whether the knowledge base owner or the + invoking user consumes retrieval quota. +11. THE Byte_Cap SHALL be enforced on **every** path that adds bytes to a managed + knowledge base, including the migration re-ingest path, not only interactive + upload. +12. WHEN a knowledge base's corpus would exceed its owner's remaining allowance, + THE Migration_Worker SHALL reserve for the whole snapshot and fail the + migration **before** entering `shadow`, rather than part-migrating a corpus + that cannot fit. +13. THE system SHALL raise account-level alarms on total managed storage, on + managed knowledge base count against the 10,000 quota, on daily + Knowledge-Base `usagetype` cost, and on a sustained non-zero orphan count. +14. THE system SHALL emit a metric when a Byte_Cap reservation is rejected, so the + chosen default can be validated against real behaviour before it hardens into + policy. + +> Rationale: §13.5 requirement 1 — managed storage is $5.00/GB-month against +> ~$0.15/GB-month today, a 35× increase. The existing 1 GB-per-user allowance +> would permit 30,000 GB at full adoption, i.e. **$150,000/month**. This is the +> only finding in the evaluation that can cause real financial damage. +> `RawDataSize` returned 0 datapoints for a directly-ingested document (§11 +> question 2), so it is unproven for this purpose. + +### Requirement 13: Deletion Sagas and Tombstones + +**User Story:** As a system operator, I want every delete to either complete or +leave a retryable work item, so that a failed delete is never a silent paying leak. + +#### Acceptance Criteria + +1. WHEN deleting a knowledge base, data source, or document, THE system SHALL write + a Tombstone **before** calling AWS. +2. THE system SHALL clear the Tombstone only after AWS confirms the resource is + gone. +3. THE system SHALL NOT treat an accepted delete call as a completed deletion. +4. THE system SHALL verify knowledge base deletion by polling until the resource is + absent, tolerating at least 6 minutes. +5. THE system SHALL NOT delete a knowledge base's service role until all of its + knowledge bases are confirmed absent. +6. THE system SHALL NOT remove the last KB_Record, nor allow TTL to remove it, + until AWS confirms deletion. +7. WHEN a knowledge base reports `DELETE_UNSUCCESSFUL`, THE system SHALL surface it + as an actionable operator state rather than a completed delete. +8. A surviving Tombstone SHALL be discoverable as a retryable work item. + +> Rationale: §12 measured deletion taking 2–6 minutes and verified only by polling +> `ListKnowledgeBases`. §12.2 documents a knowledge base stuck in +> `DELETE_UNSUCCESSFUL` since 2025-11-24 that no reconciler would ever notice. + +### Requirement 14: Daily Reconciler + +**User Story:** As a system operator, I want a daily job that finds AWS resources +our database does not know about, so that crash orphans are detected rather than +paid for indefinitely. + +#### Acceptance Criteria + +1. THE Reconciler SHALL run on a schedule and join a paginated, tag-filtered + `ListKnowledgeBases` against KB_Records. +2. WHEN a Managed_KB exists in AWS with no KB_Record, THE Reconciler SHALL treat it + as an orphan. +3. THE Reconciler SHALL age-gate orphan deletion on the **AWS-reported + `createdAt`**, NOT on the time of discovery. +4. THE Reconciler SHALL delete an orphan only when it is older than 24 hours. +5. WHEN a KB_Record references an AWS_KB_Id that does not exist, THE Reconciler + SHALL mark the record's vector state as missing and SHALL NOT delete the + record. +6. WHEN both sides agree, THE Reconciler SHALL refresh stored-byte accounting. +7. THE Reconciler SHALL run in a report-only mode that logs intended deletions + without performing them, and report-only SHALL be the initial deployed mode. +8. THE Reconciler SHALL apply a bounded per-run action limit. + +> Rationale: §7.4 — age-gating on discovery time means a reconciler that was down +> for a week deletes in-flight creates. §7.3 requires shipping in report-only mode +> and arming later, and warns specifically about the empty-string workflow-variable +> case. + +### Requirement 15: Migration State Machine + +**User Story:** As a knowledge base owner, I want my knowledge base upgraded without +downtime and without re-uploading anything, so that the upgrade is invisible until +it succeeds. + +#### Acceptance Criteria + +1. THE system SHALL migrate a knowledge base through Migration_State + `shadow → verify → promote → retain`, with `failed` as a terminal state that + returns the knowledge base to Legacy_Backend. `reclaim` is reserved in the enum + and SHALL NOT be entered in this phase. +2. THE system SHALL NOT mutate a live knowledge base in place. +3. DURING `shadow` and `verify`, THE knowledge base SHALL remain fully usable and + SHALL continue serving from Legacy_Backend. +4. THE system SHALL re-ingest source bytes from their existing S3 location and + SHALL NOT ask the user to re-supply any document. +5. THE system SHALL migrate only documents whose status is `complete`. +6. THE `verify` step SHALL compare an exact source manifest of `document_id` plus + content hash or generation, NOT document-count parity alone. +7. THE `verify` step SHALL perform at least one canary retrieval that confirms + expected content is returned from the Managed_Backend. +8. `promote` SHALL be a single conditional write flipping Retrieval_Engine to + `"managed"`. +9. THE system SHALL NOT promote unless a catch-up pass has converged. +10. WHEN two workers attempt promotion concurrently, THE conditional write SHALL + allow at most one to succeed. +11. DURING `retain`, THE system SHALL preserve legacy vector data for a rollback + window of at least 30 days. +12. THE system SHALL NOT enter `reclaim` for a knowledge base until the retention + window has expired AND that knowledge base has served managed traffic without + a rollback. +13. THE Migration_Worker SHALL take a lease so that one knowledge base is not + migrated concurrently by two workers. +14. THE Migration_Worker SHALL apply a bounded per-tick dispatch limit. + +> Rationale: §10.3. Timing recomputed from §5.1's revised figures (~73 s median +> create + ~68 s first ingest + ~2.5 s per warm small document): a 20-document +> knowledge base is **~3 minutes** and 100 documents **~6.5 minutes**. §10.3's own +> "4 min / 9.5 min" figures were computed from the **superseded** §5 numbers and are +> not used here. For a PDF-heavy corpus, per-document parse time of 37–264 s +> dominates and a 20-PDF knowledge base can exceed an hour — so this is background +> work only, and progress must be reported per-document rather than as an ETA. + +### Requirement 16: Writes and Deletes During Migration + +**User Story:** As a user, I want to keep uploading and deleting documents while my +knowledge base is upgrading, so that the upgrade does not freeze my work or corrupt +the result. + +#### Acceptance Criteria + +1. DURING migration, THE existing upload path SHALL remain authoritative and SHALL + continue writing to Legacy_Backend. +2. THE Migration_Worker SHALL snapshot the document-id set, migrate it, then run a + catch-up pass for documents created since the snapshot. +3. THE Migration_Worker SHALL repeat catch-up passes until a pass finds nothing + new. +4. THE system SHALL re-read each document's `DOC#` record immediately before + ingesting it, and SHALL skip the document if it no longer exists or is no longer + `complete`. +5. THE system SHALL NOT resurrect a document that was deleted mid-migration. +6. THE system SHALL NOT implement dual-write as the coexistence mechanism. + +### Requirement 17: Rollback + +**User Story:** As a knowledge base owner, I want an upgrade to be undoable, so that +a bad outcome is recoverable immediately rather than requiring data restoration. + +#### Acceptance Criteria + +1. THE system SHALL support rollback by writing Retrieval_Engine back to its prior + value. +2. Rollback SHALL NOT move or restore any data. +3. Rollback SHALL be available for the entire `retain` window. +4. WHEN a migration fails at any stage before `promote`, THE knowledge base SHALL + remain on Legacy_Backend and SHALL remain fully usable. +5. THE system SHALL record a rollback timestamp on the KB_Record. + +### Requirement 18: Dual-Read Pilot + +**User Story:** As a platform owner, I want real comparative evidence before +migrating anyone, so that the rollout rests on measurement rather than on the +benchmark alone. + +#### Acceptance Criteria + +1. THE system SHALL support running both backends for the same query on an opted-in + knowledge base. +2. DURING a dual read, THE system SHALL serve results from Legacy_Backend. +3. THE system SHALL record, per dual read, the overlap in returned `document_id` + values, a rank correlation, and per-backend latency. +4. THE dual-read path SHALL be opt-in per knowledge base and SHALL default to off. +5. THE dual-read path SHALL NOT increase user-visible latency beyond the legacy + path's own latency. + +### Requirement 19: Independent Feature Flags + +**User Story:** As a platform operator, I want to ship the managed backend without +starting a fleet migration, so that the two risks are separable. + +#### Acceptance Criteria + +1. THE system SHALL provide a flag controlling whether new knowledge bases are + created managed. +2. THE system SHALL provide a separate flag controlling whether the + Migration_Worker runs at all. +3. THE system SHALL provide a third, separate flag controlling whether the + Reconciler deletes rather than only reporting. +4. THE three flags SHALL be independently settable. +5. ALL three flags SHALL default to off. +6. WHEN the migration flag is off, THE Migration_Worker SHALL perform no work. +7. WHILE the Reconciler arming flag is off, THE Reconciler SHALL log intended + deletions and delete nothing. +8. THE system SHALL treat an empty-string flag value as off. + +### Requirement 20: IAM, Encryption, and Teardown + +**User Story:** As a security engineer, I want least-privilege, confused-deputy-safe +roles and a teardown that removes runtime-created resources, so that the feature +neither over-grants nor leaks resources. + +#### Acceptance Criteria + +1. THE system SHALL define a dedicated Bedrock knowledge base service role. +2. THE service role's trust policy SHALL constrain `aws:SourceAccount` and SHALL + apply an `ArnLike` condition on `AWS:SourceArn` scoped to `knowledge-base/*`. +3. THE caller's `iam:PassRole` grant SHALL be conditioned on + `iam:PassedToService`. +4. S3 access SHALL be conditioned on `aws:ResourceAccount`. +5. WHERE customer-managed encryption is required, THE system SHALL supply + `serverSideEncryptionConfiguration.kmsKeyArn`. +6. THE system SHALL scope provisioner/migrator CRUD, direct-ingestion, and + inference `bedrock:Retrieve` permissions separately. +7. WHEN synchronous AWS SDK calls are made from an async request path, THE system + SHALL execute them off the event loop. +8. THE teardown script SHALL list and delete only resources tagged for the project + and environment, and SHALL do so **before** deleting their service role and the + platform stack. +9. THE system SHALL include CDK assertions covering the IAM conditions in this + requirement. +10. THE system SHALL grant `cloudwatch:PutMetricData` scoped to the + `{projectPrefix}/ManagedKb` custom namespace on the **calling identities only**. + THE namespace SHALL NOT begin with `AWS`. THE Bedrock service role SHALL NOT + receive this grant. +11. WHEN a Managed_KB is created, THE system SHALL tag it with the project prefix, + the environment, the App_KB_Id, and the owner identity. +12. THE owner tag value SHALL be an opaque identifier and SHALL NOT be an email + address or any other personally identifying value. +13. THE identities that read Bedrock's own per-knowledge-base metrics SHALL be + granted `cloudwatch:GetMetricData` and `cloudwatch:GetMetricStatistics`. Those + metrics live in the `AWS/Bedrock/KnowledgeBases` namespace, which is a **read + source only** and is never a `PutMetricData` target under 20.10. + +> Note: tagging is a hard prerequisite, not housekeeping. Requirement 14.1's +> tag-filtered `ListKnowledgeBases` and Requirement 20.8's teardown both read these +> tags; without them the Reconciler cannot distinguish our resources from anything +> else in the account, and teardown cannot scope itself. + +> **Why 20.10's namespace is not an `AWS/...` one, and must not be "fixed" back to +> one.** CloudWatch reserves every namespace beginning with `AWS` for its own +> services: "You cannot specify a namespace that begins with AWS. Namespaces that +> begin with AWS are reserved for use by Amazon Web Services products." A +> `PutMetricData` grant scoped to `AWS/Bedrock/KnowledgeBases` therefore authorizes +> no publish that can ever succeed — it reads as correct in a policy review and +> silently does nothing. 20.10 and 20.13 cover two different directions of traffic +> that were previously conflated: +> +> - **Writing** this platform's OWN metrics (`KbByteCapRejected`, `KbOrphansFound`, +> `KbIdleGB`, `KbCount`, `KbStorageGB`, `KbQueryClamped`, +> `KbStatusFilterFailClosed`, `KbMigration{Started,Promoted,Failed,RolledBack}`) +> needs `PutMetricData` into the non-reserved `{projectPrefix}/ManagedKb` +> namespace (20.10). The project prefix keeps two environments in one account from +> blending their metrics. +> - **Reading** Bedrock's own per-KB metrics (`Invocations`, `ClientErrors`, +> `ServerErrors`, `Throttles`, `TotalIterationCount`, `RawDataSize`) needs +> `GetMetricData` / `GetMetricStatistics` against `AWS/Bedrock/KnowledgeBases` +> (20.13). Reading a reserved namespace is permitted; only writing is not. + +> Rationale: §14.5 and §14.0. Metric publishing is best-effort and +> permission-gated: omit the grant and metrics silently vanish while requests keep +> succeeding. Managed embedding and managed reranking need no Bedrock model access; +> only `CUSTOM` embedding or reranking does — and Requirement 8.5 chooses `CUSTOM` +> embedding, so that grant is required. + +### Requirement 21: Failed and Stuck Legacy Documents + +**User Story:** As a user whose upload failed months ago without telling me, I want +to find out and retry, so that migration does not quietly drop my document. + +#### Acceptance Criteria + +1. WHEN a knowledge base is migrated, THE system SHALL surface to its owner any + document not in `complete` status that will therefore not be carried across. +2. THE system SHALL offer a retry path for such documents. +3. THE system SHALL NOT silently omit non-`complete` documents without surfacing + them. +4. THE system SHALL distinguish, in user-facing messaging, an unsupported file + format from a processing failure. + +> Rationale: §7.4 measured 1,692 `DOC#` records of which 200 (11.8%) are not +> `complete` — 101 stuck `deleting`, 95 `failed`, 4 `uploading`. §10.3 ingests only +> `complete` documents, so migration would silently drop all 95 failures. §11.2 +> documents that the deployed pipeline cannot ingest `.txt` at all despite the repo +> and frontend both advertising support, producing a 56-second wait and a generic +> failure message. + +### Requirement 22: Observability + +**User Story:** As a system operator, I want to see knowledge base count, stored +bytes, orphans and migration progress, so that cost and correctness problems are +visible before they become incidents. + +#### Acceptance Criteria + +1. THE system SHALL emit metrics for at least: knowledge base count, stored + gigabytes, idle gigabytes, orphans found, and Byte_Cap rejections. THE system + SHALL NOT emit a reclaimed-gigabytes metric, because nothing reclaims in this + phase and a structurally-always-zero metric trains operators to ignore it. +2. THE system SHALL emit migration progress and failure counts. +3. THE system SHALL emit a metric when a query is clamped per Requirement 4. +4. THE system SHALL emit a metric when the Doc_Status_Filter drops chunks because + status could not be confirmed per Requirement 5. +5. THE system SHALL derive idleness from the maximum of the knowledge base's own + last-retrieved time and the last-used time of any bound agent, NOT from + retrieval alone. +6. THE system SHALL NOT write a last-retrieved timestamp on every retrieval. +7. THE system SHALL attribute cost by filtering on `usagetype`, NOT on service code + alone. +8. THE system SHALL treat a sustained non-zero orphan count as the signal that the + delete saga is leaking. + +> Rationale: §7.2 — idleness computed from retrieval alone evicts an actively used +> agent's knowledge base because its queries did not match. §7.3 requires a +> throttled conditional write rather than per-retrieval writes; §14.0 notes +> per-knowledge-base `Invocations` is a cheaper idleness signal. §8 — Managed KB +> bills under `AmazonBedrockAgentCore`, so anything keyed on `AmazonBedrock` misses +> it entirely and anything keyed on service code alone blends it into the Runtime +> memory line. + +### Requirement 23: User Experience + +**User Story:** As a knowledge base owner, I want the upgrade explained honestly and +never forced on me, so that I keep working normally and understand what changed. + +#### Acceptance Criteria + +1. WHEN a knowledge base is on Legacy_Backend and no action is required, THE system + SHALL show no badge, banner, or prompt. +2. WHEN an upgrade is available, THE system SHALL present it as an inline, opt-in + control describing only benefits proven by the §13 benchmark and stating that + the knowledge base keeps working during the upgrade. +3. DURING `shadow` and `verify`, THE system SHALL show non-blocking progress and + SHALL allow the user to navigate away. +4. WHEN promotion succeeds, THE system SHALL show a one-time dismissible notice and + SHALL NOT show a permanent badge. +5. WHEN migration fails, THE system SHALL show a plain-language reason and a retry + control, and the knowledge base SHALL remain usable on Legacy_Backend. +6. THE system SHALL NOT use the word "vector" in user-facing copy. +7. THE upgrade control SHALL be gated on existing edit permission, and viewers + SHALL NOT see it. +8. THE system SHALL NOT auto-migrate knowledge bases silently in this phase. +9. THE admin surface SHALL list knowledge bases filterable by engine with stored + bytes and document counts, and SHALL support bulk migrate and per-knowledge-base + retry. + +### Requirement 24: Minimum Test Coverage + +**User Story:** As a reviewer, I want the risky paths covered by tests before +promotion, so that correctness does not rest on manual verification. + +#### Acceptance Criteria + +1. THE test suite SHALL cover adapter parity across both backends, including score + direction. +2. THE test suite SHALL cover create, ingest, and delete idempotency. +3. THE test suite SHALL cover a crash after the AWS create call but before the + database update. +4. THE test suite SHALL cover record-only and AWS-only reconciliation outcomes. +5. THE test suite SHALL cover uploads and deletes occurring during migration. +6. THE test suite SHALL cover fail-closed document status and fail-closed access + checks. +7. THE test suite SHALL cover byte-cap reservation races. +8. THE test suite SHALL cover a mixed old/new deployment serving simultaneously. +9. THE test suite SHALL cover teardown of tagged dynamic resources. +10. THE test suite SHALL include CDK assertions for the IAM conditions in + Requirement 20. +11. THE test suite SHALL stub managed AWS APIs rather than calling them. +12. THE test suite SHALL assert that resource policies are re-applied after a + rehydration that produces a new AWS_KB_Id. +13. THE test suite SHALL assert the presence of the CloudWatch metric permissions + in Requirement 20.10. +14. THE test suite SHALL cover published-agent corpus behaviour, asserting that an + engine swap does not alter what a published agent retrieves and that a listed + agent is exempt from lifecycle reclaim. +15. THE test suite SHALL assert that `apis.shared.kb_backend` does not transitively + import `apis.shared.assistants`, so the Lambda image constraint is enforced by + test rather than by convention. + +### Requirement 25: Authorization, Isolation, and Publication Semantics + +**User Story:** As a user, I want my knowledge base readable only by people who are +allowed to read it, so that sharing an agent does not silently expose my documents. + +#### Acceptance Criteria + +1. THE system SHALL resolve the invoking user's access to a knowledge base **before** + retrieval is attempted. +2. THE system SHALL reuse the existing assistant permission model rather than + introducing a parallel one, so that owner, editor, and viewer semantics are + unchanged. +3. THE system SHALL treat the application as the authoritative authorization layer. +4. THE system SHALL NOT rely on a metadata filter as the tenant boundary. +5. THE system SHALL NOT adopt ACL-aware retrieval as an authorization mechanism in + this phase. +6. WHERE a knowledge base is shared beyond its owner, THE system SHALL apply a + resource policy for IAM-enforced `bedrock:Retrieve`. +7. WHEN a rehydration or replacement produces a new AWS_KB_Id, THE system SHALL + re-apply any resource policy that was attached to the previous identifier. +8. WHEN a knowledge base's engine is migrated, THE system SHALL NOT change what a + published agent retrieves. +9. WHILE an agent is listed in the marketplace, THE system SHALL exempt its + knowledge base from lifecycle reclaim. +10. WHEN a listed agent transitions to `taken_down`, THE system SHALL require an + explicit transition rather than allowing it to fall through to reclaim. +11. THE system SHALL NOT claim to resolve whether published agents pin a corpus + revision; that question is owned by the marketplace spec and remains open. + +> Rationale: closes evaluation gate §14.3. Managed KB ships two features whose names +> overstate what they provide. AWS's multi-tenant guidance calls metadata filtering +> *"filter-level (logical) isolation, not IAM-enforced (infrastructure) isolation"*, +> and states that ACL-aware retrieval *"is not authorization"* and does not +> authenticate users — its identity is **email only, with no alias resolution, and +> mismatches fail silently**. This platform authenticates via OIDC with claim +> mappings, so a silently-failing email match would be a worse primitive than an +> explicit app-side check. Because this phase holds `App_KB_Id == assistant_id`, the +> per-assistant boundary *is* a per-knowledge-base boundary, which is the strongest +> available isolation by construction. Resource policies are MANAGED-only and attach +> to the AWS knowledge base ARN, so a new identifier silently drops sharing (§11.1). diff --git a/.kiro/specs/managed-kb-migration/tasks.md b/.kiro/specs/managed-kb-migration/tasks.md new file mode 100644 index 000000000..ab1587a18 --- /dev/null +++ b/.kiro/specs/managed-kb-migration/tasks.md @@ -0,0 +1,749 @@ +# Implementation Plan: Managed Knowledge Base Migration + +## Overview + +Introduce Amazon Bedrock Managed Knowledge Base as a second retrieval backend +behind a single abstraction seam, then migrate knowledge bases to it one at a time, +opt-in, with rollback available throughout. + +Task order enforces the deployment rule that **backend code never deploys before +the IAM and resources it requires**. Groups 1–2 are platform-only and change no +behaviour. Groups 3–11 land backend code that stays dark behind flags. Groups +12–13 enable the pilot and opt-in migration. Groups 14–15 add the user-facing +surfaces and the pre-promotion verification gate. + +**Scope:** §14.7 phases 1–4 only. Managed-by-default, stopping legacy writes, +reclaiming legacy vectors, and removing the old pipeline are a follow-up spec. +All three flags — managed-default, migration, and reconciler arming — ship **off**. + +## Tasks + +- [x] 1. Platform: additive schema and IAM (no behaviour change) + - [x] 1.1 Add the sparse work-discovery GSI to the assistants table + - In `infrastructure/lib/constructs/rag/rag-data-construct.ts`, add GSI + `KbWorkIndex` with partition key `GSI7_PK` and sort key `GSI7_SK`, both + STRING, `projectionType: ALL` + - **GSI7, not GSI1** — the table already has six indexes using `GSI_PK`/`GSI_SK` + for the first and `GSI2_PK` through `GSI6_PK` thereafter + - Follow the sparse pattern and comment style of the adjacent `DueSyncIndex` + (GSI4), `AgentDirectoryIndex` (GSI5) and `AgentReportsIndex` (GSI6): keys are + written only while the record is eligible, so ineligible and pinned knowledge + bases are invisible to the dispatcher's query by physics rather than by filter + - Add `GSI7_PK` / `GSI7_SK` to the generic assistant-update path's immutable + attribute list, mirroring `GSI5_*`, so a routine edit cannot resurrect a work + key on a knowledge base that has left the queue + - ⚠️ **This consumes the entire `rag-assistants` GSI budget for whichever + release ships it.** DynamoDB's `UpdateTable` permits exactly ONE GSI creation + or deletion per call, and CloudFormation issues one `UpdateTable` per changed + table, so a release that adds a second index to this table fails the deploy + and rolls the whole stack back. This is not theoretical: it took production + down on 2026-08-01 in release 1.12.0, when `AgentDirectoryIndex` and + `AgentReportsIndex` arrived in separate `develop` merges and collapsed into a + single prod update. If any other in-flight spec adds a GSI to + `rag-assistants`, the two must ship in different releases. + - Regenerate the committed inventory after adding the index: + `cd infrastructure && UPDATE_GSI_INVENTORY=1 npx jest gsi-update-limit`, and + confirm the diff is exactly one line. `infrastructure/test/gsi-update-limit.test.ts` + fails until this is done, and `scripts/release/check-gsi-update-limit.mjs` + re-checks it against `origin/main` on PRs into `main`. + - _Requirements: 15.14, 15.13_ + + - [x] 1.2 Create the Bedrock knowledge base service role + - New construct `infrastructure/lib/constructs/managed-kb/managed-kb-role-construct.ts` + - Trust policy: `bedrock.amazonaws.com` with `aws:SourceAccount` equal to the + account and `ArnLike` on `AWS:SourceArn` scoped to `knowledge-base/*` + - Grant S3 read on the documents bucket conditioned on `aws:ResourceAccount` + - Grant `bedrock:InvokeModel` on `amazon.titan-embed-text-v2:0` only (required + because Requirement 8.5 pins `embeddingModelType: CUSTOM`) + - Grant `cloudwatch:PutMetricData` scoped to the non-reserved + `${prefix}/ManagedKb` namespace. NOT `AWS/Bedrock/KnowledgeBases`: CloudWatch + reserves every namespace beginning with `AWS` and rejects writes to them, so + an `AWS/...`-scoped grant authorizes nothing while looking correct. Bedrock's + own `AWS/Bedrock/KnowledgeBases` metrics are a read source (Req 20.13), not a + publish target + - Publish the role ARN to SSM at `/${prefix}/managed-kb/service-role-arn` + - _Requirements: 20.1, 20.2, 20.4, 20.5, 20.10, 8.5, 8.9_ + + - [x] 1.3 Grant caller permissions for provisioning, ingestion, and retrieval + - Separate policy statements with distinct SIDs for: provisioner/migrator CRUD + (`bedrock:CreateKnowledgeBase`, `CreateDataSource`, `DeleteKnowledgeBase`, + `DeleteDataSource`, `ListKnowledgeBases`, `GetKnowledgeBase`), direct + ingestion (`IngestKnowledgeBaseDocuments`, `DeleteKnowledgeBaseDocuments`, + `GetKnowledgeBaseDocuments`), and inference (`bedrock:Retrieve`) + - Add `iam:PassRole` on the service role conditioned on `iam:PassedToService` + equal to `bedrock.amazonaws.com` + - Attach retrieval to the AgentCore Runtime role and the App API task role; + attach CRUD only to the migration Lambdas' roles + - _Requirements: 20.3, 20.6_ + + - [x] 1.4 Write CDK assertions for the IAM conditions + - New `infrastructure/test/managed-kb.test.ts`, following + `infrastructure/test/kb-sync.test.ts` + - Assert the `aws:SourceAccount` and `ArnLike` `AWS:SourceArn` conditions, the + `iam:PassedToService` condition, the `aws:ResourceAccount` S3 condition, and + the presence of the `PutMetricData` grant on the calling identities, and its + **absence** on the service role + - Assert the S3 statement's **Resource** as well as its Condition: + `aws:ResourceAccount` scopes the account, not the bucket, so without a + Resource assertion the grant can widen to every bucket in the account (file + uploads, fine-tuning, artifacts, SPA) with all tests still green + - Assert the `PutMetricData` namespace does not begin with `AWS`, so nobody + reverts it to the reserved `AWS/Bedrock/KnowledgeBases` namespace that + authorizes no publish + - The `PutMetricData` assertion matters because metric publishing is + best-effort: omit the grant and metrics silently vanish while requests keep + succeeding + - _Requirements: 20.9, 24.10, 24.13_ + +- [x] 2. Platform: worker resources and config + - [x] 2.1 Add the migration construct with dispatcher, worker, and reconciler + - New `infrastructure/lib/constructs/managed-kb/kb-migration-construct.ts`, + following `infrastructure/lib/constructs/kb-sync/kb-sync-construct.ts` + - Three DockerImage Lambdas sharing ONE image + (`backend/Dockerfile.kb-migration`) + - Byte-stable bootstrap stub at + `infrastructure/bootstrap-assets/kb-migration/`, per the + platform-as-bootstrap pattern + - Publish generated function names to SSM under `/${prefix}/kb-migration/` + - EventBridge `rate()` schedule into the dispatcher and into the reconciler + - Wire the construct in `infrastructure/lib/platform-stack.ts` + - _Requirements: 14.1, 15.13, 15.14_ + + - [x] 2.2 Add the ingestion consumer Lambda + - Same construct; triggered by the documents bucket `ObjectCreated` + notification, wired in `platform-stack.ts` alongside the existing + notification to avoid a circular dependency + - Timeout ≥300 s (a 50 KiB PDF was measured at 264 s) and a dead-letter queue + - _Requirements: 10.1, 10.9_ + + - [x] 2.3 Add configuration properties and flags + - In `infrastructure/lib/config.ts`, add a `managedKb` section carrying + `newDefault`, `migrationEnabled`, `reconcilerArmed`, per-owner byte cap + defaults by role tier, and the retention window in days + - Follow the 7-step config pattern: `config.ts` interface → `loadConfig` → + construct → `scripts/common/load-env.sh` → `synth.sh` and `deploy.sh` + (identical context flags) → workflow job-level `env:` → GitHub variable + - All three booleans default to **false**, and an empty string resolves to + false + - _Requirements: 19.1, 19.2, 19.3, 19.4, 19.5, 19.8, 12.2, 14.7, 15.11_ + + - [x] 2.4 Add tagging for reconciliation and teardown + - Tag every runtime-created knowledge base with `prefix`, `env`, `appKbId`, and + an opaque `ownerUserId` + - The owner tag must be an opaque identifier, never an email address or other + PII + - This is a hard prerequisite, not housekeeping: the Reconciler's tag-filtered + `ListKnowledgeBases` and the teardown script both read these tags + - _Requirements: 20.11, 20.12_ + + - [x] 2.5 Add account-level alarms + - New alarms in the managed-kb construct on total managed storage, managed + knowledge base count against 80% of the 10,000 quota, daily + Knowledge-Base `usagetype` cost, and sustained non-zero `KbOrphansFound` + - Use `TreatMissingData.NOT_BREACHING`, matching the posture of the existing + kb-sync, scheduled-runs and prompt-cache observability constructs + - Per-owner caps bound one user; these bound the fleet, and the gap between + ~$169/month expected and ~$15,000/month permitted is why they are required + - _Requirements: 12.13_ + +- [x] 3. KB_Record data layer + - [x] 3.1 Define the KB_Record model + - New `backend/src/apis/shared/kb_backend/records.py` + - Keys `PK=AST#{assistant_id}`, `SK=KB#{app_kb_id}`, with + `app_kb_id == assistant_id` in this phase + - Fields per the design's data-model table, including `retrievalEngine`, + `provisioningState`, `awsKbId`, `awsDataSourceId`, immutable embedding + config, `storedBytes`, `reservedBytes`, `lastRetrievedAt`, migration state + with generation and lease, and lifecycle exemption flags + - _Requirements: 6.1, 6.2, 6.5_ + + - [x] 3.2 Implement conditional state transitions + - `create_provisioning`, `attach_aws_ids`, `promote_engine`, + `rollback_engine`, `set_migration_state`, `acquire_lease` + - Every transition uses a DynamoDB condition expression; `promote_engine` is + conditional on converged catch-up so two workers cannot both promote + - Sparse GSI attributes are written on entering an eligible state and + **removed** on reaching a terminal state + - _Requirements: 15.8, 15.10, 15.13, 17.1, 17.5_ + + - [x] 3.3 Write property test for engine resolution by absence + - **Property 1: absence means legacy** + - Using `hypothesis`, for any KB_Record shape with no `retrievalEngine` + attribute, verify resolution returns the legacy backend, and verify no code + path writes the literal `"s3vectors"` to a record that did not already carry + it + - **Validates: Requirements 1.6, 1.7, 6.6** + - File: `backend/tests/property/test_pbt_kb_engine_resolution.py` + + - [x] 3.4 Write unit tests for conditional transitions + - Concurrent `create_provisioning` yields exactly one winner + - Concurrent `promote_engine` yields exactly one winner + - Terminal transitions remove the GSI attributes + - File: `backend/tests/shared/test_kb_records.py` + - _Requirements: 7.4, 15.10, 15.13_ + +- [x] 4. Backend abstraction seam + - [x] 4.1 Define the protocol and canonical chunk shape + - New `backend/src/apis/shared/kb_backend/protocol.py` + - `KnowledgeBaseBackend` Protocol with `search`, `ingest`, `delete_document` + - Frozen `Chunk` dataclass whose score field is named `relevance` and is + documented as higher-is-more-relevant + - _Requirements: 1.1, 2.1_ + + - [x] 4.2 Implement the backend resolver + - New `backend/src/apis/shared/kb_backend/resolver.py` + - Reads `retrievalEngine` from the KB_Record; absence resolves to + `S3VectorsBackend` + - _Requirements: 1.4, 1.6_ + + - [x] 4.3 Extract the legacy backend verbatim + - New `backend/src/apis/shared/kb_backend/s3vectors_backend.py` + - Move the existing S3 Vectors search path from + `apis/shared/assistants/rag_service.py` and + `apis/shared/embeddings/bedrock_embeddings.py` without functional change + - Convert S3 Vectors cosine **distance** to **relevance** inside this adapter + - _Requirements: 1.2, 1.3, 2.2_ + + - [x] 4.4 Convert the entry point into a facade + - In `apis/shared/assistants/rag_service.py`, reduce + `search_assistant_knowledgebase_with_formatting(assistant_id, query, top_k=5)` + to resolve-then-delegate, preserving its public signature + - Keep emitting a `distance` key in the formatted result, derived from + `relevance`, so no existing consumer breaks on the field rename + - Neither of the two call sites + (`inference_api/chat/routes.py`, `app_api/assistants/routes.py`) changes + - _Requirements: 1.5, 3.4_ + + - [x] 4.5 Write property test for score direction equivalence + - **Property 2: ranking is backend-independent** + - Using `hypothesis`, for any list of chunks with distinct scores, verify both + backends return the known-best chunk first after adapter conversion + - This is the only test that can catch a silent ranking inversion; without it + the failure mode produces no error, just worse answers + - **Validates: Requirements 2.1, 2.2, 2.3, 2.4, 24.1** + - File: `backend/tests/property/test_pbt_kb_score_direction.py` + + - [x] 4.6 Apply the document-status filter above the seam, on both backends + - Move the `status == "complete"` post-filter into the facade so there is one + implementation covering both backends + - It works on the managed path only because `customDocumentIdentifier` is the + platform `document_id` (task 8.4); the filter needs a `document_id` per chunk + to join on + - Keep it on the managed path even though managed ingestion makes it largely + redundant — removing it in the same change that swaps the engine would + confound the comparison + - Apply the 2,000-character context cap in the same place, for the same reason + - _Requirements: 3.2, 3.3_ + + - [x] 4.7 Write test for parity properties on the managed path + - Assert `top_k=5`, the 2,000-character cap, the status filter, and the + 500-character citation clip all hold on the managed backend, not just legacy + - _Requirements: 3.1, 3.2, 3.3, 3.4_ + + - [x] 4.8 Write architecture test for the Lambda import constraint + - Assert `apis.shared.kb_backend` does not transitively import + `apis.shared.assistants`, whose `__init__` drags in the embeddings stack + - Add alongside the existing boundary tests in `backend/tests/architecture/` + - Keep `kb_backend/__init__.py` empty and heavy imports function-local, matching + the convention in `kb_sync/records.py` + - _Requirements: 24.15_ + +- [x] 5. Query clamp + - [x] 5.1 Implement the query guard + - New `backend/src/apis/shared/kb_backend/query_guard.py` with + `MAX_QUERY_CHARS = 10_000` + - Applied in the facade before backend dispatch so both backends are protected + identically; never raises + - Emit a `KbQueryClamped` metric on truncation + - _Requirements: 4.1, 4.2, 4.3, 4.4, 22.3_ + + - [x] 5.2 Remove the stale no-validation assertion + - In `apis/shared/embeddings/bedrock_embeddings.py`, delete the inline comment + stating the query is a "short string, no token validation needed" + - It is true only because Titan v2 tolerates ~32,000 characters; Managed KB + caps `Retrieve` input at 10,000 and the limit is not adjustable + - _Requirements: 4.5_ + + - [x] 5.3 Write property test for the clamp + - **Property 3: clamp is total and non-throwing** + - Using `hypothesis`, for any input string of any length, verify the output is + at most 10,000 characters, the function never raises, and a truncation signal + is emitted exactly when the input exceeded the cap + - **Validates: Requirements 4.1, 4.3, 4.4** + - File: `backend/tests/property/test_pbt_kb_query_clamp.py` + +- [x] 6. Fail-closed document status filter + - [x] 6.1 Make the status filter fail closed + - In `apis/shared/assistants/rag_service.py`, change + `_filter_vectors_by_document_status` so both fallback paths drop chunks + instead of returning them unfiltered: + the missing-table-name branch (currently `valid_doc_ids = doc_ids`) and the + outer exception handler (currently `valid_doc_ids = doc_ids # Graceful + degradation`) + - Leave the per-document handler as-is; it already fails closed + - Emit `KbStatusFilterFailClosed` at error level, distinct from an ordinary + empty-result log line + - _Requirements: 5.1, 5.2, 5.3, 5.4, 22.4_ + + - [x] 6.2 Record the supersession in the prior spec + - In `.kiro/specs/reliable-document-deletion/requirements.md`, annotate + Requirement 3.4 as superseded by Requirement 5 of this spec + - That requirement specified the fail-open deliberately, so retiring it is a + supersession and must be recorded rather than silently contradicted + - _Requirements: 5.5_ + + - [x] 6.3 Write property test for fail-closed behaviour + - **Property 4: unconfirmable status never leaks** + - Using `hypothesis`, for any set of vectors and any injected table-level + failure or missing table-name condition, verify zero chunks are returned + - **Validates: Requirements 5.1, 5.2, 24.6** + - File: `backend/tests/property/test_pbt_kb_status_fail_closed.py` + + - [x] 6.4 Update existing tests that assert the fail-open contract + - Search `backend/tests/` for tests asserting unfiltered fallback and invert + their expectations, citing this spec's Requirement 5 + - _Requirements: 5.5_ + +- [x] 7. Byte cap accounting + - [x] 7.1 Implement reserve / commit / release + - New `backend/src/apis/shared/kb_backend/byte_cap.py` + - `reserve` is a conditional update failing when + `storedBytes + reservedBytes + n > cap`; `commit` moves reserved to stored; + `release` returns the reservation on failure + - Resolve the per-owner cap by role tier, defaulting **below** the existing + 1 GB user-files precedent + - Determine size from an S3 `HEAD` on the stored object, never from a + client-reported value + - Do not read `RawDataSize` for enforcement; it returned 0 datapoints for a + directly-ingested document and remains unconfirmed + - _Requirements: 12.1, 12.2, 12.3, 12.4, 12.6, 12.7, 12.8_ + + - [x] 7.2 Document the retrieval-quota payer decision + - Record in the design whether the knowledge base owner or the invoking user + consumes retrieval quota, and implement accordingly + - _Requirements: 12.10_ + + - [x] 7.3 Write property test for reservation races + - **Property 5: the cap is never exceeded under concurrency** + - Using `hypothesis`, for any interleaving of N concurrent reserve/commit + operations against a cap, verify the committed total never exceeds the cap + and released reservations are fully returned + - **Validates: Requirements 12.4, 12.5, 12.6, 24.7** + - File: `backend/tests/property/test_pbt_kb_byte_cap.py` + + - [x] 7.4 Enforce the cap on the migration re-ingest path + - The Migration_Worker reserves for the **whole snapshot** before entering + `shadow`, and fails the migration up front rather than part-migrating a corpus + that cannot fit + - Surface the failure as a plain-language reason with the option to request an + elevated tier + - Migration is the largest byte-adding operation in the system and the only one + that runs unattended, so it is both the easiest and the worst place to omit + the check + - Emit `KbByteCapRejected` on rejection + - _Requirements: 12.11, 12.12, 12.14_ + + - [x] 7.5 Write test for migration byte-cap rejection + - A corpus exceeding the owner's remaining allowance fails before `shadow`, and + leaves no partially-ingested managed knowledge base behind + - _Requirements: 12.11, 12.12_ + +- [x] 8. Managed backend: provisioning and retrieval + - [x] 8.1 Implement the provisioning saga + - New `backend/src/apis/shared/kb_backend/provisioning.py` + - Write the KB_Record in `provisioning` **before** calling AWS; attach returned + ids with a conditional update + - `CreateKnowledgeBase` with `type="MANAGED"`, `roleArn`, and + `managedKnowledgeBaseConfiguration` carrying the embedding pin (it has no + required members, but the pin has nowhere else to live — NOT literally `{}`); + omit `storageConfiguration` entirely + - Build the `clientToken` programmatically to satisfy the **33-character + minimum** and persist it so a retry reuses it — a natural + `{id}-{variant}-kb` token is 31 characters and fails client-side validation + - Treat "unable to verify the specified embedding model" as **retryable**; it + was observed as pure IAM eventual consistency against a model confirmed + ACTIVE and invokable + - _Requirements: 7.1, 7.2, 7.3, 7.4, 7.5, 7.6, 7.7, 7.8, 8.1, 8.2_ + + - [x] 8.2 Create the CUSTOM connector data source + - `dataSourceConfiguration.type = "MANAGED_KNOWLEDGE_BASE_CONNECTOR"` with the + real type in + `managedKnowledgeBaseConnectorConfiguration.connectorParameters` + - `embeddingModelType: CUSTOM` pinned to `amazon.titan-embed-text-v2:0`, + `FLOAT32` (upper-case: that is the service-model enum value), 1024 dimensions + - `mediaExtractionConfiguration.imageExtractionConfiguration.imageExtractionStatus + = ENABLED` — opt-in, and silently indexes no chart or image content if left + default + - `dataDeletionPolicy = RETAIN` at creation, the documented remedy for the + `DELETE_UNSUCCESSFUL` state already present in the dev account + - _Requirements: 8.3, 8.4, 8.5, 8.6, 8.7, 8.8_ + + - [x] 8.3 Implement managed retrieval + - New `backend/src/apis/shared/kb_backend/managed_backend.py` + - Use `managedSearchConfiguration` with `numberOfResults=5` and + `rerankingModelType="MANAGED"`; never send `vectorSearchConfiguration`, which + is rejected outright for managed knowledge bases + - Do not attempt to configure hybrid search; it is not toggleable + - Constrain any isolation-critical filter to `equals` or `in` + - Run synchronous boto3 calls off the event loop + - _Requirements: 11.1, 11.2, 11.3, 11.4, 11.5, 3.1, 20.7_ + + - [x] 8.4 Implement direct ingestion and document delete + - `IngestKnowledgeBaseDocuments` batched at **10 documents maximum**, + server-enforced; the user guide's claim of 25 does not apply to managed + knowledge bases + - `customDocumentIdentifier = document_id`, which retires the + `{doc_id}#{chunk_index}` scheme including `delete_vector_tail` and the + chunk-shrinkage stash on this path + - Never call `StartIngestionJob` — 0.1 RPS account-wide and not adjustable + - Bound concurrency against the 10-per-account concurrent document-operation + limit + - _Requirements: 9.1, 9.2, 9.3, 9.4, 9.5, 9.6_ + + - [x] 8.5 Write unit tests with stubbed AWS APIs + - Stub `bedrock-agent` and the agent runtime client; never call live + - Cover create/ingest/delete idempotency, the 10-document batch boundary, + retryable embedding-verification failure, and `clientToken` length ≥33 + - File: `backend/tests/shared/test_managed_kb_backend.py` + - _Requirements: 24.2, 24.11_ + + - [x] 8.6 Write test for crash between AWS create and record update + - Simulate a crash after `CreateKnowledgeBase` returns but before the + conditional update; verify the record remains a discoverable retry anchor and + that a retry does not create a second knowledge base + - _Requirements: 7.8, 24.3_ + +- [x] 9. Ingestion control plane + - [x] 9.1 Implement the ingestion consumer + - New `backend/src/apis/app_api/kb_migration/ingestion_consumer.py` + - Follow `kb_sync/records.py`'s raw-table-access convention: importing + `apis.shared.assistants` drags in the whole embeddings stack, and keeping the + Lambda image small is a deliberate constraint + - Resolve each document's knowledge base and engine, then route legacy + documents to the existing pipeline and managed documents to direct ingestion + - Never index the same document on both backends outside a deliberate migration + or pilot + - Poll until **actually retrievable**, recording `indexedAt` and + `retrievableAt` as two distinct timestamps + - Update `DOC#` to a terminal state with bounded retries and a durable retry + anchor + - No in-process `asyncio.ensure_future` orchestration + - _Requirements: 10.2, 10.3, 10.4, 10.5, 10.6, 10.7, 10.8_ + + - [x] 9.2 Write unit tests for routing exclusivity + - Legacy document routes to the old pipeline only; managed document routes to + direct ingestion only; neither is double-indexed + - File: `backend/tests/lambdas/test_kb_ingestion_consumer.py` + - _Requirements: 10.3, 10.4, 10.5_ + +- [x] 10. Deletion sagas and reconciler + - [x] 10.1 Implement tombstones + - New `backend/src/apis/shared/kb_backend/tombstones.py` + - Write `KBTOMB#{app_kb_id}` (and the `#DOC#{document_id}` variant) **before** + calling AWS; clear only after AWS confirms absence + - No TTL on tombstones — TTL removal would recreate the silent-leak class this + design exists to close + - Verify knowledge base deletion by polling `ListKnowledgeBases` until the name + disappears, tolerating ≥6 minutes; deletion took 2–6 minutes when measured + - Never delete the service role until all of its knowledge bases are confirmed + absent, and never delete the last KB_Record before AWS confirms + - Surface `DELETE_UNSUCCESSFUL` as an actionable operator state + - _Requirements: 13.1, 13.2, 13.3, 13.4, 13.5, 13.6, 13.7, 13.8_ + + - [x] 10.2 Implement the daily reconciler + - New `backend/src/apis/app_api/kb_migration/reconciler.py` + - Join paginated, tag-filtered `ListKnowledgeBases` against KB_Records + - AWS-only ⇒ orphan, deleted only if the **AWS-reported `createdAt`** is >24 h + old; age-gating on discovery time would make a reconciler that was down for a + week delete every in-flight create + - Record-only ⇒ mark `vectorState: missing` and **never** delete the record; + the documents are still valid + - Both ⇒ refresh `storedBytes` + - Ship in **report-only** mode, which logs intended deletions and deletes + nothing; arming is a separate flag that treats an empty string as off + - Apply a bounded per-run action limit + - _Requirements: 14.1, 14.2, 14.3, 14.4, 14.5, 14.6, 14.7, 14.8, 19.7_ + + - [x] 10.3 Write reconciliation tests + - Record-only and AWS-only outcomes; age-gate honours AWS `createdAt` rather + than discovery time; report-only performs no deletes + - File: `backend/tests/lambdas/test_kb_reconciler.py` + - _Requirements: 24.4_ + +- [x] 11. Authorization, isolation, and publication + - [x] 11.1 Resolve access before retrieval + - In the facade, resolve the invoking user's access to the knowledge base + **before** attempting retrieval, reusing the existing assistant permission + model rather than introducing a parallel one + - Because this phase holds `App_KB_Id == assistant_id`, "can this user invoke + this agent" already answers "may this turn retrieve"; do not build for the + 0..N case, which is F4's problem + - _Requirements: 25.1, 25.2, 25.3_ + + - [x] 11.2 Keep filters out of the tenant boundary + - Do not use a metadata filter as the isolation mechanism; the per-knowledge-base + boundary is the tenant boundary in this phase + - Do not adopt ACL-aware retrieval: its identity is email-only with no alias + resolution and mismatches fail silently, which is a worse primitive than an + explicit app-side check on an OIDC claim-mapped platform + - _Requirements: 25.4, 25.5, 11.5_ + + - [x] 11.3 Apply resource policies for shared knowledge bases + - Where a knowledge base is shared beyond its owner, attach a resource policy + granting IAM-enforced `bedrock:Retrieve` + - Re-apply the policy whenever a new `awsKbId` is produced; policies attach to + the AWS ARN, so a replacement silently drops sharing + - _Requirements: 25.6, 25.7_ + + - [x] 11.4 Preserve published-agent semantics + - An engine migration must not change what a published agent retrieves; parity + is the contract, so a swap is not a corpus change and needs no re-review + - Exempt listed agents' knowledge bases from lifecycle reclaim; `taken_down` + requires an explicit transition rather than falling through + - Do not attempt to resolve corpus-revision pinning; it belongs to the + marketplace spec + - _Requirements: 25.8, 25.9, 25.10, 25.11_ + + - [x] 11.5 Write authorization tests + - Viewer can read through the agent but never sees the upgrade control; a user + with no access never reaches retrieval; access checks fail closed + - Published-agent corpus behaviour and reclaim exemption + - Resource policy is re-applied after a new `awsKbId` + - _Requirements: 24.6, 24.12, 24.14_ + + - [x] 11.6 Write test asserting the 1:1 binding freeze + - Assert an explicit `knowledge_base` binding is still rejected by + `binding_validation.py` and that `bindable_catalog.py` still returns an empty + list for it, so the freeze is enforced by test rather than by intention + - _Requirements: 6.7, 6.8_ + +- [x] 12. Dual-read pilot + - [x] 12.1 Implement opt-in dual read + - In the facade, when a knowledge base is flagged for the pilot, run both + backends for the same query and **serve legacy** + - Record per read: overlap in returned `document_id` values, a rank + correlation, and per-backend latency + - Default off; must not increase user-visible latency beyond the legacy path's + own latency + - _Requirements: 18.1, 18.2, 18.3, 18.4, 18.5_ + + - [x] 12.2 Write dual-read tests + - Legacy results are always the ones served; comparison metrics are emitted; + a managed-side failure does not fail the turn + - File: `backend/tests/shared/test_kb_dual_read.py` + - _Requirements: 18.2, 18.5_ + +- [x] 13. Migration dispatcher and worker + - [x] 13.1 Implement the dispatcher + - New `backend/src/apis/app_api/kb_migration/dispatcher.py`, following + `kb_sync/dispatcher.py` + - Query the sparse `KbWorkIndex`, apply a bounded per-tick dispatch limit + (mirroring `KB_SYNC_DISPATCH_LIMIT`, default 20), and no-op entirely when the + migration flag is off + - _Requirements: 19.6, 15.14_ + + - [x] 13.2 Implement the migration worker state machine + - New `backend/src/apis/app_api/kb_migration/worker.py` + - `shadow`: provision, then re-ingest every `complete` document from its + existing S3 key at `assistants/{assistant_id}/documents/{document_id}/{filename}` + — never ask the user to re-supply anything + - `verify`: compare an exact source manifest of `document_id` + content hash or + generation, **not** document-count parity, then run a canary retrieval + - `promote`: single conditional write of `retrievalEngine="managed"`, only after + a converged catch-up pass **and** only once the Byte_Cap is enforced on this + knowledge base — no traffic is promoted to an unmetered corpus + - `retain`: set `retainUntil` at least 30 days out + - Take a lease so one knowledge base is never migrated by two workers + - _Requirements: 15.1, 15.2, 15.3, 15.4, 15.5, 15.6, 15.7, 15.8, 15.9, 15.11, 15.13, 12.9_ + + - [x] 13.3 Implement catch-up convergence + - Snapshot the doc-id set, migrate, then run catch-up passes until a pass finds + nothing new — the same converge-on-quiet shape as the crawler's + consecutive-miss rule + - Re-read each document's `DOC#` record immediately before ingesting and skip it + if gone or no longer `complete`, so a document deleted mid-migration cannot + resurrect + - Do not implement dual-write; one write path stays authoritative until + promotion + - _Requirements: 16.1, 16.2, 16.3, 16.4, 16.5, 16.6_ + + - [x] 13.4 Implement rollback + - Write `retrievalEngine` back to its prior value and stamp `rolledBackAt`; + move no data + - Available for the entire `retain` window; a pre-promotion failure leaves the + knowledge base on legacy and fully usable + - _Requirements: 17.1, 17.2, 17.3, 17.4, 17.5_ + + - [x] 13.5 Write property test for migration idempotency + - **Property 6: interrupted migration converges without duplication** + - Using `hypothesis`, for any interruption point in the state machine, verify a + resumed run reaches the same terminal state, creates exactly one knowledge + base, and ingests each document at most once + - **Validates: Requirements 15.9, 15.10, 15.13, 7.4** + - File: `backend/tests/property/test_pbt_kb_migration_convergence.py` + + - [x] 13.6 Write tests for interference during migration + - Upload during migration is picked up by catch-up; delete during migration + never resurrects; concurrent promotion attempts yield one winner + - File: `backend/tests/lambdas/test_kb_migration_worker.py` + - _Requirements: 16.2, 16.4, 16.5, 24.5_ + + - [x] 13.7 Write test for resource-policy re-application after rehydration + - A rehydration producing a new `awsKbId` re-applies the resource policy; + policies attach to the AWS ARN, so a new id otherwise silently drops sharing + - _Requirements: 24.12_ + + - [x] 13.8 Write test for mixed old/new deployment + - Old and new code serving simultaneously; a record with no `retrievalEngine` + resolves to legacy under both + - _Requirements: 1.6, 24.8_ + +- [ ] 14. Surfaces, observability, and teardown + - [x] 14.0 Register the managed backend in the resolver + - **Spec gap, found during implementation.** `register_backend` was defined in + task 4.2 and called by nothing; task 8.3's note that it would register the + managed backend was never carried out, and no other task picked it up. Every + group could therefore have been completed with the feature unreachable: a + promoted record raises `BackendUnavailable`, which is a correct fail-safe and + a useless signal — visible only to the single migrated user. + - Registered at **import** rather than by a startup call, so there is no + sequence to remember and no service that can come up half-configured. Free + because both adapters' module bodies are stdlib-only and their clients are + lazy, which `test_kb_backend_boundary.py` now asserts for `managed_backend` + too. + - Does **not** make the feature live: nothing resolves to managed until a + record says so, and only a promotion writes that. + - _Requirements: 1.4, 2.5_ + + - [x] 14.1 Emit EMF metrics + - Alongside the existing PromptCache metrics: `KbCount`, `KbStorageGB`, + `KbIdleGB`, `KbOrphansFound`, `KbQueryClamped`, + `KbStatusFilterFailClosed`, and + `KbMigration{Started,Promoted,Failed,RolledBack}` + - Compute idleness as `max(own lastRetrievedAt, max(lastUsedAt) over bound + agents)`, never retrieval alone, or an actively used agent's knowledge base is + evicted because its queries did not match + - Write `lastRetrievedAt` through a throttled conditional write (one winner per + 24 h), never per retrieval; prefer per-knowledge-base `Invocations` from + `AWS/Bedrock/KnowledgeBases` where sufficient — that is a *read* of Bedrock's + own namespace and needs `cloudwatch:GetMetricData` / + `GetMetricStatistics` (Req 20.13). Our own metrics above publish to + `${prefix}/ManagedKb` (Req 20.10), never into an `AWS/...` namespace + - _Requirements: 22.1, 22.2, 22.5, 22.6, 22.8, 20.13_ + + - [x] 14.2 Document cost attribution + - Record that Managed KB bills under `AmazonBedrockAgentCore` and that queries + must filter on `usagetype` — keying on `AmazonBedrock` misses it entirely, and + keying on service code alone blends it into the AgentCore Runtime memory line + - _Requirements: 22.7_ + + - [x] 14.3 Build the upgrade UX + - In `frontend/ai.client/src/app/knowledge-base/knowledge-base-section.component.ts`, + add the opt-in upgrade card, non-blocking progress, one-time success notice, + and a failure state with a retry control + - Show nothing at all for a legacy knowledge base needing no action + - Never use the word "vector" in user-facing copy + - Gate on the existing `_require_edit_permission`; viewers never see the control + - Angular signals, `OnPush`, Tailwind utilities, both light and dark modes, + WCAG AA + - **Spec gap, found during implementation.** This task was written as + frontend-only, but **nothing enrolled a knowledge base**. The worker picks + up records already in `shadow`; the dispatcher sweeps GSI7; no code path + wrote either. Group 14 could have been called complete with the feature + still unreachable. Required a new HTTP surface: + `backend/src/apis/app_api/kb_upgrade/` (`models.py`, `service.py`, + `routes.py`) — `GET`/`POST` `…/knowledge-base/upgrade`, `POST …/retry`, + `POST …/notice`. + - Kept in its **own package**, not `app_api/kb_migration/`: that package's + modules share one size-constrained Lambda image, and this one imports + `apis.shared.assistants` for the permission model, which pulls the + embeddings stack at module scope. + - Enrolment is **two conditional writes**, not one put. `KbRecord.to_item` + does not write `GSI7_PK`/`GSI7_SK` — only `set_migration_state` maintains + them — so the obvious one-put enrolment produces a record that claims to be + migrating and is invisible to the dispatcher *forever*, behind a spinner + that never moves. Asserted by + `test_enrolment_writes_the_dispatcher_work_keys` and + `test_the_created_record_does_not_claim_to_be_migrating`. + - The **offer is gated on `MANAGED_KB_MIGRATION_ENABLED`**, the dispatcher's + own flag, read at call time with the same allow-list. Offering an upgrade + the worker cannot perform is a spinner with no engine behind it, so + "available" is made to mean actionable. Off ⇒ phase `none` ⇒ renders + nothing, which is also 23.1's required behaviour. + - Two public transitions added to `kb_backend/records.py`: + `retry_from_failed` (one atomic write — generation bump, re-enter `shadow`, + work keys, `REMOVE migrationError`; guarded on the old generation **and** + still being `failed`) and `dismiss_upgrade_notice`. + - Client (`kb-upgrade.service.ts`) **fails soft**: `getStatus` resolves to + phase `none` rather than rejecting, so a broken upgrade endpoint cannot take + down the documents section it decorates. + - _Requirements: 23.1, 23.2, 23.3, 23.4, 23.5, 23.6, 23.7, 23.8_ + + - [ ] 14.4 Surface failed and stuck documents — **surfacing done, one-click + retry deferred** (still open: see the deferral note below) + - During the upgrade flow, list any non-`complete` document that will not be + carried across and offer retry; 200 of 1,692 production `DOC#` records + (11.8%) are affected, including 95 `failed` whose owners believe the uploads + worked + - Distinguish an unsupported file format from a processing failure in messaging + - **Done:** `classify_document` splits `unsupported_format` / + `processing_failure` / `being_removed` / `still_processing` (Req 21.4), and + the card discloses them collapsed above the offer — *before* the user + commits, so the choice to fix or accept the loss is theirs (Reqs 21.1, 21.3). + - The unsupported-format set is **imported** from + `docling_processor.DOCLING_SUPPORTED_EXTENSIONS`, never copied. A copied + list is the tag-contract defect's exact shape. Its module scope is + stdlib-only, so the import is free. + - `deleting` documents are **deliberately surfaced**, though + `list_assistant_documents` filters them out as soft-deleted: they are 101 of + the 200 affected records, and a user never shown them cannot tell they are + stuck. That filter — plus its stale-document auto-fail *write* — is why this + surface runs its own raw `DOC#` query. + - **Deferred, Req 21.2 (one-click retry).** Ingestion is S3-event-triggered + (`documents/ingestion/handler.py`) and no reprocess endpoint exists, so a + retry control needs new backend that re-fires that pipeline for bytes + already in S3. Not built: it is a change to a live ingestion path and was + explicitly deferred rather than improvised. The card currently directs the + user to re-upload via "Add files", which is a retry path that works today + and needs nothing new. **Close this subtask by either building the + reprocess endpoint or amending Req 21.2 to accept re-upload.** + - _Requirements: 21.1, 21.3, 21.4 (21.2 partial — see above)_ + + - [ ] 14.5 Build the admin surface + - Knowledge bases filterable by engine, with stored bytes and document counts, + bulk migrate, and per-knowledge-base retry + - _Requirements: 23.9_ + + - [x] 14.6 Extend teardown for runtime-created resources + - In `scripts/teardown/destroy.sh`, list and delete only knowledge bases tagged + for the project and environment, **before** deleting their service role and + the platform stack + - Poll until each resource is confirmed absent; "delete call accepted" is not + "resource gone" + - _Requirements: 20.8, 13.4, 13.5_ + + - [x] 14.7 Write teardown test + - Only tagged resources are deleted, and the service role is deleted only after + all its knowledge bases are confirmed absent + - _Requirements: 24.9_ + +- [ ] 15. Pre-promotion verification + - [ ] 15.1 Run the packaged-SDK contract probe + - Using the checked-in environment with **no** `AWS_DATA_PATH` override, run a + create → ingest → retrieve smoke probe against dev-ai + - This is the contract test that the pinned `boto3==1.43.68` and its packaged + service model are genuinely sufficient, rather than the side-loaded model the + evaluation used + - _Requirements: 8.1, 9.1, 11.1_ + + - [ ] 15.2 Probe for an account-level ingestion-concurrency limit + - During the pilot, run a many-knowledge-base backfill to determine whether an + account-level ingestion-concurrency limit exists; the quota page lists none + - Do not size a wide fleet migration before this is answered + - _Requirements: 9.5_ + + - [ ] 15.3 Confirm the full test matrix passes + - Run the backend suite, the infrastructure suite, and `mypy`/`ruff` inside the + dev container + - Verify every Requirement 24 item has a corresponding passing test + - _Requirements: 24.1, 24.2, 24.3, 24.4, 24.5, 24.6, 24.7, 24.8, 24.9, 24.10, 24.11, 24.12, 24.13, 24.14, 24.15_ diff --git a/.kiro/specs/reliable-document-deletion/requirements.md b/.kiro/specs/reliable-document-deletion/requirements.md index a8b31dff4..4207026a1 100644 --- a/.kiro/specs/reliable-document-deletion/requirements.md +++ b/.kiro/specs/reliable-document-deletion/requirements.md @@ -53,7 +53,17 @@ This document specifies the requirements for reliable document deletion in the R 1. WHEN the RAG_Search_Service receives vector search results, THE RAG_Search_Service SHALL extract unique document_id values from the result metadata and look up their status in the Assistants_Table. 2. THE RAG_Search_Service SHALL return only chunks from documents where the status equals "complete" in the Assistants_Table. 3. WHEN a document record does not exist in the Assistants_Table for a given document_id, THE RAG_Search_Service SHALL exclude chunks from that document. -4. IF the Assistants_Table lookup fails due to a DynamoDB error, THEN THE RAG_Search_Service SHALL fall back to returning unfiltered vector results. +4. ~~IF the Assistants_Table lookup fails due to a DynamoDB error, THEN THE RAG_Search_Service SHALL fall back to returning unfiltered vector results.~~ + **SUPERSEDED** by Requirement 5 of `.kiro/specs/managed-kb-migration`, which + inverts this to fail **closed**: an unconfirmable status now drops the chunks. + + This was a deliberate choice here, not an oversight, so retiring it is recorded + rather than silently contradicted. What changed is evidence: the fail-open path + was measured in production, and 936 retrievals in a trailing 30-day window had + chunks removed by this filter — so the documents it guards are real, not + hypothetical, and a lookup failure would have served users content they believe + they deleted. The per-document lookup failure in criterion 3 already failed + closed and is unchanged; only the table-level fallback moved. ### Requirement 4: Inline Cleanup with Retries diff --git a/backend/src/apis/app_api/assistants/routes.py b/backend/src/apis/app_api/assistants/routes.py index 86460c238..279ea7b29 100644 --- a/backend/src/apis/app_api/assistants/routes.py +++ b/backend/src/apis/app_api/assistants/routes.py @@ -52,6 +52,7 @@ update_assistant, update_share_permission, ) +from apis.shared.assistants.kb_access import granted from apis.shared.assistants.rag_service import augment_prompt_with_context, search_assistant_knowledgebase_with_formatting logger = logging.getLogger(__name__) @@ -521,7 +522,15 @@ async def test_chat_endpoint(assistant_id: str, request: AssistantTestChatReques session_id = request.session_id or f"test-{uuid.uuid4().hex[:12]}" # 4. Search vector store for relevant context - context_chunks = await search_assistant_knowledgebase_with_formatting(assistant_id=assistant_id, query=request.message, top_k=5) + # The permission resolved in step 1 is handed to the facade rather than + # re-resolved there (Requirement 25.1): one lookup, and the grant the + # retrieval runs under is provably the one this route checked. + context_chunks = await search_assistant_knowledgebase_with_formatting( + assistant_id=assistant_id, + query=request.message, + top_k=5, + access=granted(assistant_id, user_id, permission), + ) # 5. Augment user message with retrieved context augmented_message = augment_prompt_with_context(user_message=request.message, context_chunks=context_chunks) diff --git a/backend/src/apis/app_api/kb_migration/__init__.py b/backend/src/apis/app_api/kb_migration/__init__.py new file mode 100644 index 000000000..e69de29bb diff --git a/backend/src/apis/app_api/kb_migration/dispatcher.py b/backend/src/apis/app_api/kb_migration/dispatcher.py new file mode 100644 index 000000000..fa0de5690 --- /dev/null +++ b/backend/src/apis/app_api/kb_migration/dispatcher.py @@ -0,0 +1,245 @@ +"""Migration dispatcher: hand due knowledge bases to the worker, a few at a time. + +Requirements 19.6, 15.14. One EventBridge tick reads the sparse ``KbWorkIndex``, +takes at most a bounded number of records, and asynchronously invokes the worker +once per record. It performs no migration itself and holds no state. + +Follows ``apis/app_api/kb_sync/dispatcher.py`` closely — third use of that shape on +this table, after the sync dispatcher and the scheduled-runs dispatcher — so the +things that matter about it are already established: a bounded per-tick limit, one +broken record never starving the sweep, and metrics emitted from the tick rather +than from the worker. + +Why the index makes the queue correct by physics +------------------------------------------------ +``GSI7_PK``/``GSI7_SK`` are written only while a record is work-eligible +(``shadow``, ``verify``, ``promote``) and ``REMOVE``d on reaching a terminal state. +So this dispatcher cannot see a finished knowledge base even if it wanted to: there +is no filter to get wrong, because ineligible records are not in the index. Same +convention as ``DueSyncIndex``, ``AgentDirectoryIndex`` and ``AgentReportsIndex`` +on this table. + +Why it no-ops rather than refusing to start +------------------------------------------- +With ``MANAGED_KB_MIGRATION_ENABLED`` off the tick returns its zeroed counts and +invokes nothing. The Lambda still exists, still runs on schedule, and still emits +metrics — which is what makes turning the flag on a change with a known blast +radius rather than the first time this code has ever executed in production. + +Feature: managed-kb-migration +Requirements: 19.6, 15.14 +""" + +from __future__ import annotations + +import asyncio +import json +import logging +import os +from typing import Any, Dict, List + +logger = logging.getLogger() +logger.setLevel(logging.INFO) + +#: The migration flag. Absent, empty, or anything outside the truthy set means the +#: dispatcher invokes nothing. +FLAG_MIGRATION_ENABLED = "MANAGED_KB_MIGRATION_ENABLED" + +#: Recognised affirmative spellings, matching the reconciler's. An allow-list +#: rather than a truthiness test, because the failure being designed around is a +#: value that is present but empty: ``bool("")`` is correct by luck, +#: ``bool("false")`` is not. +_TRUTHY = frozenset({"1", "true", "yes", "on", "enabled"}) + +#: Mirrors ``KB_SYNC_DISPATCH_LIMIT``'s default of 20 (Requirement 15.14). The +#: limit exists twice over: it bounds the damage of a bug in the index sweep, and +#: it keeps a burst of enrolments from colliding with ``StartIngestionJob``'s +#: 0.1 RPS account-wide ceiling — which is not adjustable, so the only way to stay +#: under it is to not ask. +DEFAULT_DISPATCH_LIMIT = 20 + +#: Ceiling on the env-var override. A larger sweep should require repeated observed +#: ticks, not a variable edit. +DISPATCH_LIMIT_CEILING = 100 + +METRIC_DISPATCHED = "KbMigrationDispatched" +METRIC_DUE = "KbMigrationDue" +METRIC_DISPATCH_FAILED = "KbMigrationDispatchFailed" + + +def migration_enabled() -> bool: + """Whether the dispatcher may invoke the worker at all. + + Read at call time. Bound as a module constant it would be captured at import + and a test overriding the variable would silently get the production value — + the mistake that cost a 33-second test on this feature already. + """ + return (os.environ.get(FLAG_MIGRATION_ENABLED) or "").strip().lower() in _TRUTHY + + +def dispatch_limit() -> int: + """Records taken per tick, bounded above by :data:`DISPATCH_LIMIT_CEILING`.""" + raw = os.environ.get("KB_MIGRATION_DISPATCH_LIMIT") + try: + value = int(raw) if raw else DEFAULT_DISPATCH_LIMIT + except ValueError: + logger.warning( + f"KB_MIGRATION_DISPATCH_LIMIT={raw!r} is not an integer; using " + f"{DEFAULT_DISPATCH_LIMIT}" + ) + return DEFAULT_DISPATCH_LIMIT + if value < 0: + return 0 + if value > DISPATCH_LIMIT_CEILING: + logger.warning( + f"KB_MIGRATION_DISPATCH_LIMIT={value} exceeds the ceiling of " + f"{DISPATCH_LIMIT_CEILING}; clamping" + ) + return DISPATCH_LIMIT_CEILING + return value + + +def _now_iso() -> str: + from apis.shared.timestamps import utc_now_iso + + return utc_now_iso() + + +def _work_states() -> List[str]: + """Every work-eligible state, drained-first. + + Derived from ``WORK_ELIGIBLE_STATES`` rather than restated, with an explicit + priority order laid over it. A record in ``promote`` is one conditional write + from being finished, so serving it ahead of new ``shadow`` work drains the queue + instead of accumulating half-migrated knowledge bases. + + Anything work-eligible but absent from the priority list is appended rather + than dropped. A state added to the records module and forgotten here then + migrates slowly, which is a scheduling nuisance; dropped, it would stall + forever with its work keys written and nothing ever reading them — invisible, + because the record still looks queued. + """ + from apis.shared.kb_backend.records import PROMOTE, SHADOW, VERIFY, WORK_ELIGIBLE_STATES + + priority = (PROMOTE, VERIFY, SHADOW) + ordered = [state for state in priority if state in WORK_ELIGIBLE_STATES] + remainder = sorted(set(WORK_ELIGIBLE_STATES) - set(priority)) + if remainder: + logger.warning( + f"work-eligible states {remainder} are not in the dispatcher's priority " + f"order; sweeping them last" + ) + return ordered + remainder + + +def _invoke_worker(payload: Dict[str, Any]) -> None: + """Async-invoke the migration worker. Same shape as the sync dispatcher's.""" + import boto3 + + function_name = os.environ.get("KB_MIGRATION_WORKER_FUNCTION_NAME") + if not function_name: + raise RuntimeError("KB_MIGRATION_WORKER_FUNCTION_NAME is not set") + + boto3.client("lambda").invoke( + FunctionName=function_name, + InvocationType="Event", + Payload=json.dumps(payload).encode("utf-8"), + ) + + +def _emit_metrics(counts: Dict[str, int]) -> None: + from apis.shared.kb_backend.metrics import emit_count + + for metric, value in ( + (METRIC_DUE, counts.get("Due", 0)), + (METRIC_DISPATCHED, counts.get("Dispatched", 0)), + (METRIC_DISPATCH_FAILED, counts.get("Failed", 0)), + ): + if value: + emit_count(metric, value) + + +async def _due_records(limit: int, now_iso: str) -> List[Dict[str, Any]]: + """Records whose ``dueAt`` has passed, across every work-eligible state. + + Queried per state because ``GSI7_PK`` *is* the state — one partition each — and + trimmed to ``limit`` overall so the bound is on the tick's total work rather + than per state, which is how a three-state sweep would quietly become a + 3× limit. + """ + from apis.shared.kb_backend.records import query_due_work + + collected: List[Dict[str, Any]] = [] + for state in _work_states(): + if len(collected) >= limit: + break + remaining = limit - len(collected) + try: + found = await asyncio.to_thread(query_due_work, state, now_iso, remaining) + except Exception as exc: + logger.error(f"KbWorkIndex query failed for state {state}: {exc}", exc_info=True) + continue + collected.extend(found) + return collected[:limit] + + +async def dispatch_once() -> Dict[str, int]: + """One dispatcher tick. Returns the metric counts (also emitted).""" + counts: Dict[str, int] = {"Due": 0, "Dispatched": 0, "Failed": 0} + + if not migration_enabled(): + logger.info(f"{FLAG_MIGRATION_ENABLED} is not truthy; dispatcher tick is a no-op") + return counts + + limit = dispatch_limit() + if limit == 0: + logger.info("dispatch limit is 0; nothing will be dispatched this tick") + return counts + + now_iso = _now_iso() + due = await _due_records(limit, now_iso) + counts["Due"] = len(due) + logger.info(f"migration dispatcher tick: {len(due)} due records (limit {limit})") + + for record in due: + app_kb_id = record.get("appKbId") + pk = record.get("PK") or "" + assistant_id = pk.split("#", 1)[1] if "#" in pk else "" + if not assistant_id or not app_kb_id: + # A record the index returned but that cannot be addressed. Logged and + # skipped rather than raised: one malformed row must not starve the + # sweep, and it will still be there next tick to be noticed. + logger.error(f"skipping unaddressable KbWorkIndex row: PK={pk!r} appKbId={app_kb_id!r}") + counts["Failed"] += 1 + continue + + try: + _invoke_worker( + { + "assistantId": assistant_id, + "appKbId": app_kb_id, + "migrationState": record.get("migrationState"), + "migrationGeneration": int(record.get("migrationGeneration") or 0), + } + ) + counts["Dispatched"] += 1 + except Exception as exc: + logger.error( + f"failed to dispatch migration for kb {app_kb_id}: {exc}", exc_info=True + ) + counts["Failed"] += 1 + + _emit_metrics(counts) + return counts + + +def lambda_handler(event, context): + """EventBridge entry point. + + Nothing is read from ``event``. The dispatcher's behaviour is a function of the + index and the environment only — the same reasoning that fixed the reconciler's + arming bypass, where an invocation field could turn a report-only job into a + deleting one. + """ + counts = asyncio.run(dispatch_once()) + return {"statusCode": 200, "body": counts} diff --git a/backend/src/apis/app_api/kb_migration/ingestion_consumer.py b/backend/src/apis/app_api/kb_migration/ingestion_consumer.py new file mode 100644 index 000000000..776beb6e2 --- /dev/null +++ b/backend/src/apis/app_api/kb_migration/ingestion_consumer.py @@ -0,0 +1,361 @@ +"""Ingestion consumer for managed knowledge bases. + +Triggered by an EventBridge ``ObjectCreated`` event on the RAG documents bucket. Its +whole job is to decide whether a newly uploaded document belongs to a managed +knowledge base and, if so, ingest it directly. + +Routing exclusivity is the point (Requirements 10.3-10.5) +--------------------------------------------------------- +The legacy pipeline is triggered by its **own**, pre-existing S3 notification on the +same bucket. That notification was deliberately left in place, so for a legacy +document this function's correct behaviour is to **do nothing at all** — the other +Lambda has already got it. Acting here as well would index the same bytes twice: two +sets of vectors, doubled ingestion cost, and duplicate chunks competing in one +result list. + +So the routing table is asymmetric, and that asymmetry is intentional: + +=============== ========================================================= +Engine This function +=============== ========================================================= +legacy (absent) returns immediately, ingesting nothing +managed ingests directly and drives ``DOC#`` to a terminal state +=============== ========================================================= + +The one exception is a deliberate migration or dual-read pilot, which the +Migration_Worker drives and which never routes through an upload event. + +Indexed is not retrievable +-------------------------- +Bedrock reports a document ``INDEXED`` up to a second before it can actually be +retrieved — measured at 0.75-1.03 s. Marking a document ``complete`` on ``INDEXED`` +alone produces the worst kind of bug report: the UI says the upload worked, the user +asks a question straight away, and the answer does not mention their document. So +this polls until a retrieval really returns the document, and records ``indexedAt`` +and ``retrievableAt`` separately so the gap stays measurable instead of becoming +folklore. + +Import boundary +--------------- +Raw DynamoDB table access rather than importing ``apis.shared.assistants``, whose +``__init__`` pulls in the embeddings stack at module scope. Keeping this Lambda's +image small is a deliberate constraint — the same reason +``apis/app_api/kb_sync/records.py`` is written this way. Module-level imports are +stdlib only; everything heavy is function-local. + +No in-process orchestration +--------------------------- +No ``asyncio.ensure_future`` fan-out (Requirement 10.8). One invocation drives its +documents to terminal or fails and lets the event source redeliver. A background +task in a Lambda is killed when the handler returns, which turns a reported success +into a silently half-finished ingestion. +""" + +from __future__ import annotations + +import logging +import os +import time +from typing import Any, Dict, List, Optional, Tuple +from urllib.parse import unquote_plus + +logger = logging.getLogger() +logger.setLevel(logging.INFO) + +#: Terminal document states, taken from ``apis/app_api/documents/models.py``'s +#: ``DocumentStatus`` rather than invented: the facade's status filter serves only +#: ``complete``, so any drift here would silently make documents unretrievable. +STATUS_COMPLETE = "complete" +STATUS_FAILED = "failed" + +#: How long to wait for a document to become genuinely retrievable after Bedrock +#: reports it INDEXED. The observed gap is 0.75-1.03 s; the margin is wide because +#: the cost of waiting is a few seconds of Lambda time and the cost of not waiting +#: is telling a user their upload worked when it is not yet usable. +RETRIEVABLE_POLL_TIMEOUT_SECONDS = 30.0 +RETRIEVABLE_POLL_INTERVAL_SECONDS = 0.5 + +#: Bounded retries on the record update. The event source already redelivers, so +#: this only covers a transient DynamoDB failure inside one invocation; unbounded +#: retries would burn the Lambda timeout and lose the DLQ signal. +MAX_RECORD_UPDATE_ATTEMPTS = 3 + + +class IngestionRoutingError(Exception): + """The event could not be routed, or the document could not be finished.""" + + +def _table(): + import boto3 + + return boto3.resource("dynamodb").Table(os.environ["DYNAMODB_ASSISTANTS_TABLE_NAME"]) + + +def _now_iso() -> str: + from apis.shared.timestamps import utc_now_iso + + return utc_now_iso() + + +def parse_object_key(key: str) -> Tuple[str, str, str]: + """Split ``assistants/{assistant_id}/documents/{document_id}/{filename}``. + + The layout is the existing one and this feature does not change it: migration is + a re-ingest of bytes already in place, never a re-upload. Parsing the key rather + than trusting an event field keeps routing independent of which producer + delivered the notification. + """ + parts = unquote_plus(key).split("/") + if len(parts) < 5 or parts[0] != "assistants" or parts[2] != "documents": + raise IngestionRoutingError( + f"object key {key!r} is not an assistant document path; expected " + f"assistants/{{assistant_id}}/documents/{{document_id}}/{{filename}}" + ) + return parts[1], parts[3], "/".join(parts[4:]) + + +def extract_records(event: Dict[str, Any]) -> List[Dict[str, str]]: + """Normalize EventBridge and raw-S3 notification shapes into one list. + + Both are accepted because the bucket carries both producers — EventBridge feeds + this function, a direct notification feeds the legacy pipeline — so a wiring + change cannot silently stop ingestion. + """ + detail = event.get("detail") + if isinstance(detail, dict) and detail.get("object"): + return [ + { + "bucket": (detail.get("bucket") or {}).get("name", ""), + "key": (detail.get("object") or {}).get("key", ""), + } + ] + + out: List[Dict[str, str]] = [] + for record in event.get("Records") or []: + s3 = record.get("s3", {}) + out.append( + { + "bucket": (s3.get("bucket") or {}).get("name", ""), + "key": (s3.get("object") or {}).get("key", ""), + } + ) + return out + + +def resolve_engine_for(assistant_id: str) -> Tuple[str, Optional[Dict[str, Any]]]: + """The engine serving this assistant's knowledge base, plus its record. + + Delegates to ``records.resolve_engine`` so "absence means legacy" has exactly one + implementation. A missing record is the overwhelmingly common case today and + resolves to legacy, which is why this cannot treat it as an error. + """ + from apis.shared.kb_backend.records import get_kb_record, resolve_engine + + record = get_kb_record(assistant_id, assistant_id) + return resolve_engine(record), record + + +def set_document_terminal( + assistant_id: str, + document_id: str, + status: str, + indexed_at: Optional[str] = None, + retrievable_at: Optional[str] = None, + error: Optional[str] = None, +) -> None: + """Drive the ``DOC#`` record to a terminal state, with bounded retries. + + ``indexedAt`` and ``retrievableAt`` are stored separately on purpose: collapsing + them would erase the only evidence of the INDEXED-to-retrievable gap, which is + what makes "my upload finished but the assistant cannot see it" diagnosable + rather than mysterious. + """ + from botocore.exceptions import ClientError + + sets = ["#status = :status", "updatedAt = :now"] + values: Dict[str, Any] = {":status": status, ":now": _now_iso()} + + if indexed_at: + sets.append("indexedAt = :indexed") + values[":indexed"] = indexed_at + if retrievable_at: + sets.append("retrievableAt = :retrievable") + values[":retrievable"] = retrievable_at + if error: + sets.append("ingestionError = :err") + values[":err"] = error + + expression = f"SET {', '.join(sets)}" + last: Optional[Exception] = None + + for attempt in range(1, MAX_RECORD_UPDATE_ATTEMPTS + 1): + try: + _table().update_item( + Key={"PK": f"AST#{assistant_id}", "SK": f"DOC#{document_id}"}, + UpdateExpression=expression, + # `status` is a DynamoDB reserved keyword. + ExpressionAttributeNames={"#status": "status"}, + ExpressionAttributeValues=values, + ) + return + except ClientError as exc: + last = exc + logger.warning( + f"attempt {attempt}/{MAX_RECORD_UPDATE_ATTEMPTS} to mark " + f"{document_id} {status} failed: {exc}" + ) + if attempt < MAX_RECORD_UPDATE_ATTEMPTS: + time.sleep(0.2 * attempt) + + # Raised rather than swallowed: the record is the durable retry anchor + # (Requirement 10.7), so a document left non-terminal must surface as a failed + # invocation and reach the DLQ instead of looking like a success. + raise IngestionRoutingError( + f"could not mark document {document_id} as {status} after " + f"{MAX_RECORD_UPDATE_ATTEMPTS} attempts: {last}" + ) + + +def wait_until_retrievable( + backend: Any, + kb_ref: str, + document_id: str, + timeout_seconds: Optional[float] = None, + interval_seconds: Optional[float] = None, + sleep: Any = time.sleep, +) -> Optional[str]: + """Poll until a retrieval actually returns ``document_id``. + + Returns the timestamp at which it first became retrievable, or ``None`` on + timeout. A probe that itself errors is treated as "not yet", not as a document + failure: the document is usually fine and merely slow, and failing it would fail + uploads that are about to work. + + The timeouts default to ``None`` and are resolved from the module constants *at + call time*, rather than being bound as default arguments. Default arguments are + evaluated once at import, which makes them unpatchable — the first version of + this function bound them directly and a test that shortened the window had no + effect at all, silently waiting the full production timeout instead. + """ + import asyncio + + if timeout_seconds is None: + timeout_seconds = RETRIEVABLE_POLL_TIMEOUT_SECONDS + if interval_seconds is None: + interval_seconds = RETRIEVABLE_POLL_INTERVAL_SECONDS + + deadline = time.monotonic() + timeout_seconds + while time.monotonic() < deadline: + try: + chunks = asyncio.run(backend.search(kb_ref, document_id, 5)) + except Exception as exc: # noqa: BLE001 - a probe failure is not a document failure + logger.warning(f"retrievability probe for {document_id} failed: {exc}") + chunks = [] + + for chunk in chunks or []: + metadata = getattr(chunk, "metadata", None) or {} + if metadata.get("document_id") == document_id: + return _now_iso() + + sleep(interval_seconds) + + logger.warning( + f"document {document_id} was not retrievable within {timeout_seconds}s; " + f"leaving it short of complete rather than claiming success" + ) + return None + + +def handle_object(bucket: str, key: str) -> Dict[str, Any]: + """Route one uploaded object. Returns a summary for logging and tests.""" + from apis.shared.kb_backend.records import ENGINE_MANAGED + + assistant_id, document_id, filename = parse_object_key(key) + engine, record = resolve_engine_for(assistant_id) + + if engine != ENGINE_MANAGED: + # The legacy pipeline's own S3 notification already owns this document. + # Anything done here would index the same bytes a second time. + logger.info( + f"document {document_id} belongs to a legacy knowledge base; " + f"leaving it to the existing pipeline" + ) + return {"routed": "legacy", "ingested": False, "document_id": document_id} + + aws_kb_id = (record or {}).get("awsKbId") + data_source_id = (record or {}).get("awsDataSourceId") + if not aws_kb_id or not data_source_id: + # Managed engine but no identifiers means provisioning has not finished. + # Failing loudly is correct: silently falling back to legacy would create + # exactly the dual-index this function exists to prevent. + raise IngestionRoutingError( + f"assistant {assistant_id} resolves to the managed engine but its " + f"knowledge base is not provisioned (awsKbId={aws_kb_id!r}, " + f"awsDataSourceId={data_source_id!r})" + ) + + import asyncio + + from apis.shared.kb_backend.managed_backend import ManagedKbBackend + from apis.shared.kb_backend.protocol import DocumentSource + + # The backend takes the App_KB_Id and resolves the AWS identifiers itself on + # every operation. Threading them in from here would defeat that: a + # dormancy/rehydration cycle replaces them, and a caller holding a stale pair + # would keep addressing a knowledge base that no longer exists. The check above + # is still worth doing - it fails fast with a precise reason - but it is a + # precondition, not a value to pass along. + backend = ManagedKbBackend(bucket=bucket) + source = DocumentSource(document_id=document_id, filename=filename, s3_key=key) + + try: + asyncio.run(backend.ingest(assistant_id, source)) + except Exception as exc: + logger.error(f"direct ingestion of {document_id} failed: {exc}", exc_info=True) + set_document_terminal(assistant_id, document_id, STATUS_FAILED, error=str(exc)) + raise + + indexed_at = _now_iso() + retrievable_at = wait_until_retrievable(backend, assistant_id, document_id) + + if retrievable_at is None: + # Ingested but not confirmed retrievable. Left non-terminal deliberately so + # the event source redelivers, rather than the record claiming a success the + # user cannot yet observe. + raise IngestionRoutingError( + f"document {document_id} was ingested but not retrievable within the " + f"poll window; leaving it for redelivery" + ) + + set_document_terminal( + assistant_id, + document_id, + STATUS_COMPLETE, + indexed_at=indexed_at, + retrievable_at=retrievable_at, + ) + return { + "routed": "managed", + "ingested": True, + "document_id": document_id, + "indexedAt": indexed_at, + "retrievableAt": retrievable_at, + } + + +def lambda_handler(event: Dict[str, Any], context: Any) -> Dict[str, Any]: + """Entry point. One invocation drives its documents to terminal, or fails.""" + records = extract_records(event) + if not records: + logger.info("no S3 records in event; nothing to do") + return {"statusCode": 200, "processed": 0, "results": []} + + results = [] + for record in records: + bucket, key = record.get("bucket", ""), record.get("key", "") + if not bucket or not key: + logger.warning(f"skipping record with missing bucket or key: {record}") + continue + results.append(handle_object(bucket, key)) + + return {"statusCode": 200, "processed": len(results), "results": results} diff --git a/backend/src/apis/app_api/kb_migration/reconciler.py b/backend/src/apis/app_api/kb_migration/reconciler.py new file mode 100644 index 000000000..4b79bd5fc --- /dev/null +++ b/backend/src/apis/app_api/kb_migration/reconciler.py @@ -0,0 +1,844 @@ +"""Daily reconciler for managed knowledge bases. + +Joins a paginated, tag-filtered ``ListKnowledgeBases`` against the KB_Records and +acts on the three ways the two sides can disagree. It exists because a managed +knowledge base is a **runtime-created, billed** resource with no CloudFormation +parent: nothing else in the system would ever notice one that our database has +forgotten about. + +The join table +-------------- +============= ============================================================ +Side Action +============= ============================================================ +AWS only Orphan. Delete **only if AWS's own ``createdAt`` is >24 h old** +Record only Mark ``vectorState: missing``. **Never delete the record** +Both Refresh ``storedBytes`` for quota accounting +============= ============================================================ + +Two of those three rows are counter-intuitive, and each is the way it is because +the intuitive version destroys something. + +**Age-gate on AWS's ``createdAt``, never on discovery time.** The tempting +implementation records when the reconciler first *saw* an unknown knowledge base +and waits 24 hours from there. That is wrong in both directions. A reconciler that +was down for a week comes back and treats every knowledge base in the account as +newly discovered — so either it waits another 24 hours on genuine week-old +orphans, or, if the comparison is written the other way round, it deletes every +knowledge base that is mid-provisioning right now, including creates that are 40 +seconds old and about to succeed. AWS's ``createdAt`` is a fact about the +resource, is identical on every run, and does not depend on this process's uptime. +It is obtained from ``GetKnowledgeBase``, because ``KnowledgeBaseSummary`` does +not carry it. + +**A record with no AWS knowledge base is a stale pointer, not a dead corpus.** It +means the *vectors* are gone. The source bytes are still in S3 and the ``DOC#`` +records are still valid and still ``complete``, so the knowledge base can be +rebuilt from them on the next ingest, and the owner never has to re-upload +anything. The record is the only pointer to that recoverable corpus, so deleting +it is the one action here that loses user data — which is why +:func:`mark_vector_state_missing` is the entire response and no code path in this +module removes a KB_Record. + +Report-only, and armed separately +--------------------------------- +This ships **disarmed** (Requirement 14.7, 19.7). It logs exactly what it would +have deleted and deletes nothing, and it runs that way for weeks so its judgement +can be checked against real data before it is trusted with a delete. Arming is one +flag, ``MANAGED_KB_RECONCILER_ARMED``, and an **empty string reads as off** +(Requirement 19.8) — an unset GitHub Actions variable expands to ``""``, which is +how a flag that is obviously off ends up looking truthy to ``if os.environ.get``. + +The per-run limit applies in **both** modes, so the report says what an armed run +would actually do. A report listing 500 intended deletions from a run that would +only ever perform 25 is a misleading artifact, and the whole point of the +report-only period is that the artifact can be trusted. + +Import boundary +--------------- +Module-level imports are stdlib plus the stdlib-only ``kb_backend`` modules; +``boto3`` is function-local, and nothing here reaches ``apis.shared.assistants``. +DynamoDB is accessed through the raw table resource, matching +``kb_migration/ingestion_consumer.py`` and ``kb_sync/records.py``. +""" + +from __future__ import annotations + +import logging +import os +from dataclasses import dataclass, field +from datetime import datetime, timedelta, timezone +from decimal import Decimal +from typing import Any, Callable, Dict, Iterator, List, Optional + +from apis.shared.kb_backend.metrics import emit_count, emit_fleet_gauges +from apis.shared.kb_backend.records import kb_pk, kb_sk + +logger = logging.getLogger() +logger.setLevel(logging.INFO) + +# ── Flags ──────────────────────────────────────────────────────────────────── +# +# The arming flag. Absent, empty, or anything not in the truthy set means the +# reconciler reports and deletes nothing. +FLAG_RECONCILER_ARMED = "MANAGED_KB_RECONCILER_ARMED" + +#: Recognised affirmative spellings. Everything else — including ``""``, ``"0"``, +#: ``"false"`` and ``"off"`` — is off. An allow-list rather than a truthiness test +#: because the failure being designed around is a value that is *present but +#: empty*: ``bool("")`` is correct by luck, ``bool("false")`` is not. +_TRUTHY = frozenset({"1", "true", "yes", "on", "enabled"}) + +# ── Tunables, resolved at call time ────────────────────────────────────────── +# +# Read inside the functions that use them rather than bound as default arguments. +# A default argument is evaluated once at import, so it cannot be patched and a +# test that overrides it silently gets the production value instead. + +#: Requirement 14.4. An orphan younger than this is very likely an in-flight +#: create: provisioning to ``ACTIVE`` was measured at 47-124 s, and the record is +#: written before the AWS call, so the only window in which a legitimate create +#: looks like an orphan is the moments between the two. 24 hours is far wider than +#: needed, which is the correct direction for a destructive action. +ORPHAN_MIN_AGE_HOURS = 24.0 + +#: Requirement 14.8. Bounds the destructive work of a single run, so a bug in the +#: join — or a tag filter that suddenly matches more than it should — costs at +#: most this many knowledge bases before someone sees the report. +MAX_DELETIONS_PER_RUN = 25 + +#: Hard ceiling on :func:`max_deletions_per_run`, above which the env var is +#: ignored. Deleting more than this in one pass is not an operation that should be +#: reachable by editing a variable; it should require repeated, observed runs. +MAX_DELETIONS_CEILING = 100 + +#: Bounds the join itself. A reconciler that walked an unbounded account would +#: time out mid-pass and produce a partial report indistinguishable from a +#: complete one. +MAX_KNOWLEDGE_BASES_PER_RUN = 2000 + +# ── Vector state ───────────────────────────────────────────────────────────── +# +# Written on a record whose AWS knowledge base has gone. Not a failure state: the +# corpus is intact and the next ingest re-provisions. +VECTOR_STATE_MISSING = "missing" + +# ── Metrics ────────────────────────────────────────────────────────────────── +METRIC_ORPHANS_FOUND = "KbOrphansFound" +METRIC_ORPHANS_DELETED = "KbOrphansDeleted" +METRIC_VECTORS_MISSING = "KbVectorsMissing" +METRIC_RECONCILER_LIMIT_REACHED = "KbReconcilerLimitReached" + + +@dataclass +class PlannedDeletion: + """An orphan the reconciler intends to delete, and why it is eligible.""" + + kb_id: str + name: str + status: str + created_at: Optional[str] + age_hours: Optional[float] + performed: bool = False + error: Optional[str] = None + + +@dataclass +class ReconcileReport: + """What one run found and what it did. + + ``armed`` is on the report rather than only in the logs so a stored artifact + is self-describing: an operator reading last night's output should not have to + go and check what the flag was set to at the time. + """ + + armed: bool = False + aws_knowledge_bases: int = 0 + records: int = 0 + matched: int = 0 + orphans: int = 0 + planned_deletions: List[PlannedDeletion] = field(default_factory=list) + skipped_too_young: List[str] = field(default_factory=list) + marked_missing: List[str] = field(default_factory=list) + refreshed_bytes: List[str] = field(default_factory=list) + limit_reached: bool = False + + #: Fleet gauges (Requirement 22.1), accumulated over the record side of the + #: join. Computed from each KB_Record as it was read, so a ``storedBytes`` + #: refresh performed later in the same pass lands in the *next* pass's gauge — + #: acceptable for a daily number, and cheaper than a second full scan. + stored_bytes: int = 0 + idle_bytes: int = 0 + #: Knowledge bases with no recorded activity at all: never retrieved and their + #: agent never used. Reported so ``KbIdleGB`` can be read honestly — these are + #: unmeasured, not idle, and counting them as idle would make every freshly + #: provisioned corpus look abandoned. + unmeasured_idleness: int = 0 + + @property + def deletions_performed(self) -> int: + return sum(1 for planned in self.planned_deletions if planned.performed) + + def to_dict(self) -> Dict[str, Any]: + return { + "armed": self.armed, + "mode": "armed" if self.armed else "report-only", + "awsKnowledgeBases": self.aws_knowledge_bases, + "records": self.records, + "matched": self.matched, + "orphans": self.orphans, + "plannedDeletions": [ + { + "knowledgeBaseId": planned.kb_id, + "name": planned.name, + "status": planned.status, + "createdAt": planned.created_at, + "ageHours": planned.age_hours, + "performed": planned.performed, + "error": planned.error, + } + for planned in self.planned_deletions + ], + "deletionsPerformed": self.deletions_performed, + "skippedTooYoung": self.skipped_too_young, + "markedMissing": self.marked_missing, + "refreshedBytes": self.refreshed_bytes, + "limitReached": self.limit_reached, + "storedBytes": self.stored_bytes, + "idleBytes": self.idle_bytes, + "unmeasuredIdleness": self.unmeasured_idleness, + } + + +# ── Flag and tunable readers ───────────────────────────────────────────────── +def reconciler_armed() -> bool: + """Whether the reconciler may delete. Defaults to **off**. + + An empty string is off (Requirement 19.8). This is not defensive + over-engineering: an unset repository or environment variable expands to the + empty string in GitHub Actions, and this repo has been bitten by that before — + a flag nobody set looking set, in the one component whose mistakes are + irreversible. + """ + raw = os.environ.get(FLAG_RECONCILER_ARMED) + if not raw: + return False + return raw.strip().lower() in _TRUTHY + + +def _env_float(name: str, default: float) -> float: + raw = os.environ.get(name) + if not raw: + return default + try: + return float(raw) + except ValueError: + logger.warning(f"{name}={raw!r} is not a number; falling back to {default}") + return default + + +def _env_int(name: str, default: int) -> int: + raw = os.environ.get(name) + if not raw: + return default + try: + return int(raw) + except ValueError: + logger.warning(f"{name}={raw!r} is not an integer; falling back to {default}") + return default + + +def orphan_min_age_hours() -> float: + return _env_float("MANAGED_KB_ORPHAN_MIN_AGE_HOURS", ORPHAN_MIN_AGE_HOURS) + + +def max_deletions_per_run() -> int: + """The per-run deletion bound, clamped so the environment cannot lift it. + + The env var may lower the limit but not raise it past + :data:`MAX_DELETIONS_CEILING` (Requirement 14.8). A bound that any environment + variable can set to a million is not a bound, and this is the one limit whose + failure mode is irreversible: it is what stops a single bad run — a wrong tag + filter, a botched migration — from deleting an account's worth of user + knowledge bases before anyone reads the report. + """ + requested = _env_int("MANAGED_KB_RECONCILER_MAX_DELETIONS", MAX_DELETIONS_PER_RUN) + if requested > MAX_DELETIONS_CEILING: + logger.warning( + f"MANAGED_KB_RECONCILER_MAX_DELETIONS={requested} exceeds the ceiling of " + f"{MAX_DELETIONS_CEILING}; clamping. Run the reconciler repeatedly rather " + f"than raising this." + ) + return MAX_DELETIONS_CEILING + return max(requested, 0) + + +def max_knowledge_bases_per_run() -> int: + return _env_int("MANAGED_KB_RECONCILER_MAX_SCANNED", MAX_KNOWLEDGE_BASES_PER_RUN) + + +# ── DynamoDB plumbing ──────────────────────────────────────────────────────── +def _table(): + import boto3 + + return boto3.resource("dynamodb").Table(os.environ["DYNAMODB_ASSISTANTS_TABLE_NAME"]) + + +def _now() -> datetime: + return datetime.now(timezone.utc) + + +def _now_iso() -> str: + from apis.shared.timestamps import utc_now_iso + + return utc_now_iso() + + +def iter_kb_records() -> Iterator[Dict[str, Any]]: + """Every KB_Record in the table, paging the scan to exhaustion. + + A scan, because the ``KbWorkIndex`` GSI is *sparse* and deliberately holds + only records that are eligible for migration work — the records this join + cares most about are precisely the ones absent from it. Paged to exhaustion + for the same reason the AWS list is: a truncated read makes every unread + record look like an orphan on the AWS side. + + ``KBTOMB#`` sort keys do not match ``begins_with(SK, "KB#")``, so tombstones + are excluded by the key prefix rather than filtered afterwards. + """ + from boto3.dynamodb.conditions import Attr + + table = _table() + kwargs: Dict[str, Any] = {"FilterExpression": Attr("SK").begins_with("KB#")} + while True: + response = table.scan(**kwargs) + for item in response.get("Items") or []: + yield item + start = response.get("LastEvaluatedKey") + if not start: + return + kwargs["ExclusiveStartKey"] = start + + +# ── Age gate (Requirement 14.3, 14.4) ──────────────────────────────────────── +def parse_aws_timestamp(value: Any) -> Optional[datetime]: + """Coerce AWS's ``createdAt`` to an aware UTC datetime, or ``None``. + + boto3 hands back a ``datetime`` here, but a value that has been through a + stubbed client, an EventBridge payload or a JSON round-trip arrives as a + string or an epoch number. All three are accepted; anything unparseable + returns ``None``, which the age gate treats as *not old enough* rather than + guessing. + """ + if value is None: + return None + if isinstance(value, datetime): + return value if value.tzinfo else value.replace(tzinfo=timezone.utc) + if isinstance(value, (int, float, Decimal)): + try: + return datetime.fromtimestamp(float(value), tz=timezone.utc) + except (OverflowError, OSError, ValueError): + return None + if isinstance(value, str): + from apis.shared.timestamps import from_iso + + try: + return from_iso(value) + except ValueError: + return None + return None + + +def orphan_age_hours(created_at: Any, now: Optional[datetime] = None) -> Optional[float]: + """Hours since **AWS's** ``createdAt``, or ``None`` if it cannot be read.""" + created = parse_aws_timestamp(created_at) + if created is None: + return None + return ((now or _now()) - created).total_seconds() / 3600.0 + + +def orphan_is_deletable( + created_at: Any, + now: Optional[datetime] = None, + min_age_hours: Optional[float] = None, +) -> bool: + """Whether an orphan has existed in AWS long enough to be deleted. + + The input is AWS's ``createdAt``. It is deliberately not "when did we first + see this": see the module docstring. Passing a discovery timestamp here would + type-check, run, pass a naive test, and delete in-flight creates in + production. + + A missing or unparseable ``createdAt`` returns ``False``. Failing closed is + the only safe direction for a destructive action: an orphan left one more day + costs pennies, and a knowledge base deleted 40 seconds into its creation costs + a user their upload. + """ + if min_age_hours is None: + min_age_hours = orphan_min_age_hours() + + created = parse_aws_timestamp(created_at) + if created is None: + return False + return (now or _now()) - created > timedelta(hours=min_age_hours) + + +# ── Record-side actions ────────────────────────────────────────────────────── +def mark_vector_state_missing(assistant_id: str, app_kb_id: str) -> None: + """Record that the AWS knowledge base behind this record has gone. + + **This never deletes the record**, and there is deliberately no function in + this module that does. The vectors are gone; the corpus is not. The uploaded + bytes are still in S3 and the ``DOC#`` records still describe them, so the + next ingest re-provisions a knowledge base and re-indexes from the documents + already present. The record carries the only mapping from ``App_KB_Id`` to + that corpus, so removing it would turn a recoverable, invisible-to-the-user + situation into permanent data loss. + + ``awsKbId``/``awsDataSourceId`` are left in place rather than cleared: they + are the evidence of which AWS resource vanished, and provisioning already + treats a record it cannot find in AWS as needing a fresh create. + """ + _table().update_item( + Key={"PK": kb_pk(assistant_id), "SK": kb_sk(app_kb_id)}, + UpdateExpression=( + "SET vectorState = :missing, vectorStateObservedAt = :now, updatedAt = :now" + ), + ExpressionAttributeValues={":missing": VECTOR_STATE_MISSING, ":now": _now_iso()}, + ) + emit_count(METRIC_VECTORS_MISSING) + + +def refresh_stored_bytes(assistant_id: str, app_kb_id: str, stored_bytes: int) -> None: + """Re-anchor quota accounting, and clear any stale ``vectorState``. + + The ``REMOVE`` matters: a record marked ``missing`` on an earlier run that has + since been re-provisioned would otherwise stay marked for ever, and the UI + would keep telling its owner their knowledge base is broken after it was + fixed. + """ + _table().update_item( + Key={"PK": kb_pk(assistant_id), "SK": kb_sk(app_kb_id)}, + UpdateExpression=( + "SET storedBytes = :bytes, updatedAt = :now " + "REMOVE vectorState, vectorStateObservedAt" + ), + ExpressionAttributeValues={":bytes": Decimal(int(stored_bytes)), ":now": _now_iso()}, + ) + + +def stored_bytes_from_s3(assistant_id: str, bucket: Optional[str] = None, s3_client=None) -> Optional[int]: + """Total size of an assistant's uploaded documents, straight from S3. + + S3 rather than a client-reported or previously-stored value, for the same + reason the byte cap uses a ``HEAD``: this number gates a $150,000/month + exposure at full adoption, and the only trustworthy source for it is the + service holding the bytes. + + Returns ``None`` when no bucket is configured or the listing fails, and the + caller then leaves ``storedBytes`` alone. Writing a zero on a failed listing + would silently hand every owner their whole allowance back. + """ + bucket = bucket or os.environ.get("S3_ASSISTANTS_DOCUMENTS_BUCKET_NAME") + if not bucket: + return None + + if s3_client is None: + import boto3 + + s3_client = boto3.client("s3") + + prefix = f"assistants/{assistant_id}/documents/" + total = 0 + token: Optional[str] = None + try: + while True: + kwargs: Dict[str, Any] = {"Bucket": bucket, "Prefix": prefix} + if token: + kwargs["ContinuationToken"] = token + response = s3_client.list_objects_v2(**kwargs) + for obj in response.get("Contents") or []: + total += int(obj.get("Size") or 0) + if not response.get("IsTruncated"): + return total + token = response.get("NextContinuationToken") + if not token: + return total + except Exception as exc: # noqa: BLE001 - a failed listing must not zero the quota + logger.warning(f"could not total stored bytes for {assistant_id}: {exc}") + return None + + +# ── The run ────────────────────────────────────────────────────────────────── +def reconcile( + client=None, + project_prefix: Optional[str] = None, + environment: Optional[str] = None, + armed: Optional[bool] = None, + now: Optional[datetime] = None, + stored_bytes_resolver: Optional[Callable[[str], Optional[int]]] = None, +) -> ReconcileReport: + """One reconciliation pass. + + ``armed`` defaults to :func:`reconciler_armed`, i.e. to the flag, i.e. to off. + It is an argument only so a test can exercise the armed path without mutating + process environment — never so a caller can conveniently turn deletion on. + """ + from apis.shared.kb_backend import tombstones as tb + + if client is None: + from apis.shared.kb_backend.managed_backend import bedrock_agent_client + + client = bedrock_agent_client() + + if armed is None: + armed = reconciler_armed() + if now is None: + now = _now() + + report = ReconcileReport(armed=armed) + scan_limit = max_knowledge_bases_per_run() + delete_limit = max_deletions_per_run() + min_age = orphan_min_age_hours() + + # ── record side ────────────────────────────────────────────────────────── + # Keyed by awsKbId, because that is the identifier the AWS side reports. A + # record with no awsKbId has not finished provisioning and is not evidence of + # anything: it is skipped rather than counted as a missing-vector record, + # which would mark every in-flight create broken. + records_by_aws_id: Dict[str, Dict[str, Any]] = {} + unprovisioned = 0 + for item in iter_kb_records(): + report.records += 1 + _accumulate_gauges(item, report, now) + aws_kb_id = item.get("awsKbId") + if aws_kb_id: + records_by_aws_id[str(aws_kb_id)] = item + else: + unprovisioned += 1 + + # ── AWS side ───────────────────────────────────────────────────────────── + seen_aws_ids: set = set() + orphan_facts: List[tb.KnowledgeBaseFacts] = [] + + for facts in tb.iter_project_knowledge_bases( + client, project_prefix=project_prefix, environment=environment + ): + if report.aws_knowledge_bases >= scan_limit: + report.limit_reached = True + logger.warning( + f"stopping the AWS walk at {scan_limit} knowledge bases; this run's " + f"join is partial and no deletion decision is made beyond this point" + ) + break + + report.aws_knowledge_bases += 1 + seen_aws_ids.add(facts.kb_id) + + record = records_by_aws_id.get(facts.kb_id) + if record is None: + orphan_facts.append(facts) + else: + report.matched += 1 + _reconcile_matched(record, stored_bytes_resolver, report) + + # ── record only: mark missing, never delete (Requirement 14.5) ──────────── + # + # Only meaningful when the AWS walk completed. On a truncated walk an unmatched + # record may simply be one this run never reached. + if not report.limit_reached: + for aws_kb_id, record in sorted(records_by_aws_id.items()): + if aws_kb_id in seen_aws_ids: + continue + app_kb_id = str(record.get("appKbId") or "") + assistant_id = _assistant_id_of(record) + if not app_kb_id or not assistant_id: + logger.warning(f"skipping malformed KB_Record {record.get('PK')}/{record.get('SK')}") + continue + logger.info( + f"KB_Record {app_kb_id} points at awsKbId {aws_kb_id}, which AWS does " + f"not have. Marking vectorState={VECTOR_STATE_MISSING}. The record is " + f"NOT deleted: its documents are still valid and the knowledge base " + f"rebuilds from them on the next ingest." + ) + mark_vector_state_missing(assistant_id, app_kb_id) + report.marked_missing.append(app_kb_id) + + # ── AWS only: orphans (Requirements 14.2, 14.3, 14.4) ──────────────────── + report.orphans = len(orphan_facts) + if report.orphans: + emit_count(METRIC_ORPHANS_FOUND, value=report.orphans) + + for facts in orphan_facts: + age = orphan_age_hours(facts.created_at, now=now) + # The gate reads AWS's createdAt. Never the time of discovery. + if not orphan_is_deletable(facts.created_at, now=now, min_age_hours=min_age): + report.skipped_too_young.append(facts.kb_id) + logger.info( + f"orphan {facts.kb_id} ({facts.name}) has no KB_Record but AWS " + f"reports createdAt={facts.created_at!r} " + f"(age={age if age is None else round(age, 2)}h < {min_age}h); " + f"leaving it alone — it is most likely an in-flight create" + ) + continue + + planned = PlannedDeletion( + kb_id=facts.kb_id, + name=facts.name, + status=facts.status, + created_at=str(facts.created_at) if facts.created_at is not None else None, + age_hours=None if age is None else round(age, 2), + ) + + # The limit applies whether or not we are armed, so the report describes + # what an armed run would really do. + if len(report.planned_deletions) >= delete_limit: + report.limit_reached = True + emit_count(METRIC_RECONCILER_LIMIT_REACHED) + logger.warning( + f"per-run deletion limit of {delete_limit} reached; {facts.kb_id} and " + f"any further orphans are left for the next run" + ) + break + + report.planned_deletions.append(planned) + + if facts.status == tb.KB_STATUS_DELETE_UNSUCCESSFUL: + # Requirement 13.7 / the tombstone table's fourth row. Retrying the + # delete does not help and the state does not clear on its own, so it + # is surfaced as an operator state instead of being counted as work. + planned.error = tb.KB_STATUS_DELETE_UNSUCCESSFUL + emit_count(tb.METRIC_DELETE_UNSUCCESSFUL) + logger.error( + f"orphan {facts.kb_id} is in {tb.KB_STATUS_DELETE_UNSUCCESSFUL} and " + f"needs operator action; it will not delete by retrying and it is " + f"still being billed" + ) + continue + + if not armed: + # Report-only. This is the shipped mode and it performs no deletes. + logger.warning( + f"[report-only] WOULD delete orphan knowledge base {facts.kb_id} " + f"({facts.name}), AWS createdAt={facts.created_at!r}, " + f"age={planned.age_hours}h. Set {FLAG_RECONCILER_ARMED} to arm." + ) + continue + + try: + _delete_orphan(facts, client) + planned.performed = True + emit_count(METRIC_ORPHANS_DELETED) + logger.info(f"deleted orphan knowledge base {facts.kb_id}") + except Exception as exc: # noqa: BLE001 - one bad orphan must not end the run + planned.error = str(exc) + logger.error(f"failed to delete orphan {facts.kb_id}: {exc}", exc_info=True) + + emit_fleet_gauges( + kb_count=report.records, + stored_bytes=report.stored_bytes, + idle_bytes=report.idle_bytes, + unmeasured=report.unmeasured_idleness, + ) + + logger.info( + f"reconcile complete: mode={'armed' if armed else 'report-only'} " + f"aws={report.aws_knowledge_bases} records={report.records} " + f"matched={report.matched} orphans={report.orphans} " + f"planned={len(report.planned_deletions)} " + f"performed={report.deletions_performed} " + f"markedMissing={len(report.marked_missing)} " + f"unprovisioned={unprovisioned} " + f"storedGB={report.stored_bytes / 1_000_000_000:.3f} " + f"idleGB={report.idle_bytes / 1_000_000_000:.3f} " + f"unmeasured={report.unmeasured_idleness}" + ) + return report + + +def _accumulate_gauges( + record: Dict[str, Any], + report: ReconcileReport, + now: Optional[datetime] = None, +) -> None: + """Fold one KB_Record into the fleet gauges (Requirements 22.1, 22.5). + + The reconciler is where this belongs because it is already the one pass that + walks every knowledge base; a second sweep to count them would double a scan + that exists. + + Idleness comes from :func:`idleness.idle_days`, which takes the **maximum** of + the knowledge base's own ``lastRetrievedAt`` and its bound agents' + ``lastUsedAt``. Never retrieval alone: an agent can be invoked all day and + retrieve nothing, because retrieval only fires when the query matches, so a + corpus judged by retrieval alone looks abandoned exactly when its agent is + busiest with questions the documents do not answer. + + A record with no activity signal at all counts toward ``unmeasured_idleness`` + and **not** toward idle bytes. It is unmeasured, not idle — that is what a + knowledge base provisioned an hour ago looks like. + """ + from apis.shared.kb_backend.idleness import idle_days + + stored = int(record.get("storedBytes") or 0) + report.stored_bytes += stored + + assistant_id = _assistant_id_of(record) + if not assistant_id: + return + + try: + days = idle_days(assistant_id, record, now=_iso_or_none(now)) + except Exception as exc: # noqa: BLE001 - a gauge must not end the pass + logger.warning(f"could not compute idleness for {assistant_id}: {exc}") + return + + if days is None: + report.unmeasured_idleness += 1 + elif days >= idle_threshold_days(): + report.idle_bytes += stored + + +def _iso_or_none(moment: Optional[datetime]) -> Optional[str]: + return moment.strftime("%Y-%m-%dT%H:%M:%SZ") if moment else None + + +def idle_threshold_days() -> int: + """Days without a sign of life before bytes count as idle. + + A **reporting** threshold. Nothing reclaims in this phase, and the number the + follow-up spec eventually evicts on should come from the distribution this + metric records rather than being inherited from this default. + """ + from apis.shared.kb_backend.metrics import IDLE_THRESHOLD_DAYS + + return _env_int("KB_IDLE_THRESHOLD_DAYS", IDLE_THRESHOLD_DAYS) + + +def _assistant_id_of(record: Dict[str, Any]) -> str: + """Recover the assistant id from the record's ``PK``.""" + pk = str(record.get("PK") or "") + return pk[len("AST#") :] if pk.startswith("AST#") else "" + + +def _reconcile_matched( + record: Dict[str, Any], + stored_bytes_resolver: Optional[Callable[[str], Optional[int]]], + report: ReconcileReport, +) -> None: + """Both sides agree: refresh stored bytes (Requirement 14.6). + + Written only when the number actually changed, or when a stale + ``vectorState`` needs clearing. A daily no-op write per knowledge base would + be pure cost and would churn ``updatedAt`` on records nothing happened to. + """ + app_kb_id = str(record.get("appKbId") or "") + assistant_id = _assistant_id_of(record) + if not app_kb_id or not assistant_id: + return + + resolver = stored_bytes_resolver or stored_bytes_from_s3 + actual = resolver(assistant_id) + if actual is None: + return + + current = int(record.get("storedBytes") or 0) + stale_state = record.get("vectorState") is not None + if actual == current and not stale_state: + return + + refresh_stored_bytes(assistant_id, app_kb_id, actual) + report.refreshed_bytes.append(app_kb_id) + + +def _delete_orphan(facts, client) -> None: + """Delete an orphan through the tombstoned saga. + + Through the saga rather than a bare ``DeleteKnowledgeBase`` because an orphan + is by definition a resource a previous delete failed to remove, so the one + thing it must not do is fail silently a second time. The saga writes the + tombstone first, polls until AWS reports the knowledge base absent, and clears + the tombstone only then. + + An orphan has no KB_Record — that is what makes it an orphan — so there is no + assistant id to anchor its tombstone on. It is anchored on the ``appKbId`` tag + the provisioner wrote, falling back to the AWS identifier, and the item is + flagged ``syntheticPartition`` so nobody reads that ``PK`` as a real assistant + and nobody expects ``iter_tombstones()`` to surface it. + ``anchorSource`` records which of the two identifiers was available, because + when this item is being triaged that is the first question. + + ``remove_record`` is left false: there is no record to remove, and this module + never removes one. + """ + from apis.shared.kb_backend import tombstones as tb + from apis.shared.kb_backend.tags import TAG_KEY_APP_KB_ID + + tags = facts.tags or {} + # The AWS tag, whose key is owned by `kb_backend.tags` — not the KB_Record + # attribute, which is separately named `appKbId` and stays that way. + tagged = tags.get(TAG_KEY_APP_KB_ID) + anchor = tagged or facts.kb_id + tb.delete_knowledge_base( + anchor, + anchor, + facts.kb_id, + client=client, + remove_record=False, + extra_attributes={ + tb.SYNTHETIC_PARTITION: True, + "anchorSource": f"tag:{TAG_KEY_APP_KB_ID}" if tagged else "aws:knowledgeBaseId", + "orphanKbName": facts.name or None, + }, + ) + + +def lambda_handler(event: Dict[str, Any], context: Any) -> Dict[str, Any]: + """Scheduled entry point. Returns the report so it lands in the invocation log. + + The invocation event is deliberately **not** consulted for arming. The + environment flag is the only way to arm (Requirement 19.7), because an event + payload is the one input an operator does not review: an EventBridge target + carrying a constant ``{"armed": true}``, or any principal holding + ``lambda:InvokeFunction``, would delete billed user resources while every + piece of reviewable configuration still said report-only, leaving nothing + behind but an ``Invoke`` in CloudTrail. If the event disagrees with the flag, + the flag wins, and the disagreement is logged rather than honoured. + """ + requested = (event or {}).get("armed") + if requested is not None: + logger.warning( + f"ignoring armed={requested!r} from the invocation event: arming is " + f"controlled only by {FLAG_RECONCILER_ARMED}" + ) + report = reconcile() + return {"statusCode": 200, "report": report.to_dict()} + + +__all__ = [ + "FLAG_RECONCILER_ARMED", + "MAX_DELETIONS_CEILING", + "MAX_DELETIONS_PER_RUN", + "MAX_KNOWLEDGE_BASES_PER_RUN", + "METRIC_ORPHANS_DELETED", + "METRIC_ORPHANS_FOUND", + "METRIC_RECONCILER_LIMIT_REACHED", + "METRIC_VECTORS_MISSING", + "ORPHAN_MIN_AGE_HOURS", + "VECTOR_STATE_MISSING", + "PlannedDeletion", + "ReconcileReport", + "iter_kb_records", + "lambda_handler", + "mark_vector_state_missing", + "max_deletions_per_run", + "max_knowledge_bases_per_run", + "orphan_age_hours", + "orphan_is_deletable", + "orphan_min_age_hours", + "parse_aws_timestamp", + "reconcile", + "reconciler_armed", + "refresh_stored_bytes", + "stored_bytes_from_s3", +] diff --git a/backend/src/apis/app_api/kb_migration/worker.py b/backend/src/apis/app_api/kb_migration/worker.py new file mode 100644 index 000000000..89165eea0 --- /dev/null +++ b/backend/src/apis/app_api/kb_migration/worker.py @@ -0,0 +1,945 @@ +"""Migration worker: shadow → verify → promote → retain, one step per invocation. + +Requirements 15, 16, 17. Each invocation takes a lease, executes **one** step for +one knowledge base, records the next state with a conditional write, and returns. +The dispatcher brings it back for the next step. + +Why one step per invocation +--------------------------- +A 20-document text corpus is about 3 minutes end to end, but a 20-PDF corpus can +exceed an hour: per-document parse time was measured at 37–264 s and dominates +everything else. A worker that tried to run the whole machine in one invocation +would therefore be a Lambda that sometimes finishes in three minutes and sometimes +hits its timeout — and a timeout mid-``shadow`` is indistinguishable, from the +outside, from a crash. Stepping means every interruption lands on a recorded state +with a conditional guard in front of it, which is what makes a resumed run converge +instead of duplicating (property test 6). + +Nothing is mutated in place +--------------------------- +The live knowledge base keeps serving from the legacy backend throughout ``shadow`` +and ``verify`` (15.2, 15.3). The managed corpus is built alongside it and becomes +visible only at ``promote``, which is one conditional write. That is also why +rollback moves no data: the legacy index was never touched, so returning to it is +an attribute ``REMOVE``. + +Re-ingest, never re-upload +-------------------------- +Source bytes are already at +``assistants/{assistant_id}/documents/{document_id}/{filename}``, so migration +hands Bedrock the S3 location it already has (15.4). No user is ever asked to +re-supply a document, and no bytes move. + +Convergence, not dual-write +--------------------------- +The existing upload path stays authoritative and keeps writing to legacy for the +whole migration (16.1, 16.6). Rather than writing to both engines — which doubles +the number of ways a write can half-fail — the worker snapshots the document-id +set, migrates it, then runs catch-up passes until a pass finds nothing new (16.2, +16.3). Same converge-on-quiet shape as the crawler's consecutive-miss rule. + +Each document's ``DOC#`` record is re-read immediately before it is ingested and +skipped if it has gone or is no longer ``complete`` (16.4, 16.5). Without that +re-read, a document deleted while the migration was working through a long PDF +queue would be resurrected in the managed corpus — the user deleted it, saw it +disappear, and it comes back on a different engine. + +Feature: managed-kb-migration +Requirements: 15.1–15.14, 16.1–16.6, 17.1–17.5, 12.9 +""" + +from __future__ import annotations + +import asyncio +import logging +import os +from dataclasses import dataclass, field +from datetime import timedelta +from typing import Any, Dict, List, Optional, Sequence, Set + +logger = logging.getLogger() +logger.setLevel(logging.INFO) + +#: Document status the migration carries across. Requirement 15.5, and the same +#: value the facade's status filter serves — a document that is not ``complete`` +#: is not retrievable on legacy either, so migrating it would create a difference +#: where the whole point is parity. +STATUS_COMPLETE = "complete" + +#: Requirement 15.11. Legacy vectors are preserved for at least this long after +#: promotion, which is the window in which rollback is a pointer flip. +RETAIN_DAYS = 30 + +#: How long a worker holds a knowledge base. Long enough to cover the slowest +#: single step observed (a PDF-heavy shadow pass), short enough that a crashed +#: worker's knowledge base is picked up again the same hour. Requirement 15.13. +LEASE_MINUTES = 15 + +#: Catch-up passes before the worker gives up waiting for quiet. A knowledge base +#: whose owner is actively uploading may never converge; stopping is correct — +#: the record stays in ``shadow``, the dispatcher brings it back, and the corpus +#: keeps serving from legacy in the meantime. +MAX_CATCHUP_PASSES = 5 + +#: Documents ingested per managed call. Server-enforced at 10 for MANAGED +#: knowledge bases; the user guide's 25 is wrong. Named here so the batching is +#: visible at this level rather than only inside the adapter. +INGEST_BATCH = 10 + +#: Seconds added to ``dueAt`` when a step defers itself. Not a retry backoff — the +#: step succeeded — so it only needs to be long enough that the dispatcher does not +#: spin. +STEP_DELAY_SECONDS = 30 + +#: Ceiling on the completed-document set stored on the record. A DynamoDB item is +#: capped at 400 KB and this set is the only unbounded thing on it. Production's +#: entire corpus is 1,692 ``DOC#`` records across *all* assistants, so no real +#: knowledge base comes close; past the cap the worker stops tracking and a resume +#: re-ingests, which is slow but not wrong — ``customDocumentIdentifier`` makes a +#: re-ingest a replace. +MAX_TRACKED_DOCUMENT_IDS = 5000 + +METRIC_STARTED = "KbMigrationStarted" +METRIC_PROMOTED = "KbMigrationPromoted" +METRIC_FAILED = "KbMigrationFailed" +METRIC_ROLLED_BACK = "KbMigrationRolledBack" +METRIC_DOCUMENTS_MIGRATED = "KbMigrationDocumentsMigrated" +METRIC_DOCUMENTS_SKIPPED = "KbMigrationDocumentsSkipped" +METRIC_LEASE_LOST = "KbMigrationLeaseLost" + + +class MigrationError(Exception): + """A migration step could not complete. Leaves the record where it was.""" + + +class LeaseLost(MigrationError): + """Another worker holds this knowledge base. Not an error condition.""" + + +class VerificationFailed(MigrationError): + """The managed corpus does not match the source manifest, or the canary + retrieval returned nothing. Sends the migration to ``failed``, which leaves the + knowledge base on legacy and fully usable (17.4).""" + + +@dataclass +class StepResult: + """What one invocation did. Returned so the handler can log and test on it.""" + + assistant_id: str + app_kb_id: str + from_state: Optional[str] + to_state: Optional[str] + documents_migrated: int = 0 + documents_skipped: int = 0 + catchup_passes: int = 0 + converged: bool = False + detail: str = "" + manifest_diff: List[str] = field(default_factory=list) + + def as_log_fields(self) -> Dict[str, Any]: + return { + "appKbId": self.app_kb_id, + "from": self.from_state, + "to": self.to_state, + "migrated": self.documents_migrated, + "skipped": self.documents_skipped, + "catchupPasses": self.catchup_passes, + "converged": self.converged, + "detail": self.detail, + } + + +# ── environment, read at call time ─────────────────────────────────────────── +def _now(): + from datetime import datetime, timezone + + return datetime.now(timezone.utc) + + +def _iso(moment) -> str: + return moment.strftime("%Y-%m-%dT%H:%M:%SZ") + + +def _now_iso() -> str: + return _iso(_now()) + + +def _documents_bucket() -> str: + bucket = os.environ.get("S3_ASSISTANTS_DOCUMENTS_BUCKET_NAME") + if not bucket: + raise MigrationError("S3_ASSISTANTS_DOCUMENTS_BUCKET_NAME is not set") + return bucket + + +def _retain_days() -> int: + raw = os.environ.get("KB_MIGRATION_RETAIN_DAYS") + try: + value = int(raw) if raw else RETAIN_DAYS + except ValueError: + return RETAIN_DAYS + # Requirement 15.11 says *at least* 30 days, so a smaller override is refused + # rather than honoured: shortening the rollback window is not a tuning knob. + return max(value, RETAIN_DAYS) + + +def _table(): + import boto3 + + return boto3.resource("dynamodb").Table(os.environ["DYNAMODB_ASSISTANTS_TABLE_NAME"]) + + +# ── document reads (raw table, no assistants import) ───────────────────────── +def list_document_items(assistant_id: str) -> List[Dict[str, Any]]: + """Every ``DOC#`` record under an assistant, paginated. + + Raw table access for the same reason ``kb_sync/records.py`` uses it: importing + ``apis.shared.assistants`` pulls in the embeddings stack at module scope and + this module ships in a size-constrained Lambda image. + """ + from boto3.dynamodb.conditions import Key + + table = _table() + items: List[Dict[str, Any]] = [] + kwargs: Dict[str, Any] = { + "KeyConditionExpression": Key("PK").eq(f"AST#{assistant_id}") + & Key("SK").begins_with("DOC#"), + } + while True: + response = table.query(**kwargs) + items.extend(response.get("Items") or []) + last = response.get("LastEvaluatedKey") + if not last: + return items + kwargs["ExclusiveStartKey"] = last + + +def get_document_item(assistant_id: str, document_id: str) -> Optional[Dict[str, Any]]: + response = _table().get_item( + Key={"PK": f"AST#{assistant_id}", "SK": f"DOC#{document_id}"} + ) + return response.get("Item") + + +def document_id_of(item: Dict[str, Any]) -> str: + sk = str(item.get("SK") or "") + return sk.split("#", 1)[1] if sk.startswith("DOC#") else "" + + +def is_complete(item: Optional[Dict[str, Any]]) -> bool: + return bool(item) and item.get("status") == STATUS_COMPLETE + + +def manifest_entry(item: Dict[str, Any]) -> str: + """One line of the source manifest: id plus a content identity. + + Requirement 15.6 forbids relying on document-count parity, and this is why the + manifest is a set of strings rather than a number. Count parity is satisfied by + a corpus with the right *number* of wrong documents — which is exactly what a + migration that raced an upload and a delete produces. + + The identity is the first available of ``contentHash``, ``etag`` or + ``updatedAt``. All three are already written by the existing pipeline; falling + through to ``updatedAt`` means a document with no hash still contributes a + changing value rather than a constant that always matches. + """ + document_id = document_id_of(item) + for key in ("contentHash", "etag", "generation", "updatedAt"): + value = item.get(key) + if value: + return f"{document_id}:{value}" + return f"{document_id}:no-identity" + + +def source_manifest(items: Sequence[Dict[str, Any]]) -> Set[str]: + return {manifest_entry(item) for item in items if is_complete(item)} + + +def s3_key_for(assistant_id: str, item: Dict[str, Any]) -> Optional[str]: + """The document's existing S3 key. Prefers the stored one. + + Reconstructed from ``filename`` only when the record has no ``s3Key``, because + the record is authoritative: a filename that was sanitised on upload would + reconstruct to a key that does not exist, and the ingest would fail per + document with an error naming the wrong cause. + """ + stored = item.get("s3Key") or item.get("s3_key") + if stored: + return str(stored) + filename = item.get("filename") + document_id = document_id_of(item) + if not filename or not document_id: + return None + return f"assistants/{assistant_id}/documents/{document_id}/{filename}" + + +def document_bytes(item: Dict[str, Any]) -> int: + for key in ("sizeBytes", "fileSize", "size"): + value = item.get(key) + if value is not None: + try: + return int(value) + except (TypeError, ValueError): + continue + return 0 + + +# ── lease ──────────────────────────────────────────────────────────────────── +async def take_lease(assistant_id: str, app_kb_id: str) -> str: + """Hold the knowledge base for :data:`LEASE_MINUTES`, or raise :class:`LeaseLost`. + + Requirement 15.13. Losing this is the ordinary outcome of two dispatcher ticks + overlapping, so it is logged at info and counted, not raised as a failure that + would move the record to ``failed`` and strand a perfectly healthy migration. + """ + from apis.shared.kb_backend import records as r + from apis.shared.kb_backend.metrics import emit_count + + now = _now() + lease_until = _iso(now + timedelta(minutes=LEASE_MINUTES)) + try: + await asyncio.to_thread( + r.acquire_lease, assistant_id, app_kb_id, lease_until, _iso(now) + ) + except Exception as exc: + emit_count(METRIC_LEASE_LOST) + raise LeaseLost( + f"another worker holds the lease on kb {app_kb_id}; leaving it alone: {exc}" + ) from exc + return lease_until + + +# ── ingestion of one snapshot ──────────────────────────────────────────────── +async def _ingest_documents( + assistant_id: str, + app_kb_id: str, + document_ids: Sequence[str], + backend, +) -> Dict[str, Any]: + """Re-ingest the named documents, re-reading each record first. + + Returns ``{"migrated": int, "skipped": int, "done": [ids]}``. ``done`` is the + documents genuinely handed to Bedrock, which is what gets persisted so a resume + can skip them — a count would not identify *which*. + + The re-read is Requirement 16.4 and it happens per document immediately before + that document is handed over, not once per batch: a PDF batch can take minutes, + and the deletion this guards against is most likely to land during exactly that + window. + """ + from apis.shared.kb_backend.protocol import DocumentSource + + migrated = 0 + skipped = 0 + done: List[str] = [] + batch: List[DocumentSource] = [] + + async def flush() -> None: + nonlocal batch, migrated + if not batch: + return + await backend.ingest_documents(app_kb_id, batch, batch_size=INGEST_BATCH) + migrated += len(batch) + done.extend(source.document_id for source in batch) + batch = [] + + for document_id in document_ids: + item = await asyncio.to_thread(get_document_item, assistant_id, document_id) + if not is_complete(item): + # Gone, or no longer complete. Requirement 16.5: not resurrected. + logger.info( + f"skipping document {document_id}: status=" + f"{(item or {}).get('status', 'NOT_FOUND')}" + ) + skipped += 1 + continue + + key = s3_key_for(assistant_id, item) + if not key: + logger.warning(f"skipping document {document_id}: no resolvable S3 key") + skipped += 1 + continue + + batch.append( + DocumentSource( + document_id=document_id, + filename=str(item.get("filename") or document_id), + s3_key=key, + metadata={"document_id": document_id, "filename": str(item.get("filename") or "")}, + ) + ) + if len(batch) >= INGEST_BATCH: + await flush() + + await flush() + return {"migrated": migrated, "skipped": skipped, "done": done} + + +# ── steps ──────────────────────────────────────────────────────────────────── +async def run_shadow( + assistant_id: str, + app_kb_id: str, + record: Dict[str, Any], + backend=None, +) -> StepResult: + """Provision, reserve the whole corpus, ingest the snapshot, then converge. + + Order matters and is not arbitrary: + + 1. **Reserve the whole snapshot first** (12.9). Migration is the largest + byte-adding operation in the system and the only one that runs unattended. + Reserving per document would let a migration run for an hour and stop + halfway, leaving a half-populated managed corpus and an owner over their cap + with no way back. + 2. **Provision.** Lazy by design, so the knowledge base may not exist yet. + 3. **Ingest the snapshot**, re-reading each record immediately before use. + 4. **Catch up until quiet** (16.2, 16.3), then move to ``verify``. + """ + from apis.shared.kb_backend import byte_cap, records as r + from apis.shared.kb_backend.metrics import emit_count + from apis.shared.kb_backend.provisioning import provision_managed_kb + + generation = int(record.get("migrationGeneration") or 0) + items = await asyncio.to_thread(list_document_items, assistant_id) + complete = [item for item in items if is_complete(item)] + + # Documents a previous invocation already ingested. Skipping them is what makes + # a resumed migration cost seconds rather than re-parsing a PDF corpus that can + # take over an hour. + done = already_migrated(record) + snapshot = [ + document_id_of(item) + for item in complete + if document_id_of(item) and document_id_of(item) not in done + ] + + total_bytes = sum(document_bytes(item) for item in complete) + if total_bytes and record.get("totalBytes") in (None, 0): + # Reserved once per migration, not once per resume: the accumulator is on + # the record, so a resumed run that reserved again would double-count its + # own corpus against the owner's cap and eventually refuse itself. + # Raises ByteCapExceeded, which the caller turns into `failed` — before + # anything has been provisioned or ingested. + await asyncio.to_thread( + byte_cap.reserve_snapshot, + assistant_id, + app_kb_id, + total_bytes, + byte_cap.per_owner_cap(bool(record.get("elevatedByteCap"))), + ) + + await provision_managed_kb( + assistant_id, + app_kb_id, + owner_user_id=str(record.get("ownerUserId") or ""), + ) + + backend = backend or _managed_backend() + emit_count(METRIC_STARTED) + + counts = await _ingest_documents(assistant_id, app_kb_id, snapshot, backend) + migrated_ids = set(snapshot) | done + + passes, converged, extra = await catch_up( + assistant_id, app_kb_id, migrated_ids, backend + ) + counts["migrated"] += extra["migrated"] + counts["skipped"] += extra["skipped"] + newly_done = list(counts["done"]) + list(extra["done"]) + + total_done = len(done) + counts["migrated"] + await _record_progress( + assistant_id, + app_kb_id, + migrated=total_done, + total=total_done, + skipped=counts["skipped"], + newly_done=newly_done, + ) + + if not converged: + # Still busy. Stay in `shadow`; the dispatcher brings this back, and the + # corpus keeps serving from legacy in the meantime. + await asyncio.to_thread( + r.set_migration_state, + assistant_id, + app_kb_id, + r.SHADOW, + generation, + _iso(_now() + timedelta(seconds=STEP_DELAY_SECONDS)), + [r.SHADOW], + ) + return StepResult( + assistant_id, + app_kb_id, + r.SHADOW, + r.SHADOW, + counts["migrated"], + counts["skipped"], + passes, + False, + "catch-up did not converge; staying in shadow", + ) + + await asyncio.to_thread( + r.set_migration_state, + assistant_id, + app_kb_id, + r.VERIFY, + generation, + _iso(_now() + timedelta(seconds=STEP_DELAY_SECONDS)), + [r.SHADOW], + ) + return StepResult( + assistant_id, + app_kb_id, + r.SHADOW, + r.VERIFY, + counts["migrated"], + counts["skipped"], + passes, + True, + ) + + +async def catch_up( + assistant_id: str, + app_kb_id: str, + already: Set[str], + backend, + max_passes: int = None, +) -> tuple: + """Ingest documents that appeared since the snapshot, until a pass finds none. + + Requirements 16.2, 16.3. Returns ``(passes, converged, counts)``. + + Converged means a pass found nothing new — not that a fixed number of passes + ran. The distinction matters because the number of passes needed depends on how + fast the owner is uploading, which is not something this code can know in + advance. ``max_passes`` bounds the invocation, and *not* converging is a normal + outcome that leaves the record in ``shadow``. + """ + limit = MAX_CATCHUP_PASSES if max_passes is None else max_passes + counts: Dict[str, Any] = {"migrated": 0, "skipped": 0, "done": []} + + for attempt in range(1, limit + 1): + items = await asyncio.to_thread(list_document_items, assistant_id) + pending = [ + document_id_of(item) + for item in items + if is_complete(item) and document_id_of(item) not in already + ] + if not pending: + logger.info(f"catch-up converged for kb {app_kb_id} after {attempt} pass(es)") + return attempt, True, counts + + logger.info(f"catch-up pass {attempt} for kb {app_kb_id}: {len(pending)} new document(s)") + pass_counts = await _ingest_documents(assistant_id, app_kb_id, pending, backend) + counts["migrated"] += pass_counts["migrated"] + counts["skipped"] += pass_counts["skipped"] + counts["done"].extend(pass_counts["done"]) + already.update(pending) + + return limit, False, counts + + +async def run_verify( + assistant_id: str, + app_kb_id: str, + record: Dict[str, Any], + backend=None, +) -> StepResult: + """Compare an exact manifest, then prove retrieval works. + + Requirements 15.6, 15.7. Two checks, and both are needed: + + * The **manifest** is a set of ``document_id:identity`` strings, not a count. + A count is satisfied by the right number of wrong documents. + * The **canary retrieval** proves the corpus is genuinely queryable. Bedrock + reporting a document ``INDEXED`` precedes it being retrievable by + 0.75–1.03 s, and a knowledge base can hold documents while returning nothing + — so "we ingested everything" and "retrieval works" are separate claims. + """ + from apis.shared.kb_backend import records as r + + generation = int(record.get("migrationGeneration") or 0) + backend = backend or _managed_backend() + + items = await asyncio.to_thread(list_document_items, assistant_id) + expected = source_manifest(items) + + complete = [item for item in items if is_complete(item)] + if not complete: + raise VerificationFailed( + f"kb {app_kb_id} has no complete documents to verify; there is nothing " + f"to promote" + ) + + canary_text = _canary_query(complete) + chunks = await backend.search(app_kb_id, canary_text, 5) + if not chunks: + raise VerificationFailed( + f"canary retrieval on kb {app_kb_id} returned nothing; the managed " + f"corpus is not queryable yet" + ) + + retrieved_ids = {chunk.document_id for chunk in chunks if chunk.document_id} + expected_ids = {document_id_of(item) for item in complete} + if not retrieved_ids & expected_ids: + raise VerificationFailed( + f"canary retrieval on kb {app_kb_id} returned only documents this " + f"assistant does not own: {sorted(retrieved_ids)}" + ) + + await asyncio.to_thread( + r.set_migration_state, + assistant_id, + app_kb_id, + r.PROMOTE, + generation, + _iso(_now() + timedelta(seconds=STEP_DELAY_SECONDS)), + [r.VERIFY], + ) + return StepResult( + assistant_id, + app_kb_id, + r.VERIFY, + r.PROMOTE, + detail=f"manifest of {len(expected)} document(s) verified; canary returned " + f"{len(chunks)} chunk(s)", + ) + + +def _canary_query(complete: Sequence[Dict[str, Any]]) -> str: + """A query built from the corpus's own filenames. + + Not a fixed string. A constant like "test" can legitimately match nothing in a + real corpus, which would make verification fail for healthy knowledge bases and + train whoever is watching to ignore it. + """ + names = [str(item.get("filename") or "") for item in complete[:3]] + text = " ".join(name.rsplit(".", 1)[0].replace("_", " ").replace("-", " ") for name in names) + return text.strip() or "summary" + + +async def run_promote( + assistant_id: str, + app_kb_id: str, + record: Dict[str, Any], +) -> StepResult: + """The cutover: one conditional write, then straight into ``retain``. + + Requirements 15.8, 15.9, 15.10. Everything that makes this safe lives in + ``records.promote_engine``'s condition — the state, the generation, and + ``migrationProgress.migrated == migrationProgress.total``, so a promotion + cannot happen on a knowledge base whose catch-up never converged. Two + concurrent workers issue the same write and DynamoDB picks one. + + The byte cap must already be enforced on this knowledge base (12.9): no + traffic is promoted to an unmetered corpus, so a record carrying no + ``totalBytes`` accumulator is refused here rather than discovered later. + """ + from apis.shared.kb_backend import records as r + from apis.shared.kb_backend.metrics import emit_count + + generation = int(record.get("migrationGeneration") or 0) + + if record.get("totalBytes") is None: + raise MigrationError( + f"refusing to promote kb {app_kb_id}: it has no totalBytes accumulator, " + f"so the byte cap is not being enforced on it (Requirement 12.9)" + ) + + # Resuming after a crash *between* the promotion and the state transition. The + # promotion write is guarded on `attribute_not_exists(retrievalEngine)`, so + # retrying it here would be refused — and treating that refusal as a failure + # would mark a migration that actually succeeded as `failed`, leaving a promoted + # knowledge base with no retention window and no path to `retain`. Found by the + # convergence property test, which crashed at exactly that transition. + already_promoted = record.get("retrievalEngine") == r.ENGINE_MANAGED + + if not already_promoted: + try: + await asyncio.to_thread( + r.promote_engine, assistant_id, app_kb_id, generation, _now_iso() + ) + emit_count(METRIC_PROMOTED) + except Exception: + # Re-read before deciding. The write may have been refused because + # somebody else promoted first, which is success, or because a guard + # genuinely failed, which is not. + fresh = await asyncio.to_thread(r.get_kb_record, assistant_id, app_kb_id) + if (fresh or {}).get("retrievalEngine") != r.ENGINE_MANAGED: + raise + logger.info( + f"kb {app_kb_id} was already promoted by another attempt; " + f"continuing to retain rather than failing" + ) + already_promoted = True + + retain_until = _iso(_now() + timedelta(days=_retain_days())) + await asyncio.to_thread( + _set_retain_until, assistant_id, app_kb_id, retain_until + ) + await asyncio.to_thread( + r.set_migration_state, + assistant_id, + app_kb_id, + r.RETAIN, + generation, + None, + [r.PROMOTE], + ) + return StepResult( + assistant_id, + app_kb_id, + r.PROMOTE, + r.RETAIN, + detail=( + f"{'already promoted; ' if already_promoted else ''}legacy vectors " + f"retained until {retain_until}" + ), + ) + + +def _set_retain_until(assistant_id: str, app_kb_id: str, retain_until: str) -> None: + """Stamp the rollback deadline. Unconditional, and deliberately so. + + The promotion write immediately before this one is the guarded one. If this + write were also guarded and lost, the record would be promoted with no + ``retainUntil`` — which reads as "no rollback window" to anything that checks + it. Writing the later date twice is harmless; writing it never is not. + """ + import boto3 + + boto3.resource("dynamodb").Table(os.environ["DYNAMODB_ASSISTANTS_TABLE_NAME"]).update_item( + Key={"PK": f"AST#{assistant_id}", "SK": f"KB#{app_kb_id}"}, + UpdateExpression="SET retainUntil = :until", + ExpressionAttributeValues={":until": retain_until}, + ) + + +async def rollback(assistant_id: str, app_kb_id: str) -> StepResult: + """Return a promoted knowledge base to legacy. Moves no data. + + Requirement 17. The legacy index was never mutated — that is what ``shadow`` + building alongside it bought — so rollback is one attribute ``REMOVE`` plus a + timestamp. It is available for the whole ``retain`` window because that window + is exactly the promise not to reclaim the legacy vectors. + + Note this does **not** delete the managed knowledge base. A rolled-back corpus + that still exists costs storage but can be re-promoted without a second + migration; deleting it here would turn a reversible decision into an + irreversible one at the moment somebody is least sure. + """ + from apis.shared.kb_backend import records as r + from apis.shared.kb_backend.metrics import emit_count + + await asyncio.to_thread(r.rollback_engine, assistant_id, app_kb_id, _now_iso()) + emit_count(METRIC_ROLLED_BACK) + return StepResult( + assistant_id, + app_kb_id, + r.RETAIN, + r.RETAIN, + detail="rolled back to the legacy engine; no data moved", + ) + + +async def _record_progress( + assistant_id: str, + app_kb_id: str, + *, + migrated: int, + total: int, + skipped: int, + newly_done: Optional[Sequence[str]] = None, +) -> None: + """Write ``migrationProgress``, which the promotion condition reads. + + ``total`` is a DynamoDB reserved keyword, so both progress paths are aliased. + Unaliased, the write is rejected outright with a ``ValidationException`` — loud, + but only because it never validates at all. + + ``newly_done`` is ``ADD``ed to the ``migratedDocIds`` string set rather than + written into the progress map. Two reasons, and both are the difference between + a resumed migration costing seconds and costing an hour: + + * **``ADD`` is additive**, so a crash between batches loses only the batch in + flight. A read-modify-write of a list would lose everything since the last + read, and would also let two workers clobber each other. + * **It is a separate attribute** from ``migrationProgress``, which this function + overwrites wholesale. Keeping the completed-document set inside a map that + gets replaced is how a resume silently re-ingests a corpus it had already + finished — found by the convergence property test, which counted a document + ingested twice across a crash and a retry. + """ + from decimal import Decimal + + expression = "SET #progress = :progress" + names = {"#progress": "migrationProgress"} + values: Dict[str, Any] = { + ":progress": { + "migrated": Decimal(migrated), + "total": Decimal(total), + "skipped": Decimal(skipped), + "updatedAt": _now_iso(), + } + } + + ids = [document_id for document_id in (newly_done or []) if document_id] + if ids and len(ids) <= MAX_TRACKED_DOCUMENT_IDS: + # DynamoDB string sets cannot be empty, hence the guard above. + expression += " ADD #done :done" + names["#done"] = "migratedDocIds" + values[":done"] = set(ids) + elif ids: + logger.warning( + f"kb {app_kb_id}: {len(ids)} document ids exceeds the tracking cap of " + f"{MAX_TRACKED_DOCUMENT_IDS}; a resumed migration will re-ingest, which " + f"is safe but slow (customDocumentIdentifier makes re-ingest a replace)" + ) + + _table().update_item( + Key={"PK": f"AST#{assistant_id}", "SK": f"KB#{app_kb_id}"}, + UpdateExpression=expression, + ExpressionAttributeNames=names, + ExpressionAttributeValues=values, + ) + + +def already_migrated(record: Dict[str, Any]) -> Set[str]: + """Documents a previous invocation already ingested. + + Read from the ``migratedDocIds`` string set. Empty for a record that has never + ingested anything, which is also what a corpus past the tracking cap looks + like — and that degradation is safe: re-ingesting a document replaces it, + because ``customDocumentIdentifier`` is the platform document id. + """ + stored = record.get("migratedDocIds") + if not stored: + return set() + try: + return {str(document_id) for document_id in stored} + except TypeError: + logger.warning(f"migratedDocIds is not iterable on this record: {stored!r}") + return set() + + +def _managed_backend(): + from apis.shared.kb_backend.managed_backend import ManagedKbBackend + + return ManagedKbBackend(bucket=_documents_bucket()) + + +# ── one invocation ─────────────────────────────────────────────────────────── +async def run_step( + assistant_id: str, + app_kb_id: Optional[str] = None, + backend=None, +) -> StepResult: + """Take the lease and execute the one step this record's state calls for. + + Dispatches on the *record's* state, never on the invocation event's. The event + carries a state for logging, but trusting it would let a hand-crafted invocation + promote a knowledge base that never verified — the same class of bypass that + let an event field arm the reconciler. + """ + from apis.shared.kb_backend import byte_cap, records as r + from apis.shared.kb_backend.metrics import emit_count + + app_kb_id = app_kb_id or assistant_id + + record = await asyncio.to_thread(r.get_kb_record, assistant_id, app_kb_id) + if not record: + raise MigrationError(f"no KB_Record for {assistant_id}/{app_kb_id}") + + state = record.get("migrationState") + if state not in r.WORK_ELIGIBLE_STATES: + # Terminal, or never enrolled. Not an error: the dispatcher reads an index + # that is eventually consistent, so a record finished a moment ago can + # still be handed over once. + return StepResult( + assistant_id, app_kb_id, state, state, detail="not work-eligible; nothing to do" + ) + + generation = int(record.get("migrationGeneration") or 0) + + try: + # Inside the try, deliberately. A ``LeaseLost`` must reach the caller as + # itself — losing a lease is two dispatcher ticks overlapping, not a broken + # migration — and the ``except LeaseLost: raise`` below is what guarantees + # that even once a step starts taking sub-leases of its own. Outside the try + # the clause would be unreachable, which is how a guard becomes decoration. + await take_lease(assistant_id, app_kb_id) + + if state == r.SHADOW: + result = await run_shadow(assistant_id, app_kb_id, record, backend) + elif state == r.VERIFY: + result = await run_verify(assistant_id, app_kb_id, record, backend) + else: + result = await run_promote(assistant_id, app_kb_id, record) + except LeaseLost: + raise + except (VerificationFailed, byte_cap.ByteCapExceeded) as exc: + # Expected failure modes. The knowledge base stays on legacy and stays + # usable (17.4); `failed` is terminal and removes the work keys. + await _fail(assistant_id, app_kb_id, generation, str(exc)) + emit_count(METRIC_FAILED) + return StepResult( + assistant_id, app_kb_id, state, r.MIGRATION_FAILED, detail=str(exc) + ) + except Exception as exc: + # Unexpected. Also terminal, for the same reason: an unbounded retry on an + # unknown fault is how a migration loop bills for a week. + logger.error(f"migration step failed for kb {app_kb_id}: {exc}", exc_info=True) + await _fail(assistant_id, app_kb_id, generation, f"{type(exc).__name__}: {exc}") + emit_count(METRIC_FAILED) + return StepResult( + assistant_id, app_kb_id, state, r.MIGRATION_FAILED, detail=str(exc) + ) + + logger.info(f"migration step: {result.as_log_fields()}") + return result + + +async def _fail(assistant_id: str, app_kb_id: str, generation: int, reason: str) -> None: + from apis.shared.kb_backend import records as r + + try: + await asyncio.to_thread( + r.set_migration_state, + assistant_id, + app_kb_id, + r.MIGRATION_FAILED, + generation, + None, + None, + reason[:1000], + ) + except Exception as exc: + # Nothing further to do: the record keeps its work keys and the dispatcher + # will bring it back, which is the safe direction — a knowledge base stuck + # in `shadow` still serves from legacy. + logger.error(f"could not record migration failure for kb {app_kb_id}: {exc}") + + +def lambda_handler(event, context): + """Async-invoked by the dispatcher. + + Reads only the two identifiers from the event. Everything that decides what + happens — the state, the generation, the flags — comes from the record and the + environment. + """ + assistant_id = (event or {}).get("assistantId") + app_kb_id = (event or {}).get("appKbId") + if not assistant_id: + raise MigrationError("event carries no assistantId") + + try: + result = asyncio.run(run_step(assistant_id, app_kb_id)) + except LeaseLost as exc: + logger.info(str(exc)) + return {"statusCode": 200, "body": {"leaseLost": True}} + + return {"statusCode": 200, "body": result.as_log_fields()} diff --git a/backend/src/apis/app_api/kb_upgrade/__init__.py b/backend/src/apis/app_api/kb_upgrade/__init__.py new file mode 100644 index 000000000..40eaeef10 --- /dev/null +++ b/backend/src/apis/app_api/kb_upgrade/__init__.py @@ -0,0 +1,12 @@ +"""The owner-facing knowledge base upgrade surface (Requirements 21, 23). + +Deliberately a **separate package from** ``apis.app_api.kb_migration``. That +package holds the four Lambda handlers, which share one size-constrained image; +this one is HTTP-only and imports ``apis.shared.assistants`` for the permission +model, which pulls the embeddings stack at module scope. Putting the two in the +same package invites a handler import that blows the image-size budget — the +failure ``tests/architecture/test_kb_backend_boundary.py`` exists to prevent. + +Nothing here writes ``retrievalEngine``. Enrolment only moves a record into +``shadow``; the worker promotes, and only after verification. +""" diff --git a/backend/src/apis/app_api/kb_upgrade/models.py b/backend/src/apis/app_api/kb_upgrade/models.py new file mode 100644 index 000000000..5fffa8cfe --- /dev/null +++ b/backend/src/apis/app_api/kb_upgrade/models.py @@ -0,0 +1,95 @@ +"""Wire models for the knowledge base upgrade surface. + +Field names are camelCase on the wire (``populate_by_name`` + aliases), matching +every other app_api surface the Angular client consumes. + +The word "vector" appears nowhere in any user-facing string in this module, per +Requirement 23.6. It is fine in comments; it is not fine in ``message``. +""" + +from typing import List, Literal, Optional + +from pydantic import BaseModel, ConfigDict, Field + +#: The derived, UI-facing phase. Deliberately NOT the record's ``migrationState``: +#: the client should not have to know that ``shadow``, ``verify`` and ``promote`` +#: are all "working on it", nor that absence means legacy. +#: +#: ``none`` is the state that renders nothing at all (Requirement 23.1). +UpgradePhase = Literal["none", "available", "in_progress", "succeeded", "failed"] + +#: Why a document will not be carried across. Requirement 21.4 requires an +#: unsupported format to be distinguishable from a processing failure, because the +#: user's next action differs: convert and re-upload, versus just retry. +DocumentIssueKind = Literal[ + "unsupported_format", + "processing_failure", + "still_processing", + "being_removed", +] + + +class UpgradeProgress(BaseModel): + """Non-blocking progress for the ``in_progress`` phase (Requirement 23.3).""" + + model_config = ConfigDict(populate_by_name=True) + + completed: int = Field(0, description="Documents carried across so far") + total: int = Field(0, description="Documents in the snapshot being carried") + skipped: int = Field(0, description="Documents the snapshot could not include") + + +class DocumentNotCarried(BaseModel): + """A document the upgrade will not carry across (Requirement 21.1). + + Surfaced *before* the user commits, not after, so the choice to retry or + accept the loss is made with the facts in hand. + """ + + model_config = ConfigDict(populate_by_name=True) + + document_id: str = Field(..., alias="documentId") + filename: str + status: str = Field(..., description="The stored processing status, verbatim") + kind: DocumentIssueKind + message: str = Field( + ..., + description="Plain-language explanation, safe to render directly", + ) + retryable: bool = Field( + ..., + description="Whether re-processing this document could succeed as-is", + ) + + +class UpgradeStatusResponse(BaseModel): + """Everything the card needs to render, in one round trip.""" + + model_config = ConfigDict(populate_by_name=True) + + phase: UpgradePhase + #: True only for an owner/editor looking at an ``available`` knowledge base. + #: The client hides the control on this alone; the server re-checks on write, + #: so a client that ignores it gains nothing (Requirement 23.7). + can_upgrade: bool = Field(False, alias="canUpgrade") + progress: Optional[UpgradeProgress] = None + #: Plain-language failure reason for the ``failed`` phase (Requirement 23.5). + reason: Optional[str] = None + #: Whether the one-time success notice is still owed (Requirement 23.4). + #: Never sticky: dismissing it sets a timestamp and this goes false forever. + notice_pending: bool = Field(False, alias="noticePending") + documents_not_carried: List[DocumentNotCarried] = Field( + default_factory=list, alias="documentsNotCarried" + ) + + +class EnrollResponse(BaseModel): + """Result of an enrol or retry.""" + + model_config = ConfigDict(populate_by_name=True) + + phase: UpgradePhase + #: True when this call is what started the upgrade, false when it found one + #: already running. Both are successes — a double-click is not an error. + started: bool + message: str diff --git a/backend/src/apis/app_api/kb_upgrade/routes.py b/backend/src/apis/app_api/kb_upgrade/routes.py new file mode 100644 index 000000000..c5dd33822 --- /dev/null +++ b/backend/src/apis/app_api/kb_upgrade/routes.py @@ -0,0 +1,144 @@ +"""HTTP surface for the owner-facing knowledge base upgrade (Requirements 21, 23). + +Four endpoints, all under the assistant that owns the knowledge base: + +* ``GET /assistants/{id}/knowledge-base/upgrade`` — what to render +* ``POST /assistants/{id}/knowledge-base/upgrade`` — opt in +* ``POST /assistants/{id}/knowledge-base/upgrade/retry`` — after a failure +* ``POST /assistants/{id}/knowledge-base/upgrade/notice`` — dismiss the notice + +Permission is resolved the same way the documents surface resolves it, through +``resolve_assistant_permission``. The read is allowed for any resolvable +permission and reports ``canUpgrade: false`` to a viewer; the three writes +require owner or editor. Requirement 23.7's "viewers never see the control" is +therefore enforced on the server, with the client's hiding as presentation only. +""" + +import logging + +from fastapi import APIRouter, Depends, HTTPException, status + +from apis.app_api.kb_upgrade.models import EnrollResponse, UpgradeStatusResponse +from apis.app_api.kb_upgrade.service import ( + UpgradeUnavailable, + dismiss_notice, + enroll, + get_upgrade_status, + retry, +) +from apis.shared.assistants.service import resolve_assistant_permission +from apis.shared.auth import User, get_current_user_from_session + +logger = logging.getLogger(__name__) + +router = APIRouter( + prefix="/assistants/{assistant_id}/knowledge-base/upgrade", + tags=["knowledge-base-upgrade"], +) + +_EDIT_PERMISSIONS = ("owner", "editor") + + +async def _resolve(assistant_id: str, current_user: User): + """Return ``(assistant, permission)`` or raise 404. + + A permission of ``None`` from a resolvable assistant means the user has no + access at all, which is reported as 404 rather than 403 so the endpoint does + not confirm the existence of an assistant the caller cannot see. + """ + assistant, permission = await resolve_assistant_permission( + assistant_id=assistant_id, + user_id=current_user.user_id, + user_email=current_user.email, + ) + if not assistant or not permission: + raise HTTPException( + status_code=status.HTTP_404_NOT_FOUND, + detail=f"Assistant not found: {assistant_id}", + ) + return assistant, permission + + +async def _require_edit_permission(assistant_id: str, current_user: User): + """Resolve and require owner|editor. Returns ``(assistant, permission)``.""" + assistant, permission = await _resolve(assistant_id, current_user) + if permission not in _EDIT_PERMISSIONS: + raise HTTPException( + status_code=status.HTTP_403_FORBIDDEN, + detail="You do not have permission to upgrade this knowledge base", + ) + return assistant, permission + + +@router.get("", response_model=UpgradeStatusResponse) +async def read_upgrade_status( + assistant_id: str, + current_user: User = Depends(get_current_user_from_session), +) -> UpgradeStatusResponse: + """What the upgrade card should render, or ``phase: "none"`` for nothing.""" + _, permission = await _resolve(assistant_id, current_user) + try: + return await get_upgrade_status( + assistant_id, can_edit=permission in _EDIT_PERMISSIONS + ) + except Exception as exc: # noqa: BLE001 — see below + # Fail to "nothing to show" rather than 500. This endpoint is decoration + # on a working page: a knowledge base that cannot be described still + # serves retrieval, and taking the whole documents section down over an + # unavailable upgrade card would be a strictly worse outcome. Logged at + # error so the failure is not silent. + logger.error( + f"kb {assistant_id}: could not derive upgrade status: {exc}", + exc_info=True, + ) + return UpgradeStatusResponse(phase="none", canUpgrade=False) + + +@router.post("", response_model=EnrollResponse, status_code=status.HTTP_202_ACCEPTED) +async def start_upgrade( + assistant_id: str, + current_user: User = Depends(get_current_user_from_session), +) -> EnrollResponse: + """Opt this knowledge base into the upgrade (Requirement 23.2, 23.8). + + 202 rather than 200: enrolment queues work for the migration worker and + returns before any of it has happened. + """ + assistant, _ = await _require_edit_permission(assistant_id, current_user) + try: + return await enroll( + assistant_id, + owner_user_id=assistant.owner_id, + visibility=str(getattr(assistant, "visibility", "PRIVATE") or "PRIVATE"), + ) + except UpgradeUnavailable as exc: + raise HTTPException( + status_code=status.HTTP_409_CONFLICT, detail=str(exc) + ) from exc + + +@router.post( + "/retry", response_model=EnrollResponse, status_code=status.HTTP_202_ACCEPTED +) +async def retry_upgrade( + assistant_id: str, + current_user: User = Depends(get_current_user_from_session), +) -> EnrollResponse: + """Restart a failed upgrade on a fresh generation (Requirement 23.5).""" + assistant, _ = await _require_edit_permission(assistant_id, current_user) + try: + return await retry(assistant_id, owner_user_id=assistant.owner_id) + except UpgradeUnavailable as exc: + raise HTTPException( + status_code=status.HTTP_409_CONFLICT, detail=str(exc) + ) from exc + + +@router.post("/notice", status_code=status.HTTP_204_NO_CONTENT) +async def dismiss_upgrade_notice( + assistant_id: str, + current_user: User = Depends(get_current_user_from_session), +) -> None: + """Dismiss the one-time post-upgrade notice (Requirement 23.4).""" + await _require_edit_permission(assistant_id, current_user) + await dismiss_notice(assistant_id) diff --git a/backend/src/apis/app_api/kb_upgrade/service.py b/backend/src/apis/app_api/kb_upgrade/service.py new file mode 100644 index 000000000..2bb9de264 --- /dev/null +++ b/backend/src/apis/app_api/kb_upgrade/service.py @@ -0,0 +1,507 @@ +"""Enrolment and status for the owner-facing knowledge base upgrade. + +Three things happen here and nothing else: derive what the card should show, +move a record into ``shadow``, and dismiss the one-time success notice. Promotion +belongs to the worker, after verification — this module never writes +``retrievalEngine``, so no HTTP request can put a knowledge base on the managed +backend without the corpus having been carried across and checked first. + +Enrolment is deliberately two conditional writes rather than one put: + +1. ``create_provisioning`` — guarded on ``attribute_not_exists(PK)``. +2. ``set_migration_state(SHADOW, ...)`` — guarded on the generation. + +A single ``put_item`` with ``migrationState="shadow"`` baked in would look +simpler and would be **wrong**: ``KbRecord.to_item`` does not write the +``GSI7_PK``/``GSI7_SK`` work keys, which only ``set_migration_state`` maintains. +The record would exist, claim to be migrating, and be invisible to the +dispatcher's sparse-index sweep forever. +""" + +from __future__ import annotations + +import logging +import os +from datetime import datetime, timedelta, timezone +from typing import Any, Dict, List, Optional, Tuple + +from apis.app_api.kb_upgrade.models import ( + DocumentNotCarried, + EnrollResponse, + UpgradeProgress, + UpgradeStatusResponse, +) + +logger = logging.getLogger(__name__) + + +class UpgradeUnavailable(Exception): + """The upgrade cannot be offered right now. Carries user-safe copy.""" + + +#: Gate on the same flag the dispatcher reads. If the worker cannot run, offering +#: the upgrade would park a record in ``shadow`` that nothing ever picks up — a +#: spinner with no engine behind it. Requirement 23.1 says show nothing when no +#: action is available, and "available" has to mean actionable. +FLAG_MIGRATION_ENABLED = "MANAGED_KB_MIGRATION_ENABLED" + +#: Affirmative spellings, matching ``dispatcher._TRUTHY`` exactly. An allow-list +#: rather than truthiness, because the value being designed around is present but +#: empty: ``bool("")`` is right by luck and ``bool("false")`` is not. +_TRUTHY = frozenset({"1", "true", "yes", "on", "enabled"}) + +#: How soon the dispatcher may pick up a freshly enrolled record. Now, not later: +#: the user just asked for it, and the dispatcher is already rate-bounded. +_DUE_IMMEDIATELY = timedelta(0) + +STATUS_COMPLETE = "complete" + + +def migration_enabled() -> bool: + """Whether the upgrade may be offered at all. + + Read at call time, never bound as a default argument — a module-level default + is captured at import and makes the flag unpatchable, which already cost this + feature a 33-second test that ignored its own override. + """ + return (os.environ.get(FLAG_MIGRATION_ENABLED) or "").strip().lower() in _TRUTHY + + +def _now() -> datetime: + return datetime.now(timezone.utc) + + +def _iso(moment: datetime) -> str: + return moment.isoformat().replace("+00:00", "Z") + + +# ── Document classification (Requirement 21) ───────────────────────────────── +def _extension_of(filename: str) -> str: + _, _, tail = str(filename or "").rpartition(".") + return f".{tail.lower()}" if tail else "" + + +def _supported_extensions() -> frozenset: + """The ingestion pipeline's own extension set, read from its module. + + Imported here rather than copied. A copied list is exactly the drift that + produced the tag-contract defect: three files agreeing only because they all + fell back to the same hardcoded default. ``docling_processor``'s module scope + is stdlib-only, so this costs nothing at request time. + """ + from apis.app_api.documents.ingestion.processors.docling_processor import ( + DOCLING_SUPPORTED_EXTENSIONS, + ) + + return frozenset(DOCLING_SUPPORTED_EXTENSIONS) + + +def classify_document(item: Dict[str, Any]) -> Optional[DocumentNotCarried]: + """Describe why a document will not be carried across, or ``None`` if it will. + + Requirement 21.4 turns on the ``unsupported_format`` / ``processing_failure`` + split: the two demand different actions from the user. Telling someone to + "retry" a ``.pages`` file wastes a minute and teaches them the retry button + does not work. + + Note ``deleting`` is reported rather than hidden. The ordinary document list + filters that status out as soft-deleted, which is right there and wrong here: + 101 of the 200 affected production records are stuck in it, and a user who is + never shown them cannot tell that they are stuck. + """ + status = str(item.get("status") or "").strip() + if status == STATUS_COMPLETE: + return None + + document_id = str(item.get("documentId") or "") + if not document_id: + sk = str(item.get("SK") or "") + document_id = sk.split("#", 1)[1] if sk.startswith("DOC#") else "" + filename = str(item.get("filename") or "(unnamed file)") + extension = _extension_of(filename) + unsupported = bool(extension) and extension not in _supported_extensions() + + if status == "failed" and unsupported: + return DocumentNotCarried( + documentId=document_id, + filename=filename, + status=status, + kind="unsupported_format", + message=( + f"This platform cannot read {extension} files, so this document was " + "never added to your knowledge base. Save it as a PDF or Word " + "document and upload it again." + ), + retryable=False, + ) + if status == "failed": + stored = str(item.get("errorMessage") or "").strip() + detail = f" The reason given was: {stored}" if stored else "" + return DocumentNotCarried( + documentId=document_id, + filename=filename, + status=status, + kind="processing_failure", + message=( + "This document could not be processed, so it is not in your " + f"knowledge base and the upgrade cannot carry it across.{detail}" + ), + retryable=True, + ) + if status == "deleting": + return DocumentNotCarried( + documentId=document_id, + filename=filename, + status=status, + kind="being_removed", + message=( + "This document is part-way through being removed. It will not be " + "carried across. If you still want it, upload it again once the " + "removal finishes." + ), + retryable=False, + ) + return DocumentNotCarried( + documentId=document_id, + filename=filename, + status=status, + kind="still_processing", + message=( + "This document is still being processed. Documents that are not " + "finished when the upgrade starts will not be carried across." + ), + retryable=True, + ) + + +def _document_items(assistant_id: str) -> List[Dict[str, Any]]: + """Every ``DOC#`` item under an assistant, unfiltered. + + Raw query rather than ``list_assistant_documents``, on purpose and for two + reasons: that function drops ``deleting`` documents, which are the single + largest group Requirement 21 exists to surface, and it auto-fails stale ones + as a side effect — a write triggered by rendering a card. + """ + import boto3 + from boto3.dynamodb.conditions import Key + + table_name = os.environ.get("DYNAMODB_ASSISTANTS_TABLE_NAME") + if not table_name: + # Fail closed and loudly enough to see, but do not take the card down: a + # misconfigured table name must not make an upgradeable KB look clean. + raise RuntimeError("DYNAMODB_ASSISTANTS_TABLE_NAME is not set") + + table = boto3.resource("dynamodb").Table(table_name) + items: List[Dict[str, Any]] = [] + kwargs: Dict[str, Any] = { + "KeyConditionExpression": Key("PK").eq(f"AST#{assistant_id}") + & Key("SK").begins_with("DOC#"), + } + while True: + response = table.query(**kwargs) + items.extend(response.get("Items") or []) + last = response.get("LastEvaluatedKey") + if not last: + return items + kwargs["ExclusiveStartKey"] = last + + +def _partition_documents( + items: List[Dict[str, Any]], +) -> Tuple[int, List[DocumentNotCarried]]: + """Split into (count carried, descriptions of those not carried).""" + carried = 0 + stranded: List[DocumentNotCarried] = [] + for item in items: + issue = classify_document(item) + if issue is None: + carried += 1 + else: + stranded.append(issue) + return carried, stranded + + +# ── Status ─────────────────────────────────────────────────────────────────── +def _progress_of(record: Dict[str, Any]) -> Optional[UpgradeProgress]: + stored = record.get("migrationProgress") or {} + if not stored: + return None + return UpgradeProgress( + completed=int(stored.get("migrated") or 0), + total=int(stored.get("total") or 0), + skipped=int(stored.get("skipped") or 0), + ) + + +#: Failure reasons in the user's language. The stored ``migrationError`` is +#: written for an operator; rendering it raw is how a user ends up reading +#: "ByteCapExceeded". +_FAILURE_COPY = { + "ByteCapExceeded": ( + "Your knowledge base is larger than the current upgrade size limit, so " + "the upgrade stopped before changing anything." + ), + "VerificationFailed": ( + "The upgraded copy did not return the same results as your current one, " + "so it was discarded rather than switched over." + ), +} + +_FAILURE_FALLBACK = ( + "Something went wrong part-way through the upgrade, so it was stopped and " + "nothing was changed." +) + + +def _failure_reason(record: Dict[str, Any]) -> str: + stored = str(record.get("migrationError") or "") + for token, copy in _FAILURE_COPY.items(): + if token in stored: + return copy + return _FAILURE_FALLBACK + + +async def get_upgrade_status( + assistant_id: str, + *, + can_edit: bool, +) -> UpgradeStatusResponse: + """Derive everything the card renders. + + ``can_edit`` only ever *removes* the control (Requirement 23.7). A viewer + still gets an honest phase — they may legitimately see that an upgrade is + running — but never ``canUpgrade``. + """ + import asyncio + + from apis.shared.kb_backend import records as r + + record = await asyncio.to_thread(r.get_kb_record, assistant_id, assistant_id) + state = str((record or {}).get("migrationState") or "") + + if record and r.resolve_engine(record) == r.ENGINE_MANAGED: + # Already upgraded. The only thing owed is the one-time notice, and only + # until it is dismissed — never a permanent badge (Requirement 23.4). + pending = not record.get("upgradeNoticeDismissedAt") + return UpgradeStatusResponse( + phase="succeeded", + canUpgrade=False, + noticePending=bool(pending and can_edit), + progress=_progress_of(record), + ) + + if state in (r.SHADOW, r.VERIFY, r.PROMOTE): + return UpgradeStatusResponse( + phase="in_progress", + canUpgrade=False, + progress=_progress_of(record or {}), + ) + + if state == r.MIGRATION_FAILED: + # Still on the legacy backend, which keeps working. Retry is offered to + # editors; the phase itself is not hidden (Requirement 23.5). + return UpgradeStatusResponse( + phase="failed", + canUpgrade=can_edit, + reason=_failure_reason(record or {}), + progress=_progress_of(record or {}), + ) + + if not (can_edit and migration_enabled()): + return UpgradeStatusResponse(phase="none", canUpgrade=False) + + items = await asyncio.to_thread(_document_items, assistant_id) + if not items: + # An empty knowledge base has nothing to carry across, so there is no + # action to take and therefore nothing to show (Requirement 23.1). + return UpgradeStatusResponse(phase="none", canUpgrade=False) + + carried, stranded = _partition_documents(items) + if not carried: + # Every document is already stranded. Offering an upgrade that would + # carry nothing is worse than useless, but the stranded list is exactly + # what this owner needs to see (Requirement 21.3). + return UpgradeStatusResponse( + phase="none", + canUpgrade=False, + documentsNotCarried=stranded, + ) + + return UpgradeStatusResponse( + phase="available", + canUpgrade=True, + progress=UpgradeProgress(completed=0, total=carried, skipped=len(stranded)), + documentsNotCarried=stranded, + ) + + +# ── Enrolment ──────────────────────────────────────────────────────────────── +async def enroll( + assistant_id: str, + *, + owner_user_id: str, + visibility: str = "PRIVATE", +) -> EnrollResponse: + """Move this knowledge base into ``shadow``, or report one already running. + + Idempotent by construction: both writes are conditional, so a double-click + produces one migration and one "already running" answer rather than two + provisioning sagas racing over the same corpus. + """ + import asyncio + + from apis.shared.kb_backend import records as r + + if not migration_enabled(): + raise UpgradeUnavailable( + "Upgrades are not being accepted at the moment. Nothing has changed." + ) + + record = await asyncio.to_thread(r.get_kb_record, assistant_id, assistant_id) + + if record and r.resolve_engine(record) == r.ENGINE_MANAGED: + return EnrollResponse( + phase="succeeded", + started=False, + message="This knowledge base has already been upgraded.", + ) + + state = str((record or {}).get("migrationState") or "") + if state in (r.SHADOW, r.VERIFY, r.PROMOTE): + return EnrollResponse( + phase="in_progress", + started=False, + message="The upgrade is already running.", + ) + + generation = int((record or {}).get("migrationGeneration") or 0) + + if record is None: + # Zero-backfill: legacy knowledge bases have no KB record at all, so + # enrolment is where the record first comes into existence. + fresh = r.KbRecord( + app_kb_id=assistant_id, + owner_user_id=owner_user_id, + visibility=visibility, + provisioning_state=r.PROVISIONING, + ) + try: + await asyncio.to_thread(r.create_provisioning, assistant_id, fresh) + except r.TransitionLost: + # Another request created it between our read and our write. Not an + # error: fall through and let the state transition arbitrate. + logger.info( + f"kb {assistant_id}: record created concurrently during enrolment" + ) + record = await asyncio.to_thread(r.get_kb_record, assistant_id, assistant_id) + generation = int((record or {}).get("migrationGeneration") or 0) + + due_at = _iso(_now() + _DUE_IMMEDIATELY) + try: + await asyncio.to_thread( + r.set_migration_state, + assistant_id, + assistant_id, + r.SHADOW, + generation, + due_at=due_at, + ) + except r.TransitionLost: + return EnrollResponse( + phase="in_progress", + started=False, + message="The upgrade is already running.", + ) + + logger.info(f"kb {assistant_id}: enrolled into shadow at generation {generation}") + return EnrollResponse( + phase="in_progress", + started=True, + message=( + "Upgrade started. Your knowledge base keeps working while it runs, " + "and you can leave this page." + ), + ) + + +async def retry(assistant_id: str, *, owner_user_id: str) -> EnrollResponse: + """Re-enter ``shadow`` from ``failed``, on a fresh generation. + + The generation bump is what makes the retry safe: every conditional write + belonging to the abandoned attempt is guarded on the old generation, so a + straggler worker from the failed run cannot land a write on the new one. + """ + import asyncio + + from apis.shared.kb_backend import records as r + + if not migration_enabled(): + raise UpgradeUnavailable( + "Upgrades are not being accepted at the moment. Nothing has changed." + ) + + record = await asyncio.to_thread(r.get_kb_record, assistant_id, assistant_id) + if record is None: + return await enroll(assistant_id, owner_user_id=owner_user_id) + + state = str(record.get("migrationState") or "") + if state != r.MIGRATION_FAILED: + # Nothing to retry. Report the truth rather than starting a second run. + if state in (r.SHADOW, r.VERIFY, r.PROMOTE): + return EnrollResponse( + phase="in_progress", + started=False, + message="The upgrade is already running.", + ) + if r.resolve_engine(record) == r.ENGINE_MANAGED: + return EnrollResponse( + phase="succeeded", + started=False, + message="This knowledge base has already been upgraded.", + ) + return await enroll( + assistant_id, + owner_user_id=owner_user_id, + visibility=str(record.get("visibility") or "PRIVATE"), + ) + + generation = int(record.get("migrationGeneration") or 0) + try: + await asyncio.to_thread( + r.retry_from_failed, + assistant_id, + assistant_id, + generation, + _iso(_now() + _DUE_IMMEDIATELY), + ) + except r.TransitionLost: + # A concurrent retry got there first. Its attempt is running, so this is + # a success from the user's point of view. + return EnrollResponse( + phase="in_progress", + started=False, + message="The upgrade is already running.", + ) + logger.info(f"kb {assistant_id}: retried into generation {generation + 1}") + return EnrollResponse( + phase="in_progress", + started=True, + message=( + "Upgrade restarted. Your knowledge base keeps working while it runs." + ), + ) + + +async def dismiss_notice(assistant_id: str) -> None: + """Retire the one-time success notice (Requirement 23.4).""" + import asyncio + + from apis.shared.kb_backend import records as r + + try: + await asyncio.to_thread( + r.dismiss_upgrade_notice, assistant_id, assistant_id, _iso(_now()) + ) + except r.TransitionLost: + # No record, so no notice to dismiss. Nothing owed, nothing to report. + logger.info(f"kb {assistant_id}: notice dismissal for a record that is absent") diff --git a/backend/src/apis/app_api/main.py b/backend/src/apis/app_api/main.py index c47c9978e..cb9349532 100644 --- a/backend/src/apis/app_api/main.py +++ b/backend/src/apis/app_api/main.py @@ -192,6 +192,7 @@ async def lifespan(app: FastAPI): from apis.app_api.assistants.routes import router as assistants_router from apis.app_api.agent_designer.routes import router as agents_router from apis.app_api.documents.routes import router as documents_router +from apis.app_api.kb_upgrade.routes import router as kb_upgrade_router from apis.app_api.users.routes import router as users_router from apis.app_api.user_settings.routes import router as user_settings_router from apis.app_api.connectors.routes import router as connectors_router @@ -217,6 +218,7 @@ async def lifespan(app: FastAPI): app.include_router(assistants_router) app.include_router(agents_router) # Agent Designer /agents surface; 404s while AGENTS_API_ENABLED off app.include_router(documents_router) +app.include_router(kb_upgrade_router) # Owner-facing KB upgrade card; phase "none" (renders nothing) while MANAGED_KB_MIGRATION_ENABLED is off app.include_router(users_router) app.include_router(user_settings_router) app.include_router(models_router) diff --git a/backend/src/apis/inference_api/chat/routes.py b/backend/src/apis/inference_api/chat/routes.py index 11c008bc5..f1a5c2c3e 100644 --- a/backend/src/apis/inference_api/chat/routes.py +++ b/backend/src/apis/inference_api/chat/routes.py @@ -1542,6 +1542,7 @@ async def invocations(request: InvocationRequest, current_user: User = Depends(g if input_data.rag_assistant_id and not is_resume and not is_continuation: # Local imports to avoid circular dependency + from apis.shared.assistants.kb_access import granted from apis.shared.assistants.rag_service import ( augment_prompt_with_context, search_assistant_knowledgebase_with_formatting, @@ -1661,9 +1662,25 @@ async def invocations(request: InvocationRequest, current_user: User = Depends(g assistant = await _get_assistant_cloud_without_ownership_check( input_data.rag_assistant_id, table_name ) + # No permission was resolved, because this path deliberately bypassed the + # gate that resolves one. Left as None so the knowledge base read below + # fails closed (Requirement 25.1): `granted(...)` treats None as "grants + # nothing", and the marketplace scope is authority to *review a + # submission*, not evidence of a read grant on that owner's corpus. + # + # ⚠️ Consequence worth owning: a reviewer test-drives with an empty + # knowledge base, which is a degraded review of a RAG-backed Agent — the + # same class of problem develop's comment above warns about. Whether a + # marketplace reviewer should receive corpus read is a policy decision for + # the marketplace owner, not something to settle inside a merge conflict. + # Tracked rather than guessed; failing closed is the safe default meanwhile. + assistant_permission = None else: logger.info("Loading assistant with access check...") - assistant, _ = await get_assistant_with_access_check( + # The permission is kept, not discarded: it is what the knowledge base + # retrieval below runs under (Requirement 25.1), so the grant that governs + # the corpus read is provably the same one that admitted this turn. + assistant, assistant_permission = await get_assistant_with_access_check( assistant_id=input_data.rag_assistant_id, user_id=user_id, user_email=current_user.email, @@ -1795,7 +1812,10 @@ async def invocations(request: InvocationRequest, current_user: User = Depends(g try: logger.info("Searching knowledge base for assistant...") context_chunks = await search_assistant_knowledgebase_with_formatting( - assistant_id=input_data.rag_assistant_id, query=input_data.message, top_k=5 + assistant_id=input_data.rag_assistant_id, + query=input_data.message, + top_k=5, + access=granted(input_data.rag_assistant_id, user_id, assistant_permission), ) logger.info(f"Knowledge base search returned {len(context_chunks) if context_chunks else 0} chunks") if context_chunks: diff --git a/backend/src/apis/shared/assistants/kb_access.py b/backend/src/apis/shared/assistants/kb_access.py new file mode 100644 index 000000000..68662405f --- /dev/null +++ b/backend/src/apis/shared/assistants/kb_access.py @@ -0,0 +1,237 @@ +"""Who may read a knowledge base — resolved before retrieval is attempted. + +Requirement 25.1–25.3. The application is the authorization authority for +knowledge base reads. Neither of the two mechanisms Bedrock offers is trusted to +be that authority: + +* **Metadata filters** give what AWS's own multi-tenant guidance calls + "filter-level (logical) isolation, *not* IAM-enforced (infrastructure) + isolation". A filter is a query argument; anything that can issue a query can + omit it. +* **ACL-aware retrieval** fails closed, which is better than the document-status + filter used to be, but AWS states plainly that it "is not authorization" and + does not authenticate users. Its identity is **email only, with no alias + resolution, and a mismatch fails silently**. On a platform that authenticates + via OIDC with claim mappings, a silently-failing email comparison is a worse + primitive than an explicit check against the permission model that already + governs the agent. + +So the check happens here, in the application, on the way in. + +Why this lives in ``apis.shared.assistants`` and not in ``kb_backend`` +--------------------------------------------------------------------- +It reuses :func:`apis.shared.assistants.service.resolve_assistant_permission` +rather than introducing a parallel permission model (Requirement 25.2), and +``kb_backend`` may not import the assistants package — that boundary is what keeps +the migration Lambda images small, and it is enforced by +``tests/architecture/test_kb_backend_boundary.py``. Authorization is also +*above* the seam by nature: it is the same answer whichever engine serves the +query, so implementing it once above both adapters is the only way it cannot +differ between them. + +Why the facade takes a resolved grant rather than a user +------------------------------------------------------- +Both production callers resolve the invoking user's permission a few lines before +they retrieve — ``inference_api/chat/routes.py`` via +``get_assistant_with_access_check``, ``app_api/assistants/routes.py`` via +``resolve_assistant_permission``. Re-resolving inside the facade would add a +second DynamoDB read per turn to answer a question the caller has already +answered. + +Passing the answer instead is not weaker, because the parameter is **required and +keyword-only**: a caller that forgets it raises ``TypeError`` at the call site, +which no test suite can miss, while a caller that genuinely has no grant passes +``None`` and gets nothing back. Trusting a caller-supplied *string* would be +weaker; :class:`KbAccess` cannot be constructed with a permission outside +:data:`KB_READ_PERMISSIONS`, so "I have a grant object" is not something a caller +can assert without having gone through :func:`granted` or +:func:`resolve_kb_access`. + +The 1:1 binding is what makes this simple +----------------------------------------- +This phase holds ``App_KB_Id == assistant_id``, so an agent's knowledge base is +exactly that agent's own and "may this user invoke this agent" already answers +"may this turn retrieve". There is deliberately no handling for the 0..N case — +whether one inaccessible knowledge base among several should fail the whole turn +is F4's question, and answering it here would bake in a guess. + +Feature: managed-kb-migration +Requirements: 25.1, 25.2, 25.3 +""" + +from __future__ import annotations + +import logging +from dataclasses import dataclass +from typing import Optional + +logger = logging.getLogger(__name__) + +#: Permissions that may read a knowledge base through its agent. A viewer reads +#: (that is what sharing an agent is *for*) but never sees the upgrade control, +#: which is why the two sets below are separate rather than one ranked scale. +KB_READ_PERMISSIONS = frozenset({"owner", "editor", "viewer"}) + +#: Permissions that may change a knowledge base — upload, delete, or trigger a +#: migration. Deliberately a strict subset: an engine upgrade spends money and +#: mutates the corpus, so it is an owner/editor act. +KB_WRITE_PERMISSIONS = frozenset({"owner", "editor"}) + + +class KbAccessDenied(PermissionError): + """The invoking user may not read this knowledge base. + + Raised only by callers that want an error; :func:`granted` and + :func:`resolve_kb_access` return ``None`` instead, because the retrieval path + turns a denial into "no context" rather than into a failed turn. + """ + + +@dataclass(frozen=True) +class KbAccess: + """A resolved grant to read one knowledge base. + + Frozen, and only ever produced by :func:`granted` or + :func:`resolve_kb_access`, so its existence *is* the statement that the + permission model was consulted. Holding a permission string proves nothing; + holding one of these does. + """ + + assistant_id: str + app_kb_id: str + user_id: str + permission: str + + @property + def may_read(self) -> bool: + """True for every instance. Kept as a named predicate rather than an + implicit invariant so a call site reads as a check, and so the day a + write-only grant is added there is somewhere for it to go.""" + return self.permission in KB_READ_PERMISSIONS + + @property + def may_upgrade(self) -> bool: + """Whether this grant may trigger a migration for the knowledge base.""" + return self.permission in KB_WRITE_PERMISSIONS + + +def granted( + assistant_id: str, + user_id: str, + permission: Optional[str], + app_kb_id: Optional[str] = None, +) -> Optional[KbAccess]: + """Wrap an already-resolved permission as a grant, or ``None`` if it grants nothing. + + For the callers that resolved the permission themselves a moment earlier. + ``None``, an empty string, and any unrecognized value all return ``None``: + an unknown permission written by newer code is not evidence of access, and + guessing in the permissive direction is how a viewer-shaped bug becomes a + disclosure. + + ``app_kb_id`` defaults to ``assistant_id`` — the 1:1 binding this phase keeps. + """ + if not permission or permission not in KB_READ_PERMISSIONS: + logger.warning( + f"knowledge base access denied for user {user_id} on assistant " + f"{assistant_id}: permission {permission!r} does not grant read" + ) + return None + + return KbAccess( + assistant_id=assistant_id, + app_kb_id=app_kb_id or assistant_id, + user_id=user_id, + permission=permission, + ) + + +async def resolve_kb_access( + assistant_id: str, + user_id: str, + user_email: Optional[str] = None, + app_kb_id: Optional[str] = None, +) -> Optional[KbAccess]: + """Resolve a grant from the assistant permission model, failing closed. + + For callers that do not already hold a permission. Delegates to + ``resolve_assistant_permission`` — the same function the document routes and + the listing service gate on — so owner/editor/viewer semantics are whatever + that function says they are and cannot drift here (Requirement 25.2). + + **Any** failure denies: a missing table, an unreachable table, a malformed + record. This is the opposite of the resolver's choice to treat an unreadable + KB_Record as legacy, and deliberately so. There, both answers serve the user's + own documents and one of them is always safe. Here the two answers are "your + documents" and "someone else's", and an error tells us which is which is + exactly what we do not know (Requirement 24.6). + """ + # Function-local: importing ``.service`` at module scope would run inside the + # package ``__init__``'s own import of ``rag_service``, and the ordering that + # makes that work today is not a property worth depending on. + from apis.shared.assistants.service import resolve_assistant_permission + + try: + _assistant, permission = await resolve_assistant_permission( + assistant_id=assistant_id, user_id=user_id, user_email=user_email + ) + except Exception as exc: + logger.error( + f"knowledge base access check failed for user {user_id} on assistant " + f"{assistant_id}; denying because access cannot be confirmed: {exc}", + exc_info=True, + ) + return None + + return granted(assistant_id, user_id, permission, app_kb_id) + + +async def is_shared_beyond_owner(assistant_id: str, owner_id: str, visibility: Optional[str] = None) -> bool: + """Whether anyone other than the owner can reach this assistant's documents. + + The input to Requirement 25.6's "where a knowledge base is shared beyond its + owner". Lives here rather than in ``kb_backend`` for the same reason the access + check does: it is an application fact, and the seam may not read it. + + Two independent sources, because either alone under-detects: + + * **Visibility.** ``PUBLIC`` shares with everyone and ``SHARED`` announces the + intent, so neither is owner-only. + * **Share records.** A ``PRIVATE`` assistant can still carry explicit shares — + ``resolve_assistant_permission`` resolves an editor share on a private + assistant to ``editor``. Judging by visibility alone would leave those + knowledge bases unprotected, which is the case most likely to exist and least + likely to be noticed. + + Fails **shared** on error. The two answers are "apply a narrowing policy that + was not strictly needed" and "leave a multi-user corpus reachable by anything + in the account with a wildcard grant"; the first costs one control-plane call. + """ + if visibility in ("PUBLIC", "SHARED"): + return True + + from apis.shared.assistants.service import list_assistant_shares + + try: + shares = await list_assistant_shares(assistant_id, owner_id) + except Exception as exc: + logger.error( + f"could not determine whether assistant {assistant_id} is shared; " + f"assuming it is, so that a resource policy is applied rather than " + f"skipped: {exc}", + exc_info=True, + ) + return True + + return bool(shares) + + +__all__ = [ + "KB_READ_PERMISSIONS", + "KB_WRITE_PERMISSIONS", + "KbAccess", + "KbAccessDenied", + "granted", + "is_shared_beyond_owner", + "resolve_kb_access", +] diff --git a/backend/src/apis/shared/assistants/kb_publication.py b/backend/src/apis/shared/assistants/kb_publication.py new file mode 100644 index 000000000..bae24dbaf --- /dev/null +++ b/backend/src/apis/shared/assistants/kb_publication.py @@ -0,0 +1,123 @@ +"""What publication means for a knowledge base's lifecycle. + +Requirements 25.8–25.11. Three positions, and one question deliberately left open. + +**An engine migration is not a corpus change (25.8).** Parity is the entire +contract of this migration: the same documents, the same ``top_k``, the same +context cap, the same answer model. So swapping which engine serves a published +agent's knowledge base does not change what that agent retrieves and therefore +needs no re-review. :func:`migration_requires_review` says so in one place, with a +test, rather than leaving it as an assumption spread across the worker. + +**A listed agent's knowledge base is exempt from reclaim (25.9).** Nothing reclaims +in this phase — ``reclaim`` is reserved in the state enum and never entered — so +this is a guard placed before the mechanism that will need it. Written now because +the follow-up spec's eviction pass will be the first thing to delete a corpus, and +"is this on the store shelf right now?" is not a question it should be answering +for the first time under a deadline. + +**A takedown must be walked, not fallen into (25.10).** Reclaim eligibility is +computed from the listing state as it is, and ``taken_down`` is reached only by the +listing state machine's explicit edge. Nothing here infers a takedown from a +missing listing, an expired timestamp, or an error. + +**Corpus-revision pinning stays open (25.11).** A marketplace listing freezes a +knowledge base *reference*, not its contents, so a published agent's answers can +change after review without any re-review. That is a real review bypass and this +module does not pretend to close it: exemption from cleanup is not revision +pinning. The question belongs to the marketplace spec. + +Why ``is_on_shelf`` and not ``is_listed`` +---------------------------------------- +The listing module documents the trap and it applies exactly here: an admin +requesting changes on a *live* listing leaves it serving but moves its state to +``changes_requested``, which is not in ``LISTED_STATES``. Asked by state name +alone, such an agent reads as unlisted — so a reclaim pass would delete the corpus +behind an agent users can still see in the store. ``is_on_shelf`` asks the fact +(is a version of this queryable in the store right now) rather than the state name, +and that is the only correct question for a destructive pass. + +Feature: managed-kb-migration +Requirements: 25.8, 25.9, 25.10, 25.11 +""" + +from __future__ import annotations + +import logging +from typing import Any, Mapping, Optional + +from apis.shared.assistants.listing import is_on_shelf + +logger = logging.getLogger(__name__) + + +def migration_requires_review(from_engine: Optional[str], to_engine: Optional[str]) -> bool: + """Whether changing engines needs a listed agent to be reviewed again. + + Always ``False``, and a function rather than a comment so the claim is + something a test can hold. An engine swap moves the same documents to a + different index; the reviewed artefact — instructions, bindings, the corpus + itself — is untouched. If a future change makes an engine swap alter retrieval + results, this must stop returning ``False``, and the test that pins it is where + that argument has to be had. + """ + return False + + +def is_reclaim_exempt( + kb_record: Optional[Mapping[str, Any]], + listing_state: Optional[str] = None, + published_version: Optional[int] = None, +) -> bool: + """Whether this knowledge base must be left alone by a lifecycle reclaim pass. + + Exempt when any of these holds: + + * the record carries ``exemptFromReclaim`` — an operator's explicit hold; + * the record is ``pinned``; + * the agent is on the store shelf right now. + + A **missing** record — ``None``, which is what ``get_kb_record`` returns when + it cannot find one — is exempt. Reclaim acts on knowledge bases it can + describe, and "I could not read the record" is not a description; the same + fail-closed reasoning the access check uses, applied to deletion, where it + matters more. An *empty* mapping is a different thing: a record that was read + and carries no holds. Treating the two alike would exempt every unheld + knowledge base and make the whole predicate vacuous. + """ + if kb_record is None: + return True + + if kb_record.get("exemptFromReclaim") or kb_record.get("pinned"): + return True + + return is_on_shelf(listing_state, published_version) + + +def reclaim_exemption_reason( + kb_record: Optional[Mapping[str, Any]], + listing_state: Optional[str] = None, + published_version: Optional[int] = None, +) -> Optional[str]: + """Why this knowledge base is exempt, or ``None`` if it is not. + + For the report-only output of a reclaim pass. A pass that logs "skipped 400 + knowledge bases" tells an operator nothing they can act on; one that says which + are on the shelf and which an operator pinned by hand does. + """ + if kb_record is None: + return "no KB_Record could be read" + if kb_record.get("exemptFromReclaim"): + return "exemptFromReclaim is set on the record" + if kb_record.get("pinned"): + return "the knowledge base is pinned" + if is_on_shelf(listing_state, published_version): + return f"the agent is on the store shelf (listing state {listing_state!r})" + return None + + +__all__ = [ + "is_reclaim_exempt", + "migration_requires_review", + "reclaim_exemption_reason", +] diff --git a/backend/src/apis/shared/assistants/rag_service.py b/backend/src/apis/shared/assistants/rag_service.py index 6c3d76e4f..9e39ff763 100644 --- a/backend/src/apis/shared/assistants/rag_service.py +++ b/backend/src/apis/shared/assistants/rag_service.py @@ -1,28 +1,99 @@ """RAG service for assistant knowledge base search and prompt augmentation -This service handles searching the vector store for assistant-specific -knowledge and augmenting user prompts with retrieved context. +This module is the **facade** over the knowledge base seam. It resolves which +backend serves an assistant's knowledge base, delegates the search, and then +applies the properties that must hold identically on every backend. It contains +no retrieval logic of its own: what an S3 Vectors response looks like now lives +in ``apis.shared.kb_backend.s3vectors_backend``. + +What lives here, and why here +----------------------------- +Four rules sit above the seam rather than inside either adapter, because a rule +implemented twice is a rule that will eventually differ (Requirement 3): + +* **The access check.** No backend is contacted until the invoking user's grant + has been resolved (Requirement 25.1). It is above the seam because the answer + does not depend on the engine, and because Bedrock's own isolation features are + not trusted to be the authority — see ``kb_access``. +* **The document-status filter.** Dropped chunks whose parent document is not + ``complete``. Kept on both backends during parity even though managed + ingestion makes it largely redundant — removing it in the same change that + swaps the engine would make any difference in results unattributable. +* **``top_k`` narrowing.** Applied *after* the status filter, which is the order + the legacy path has always used: filter-then-slice, so an incomplete document + cannot silently shrink a five-chunk answer. +* **The 2,000-character context cap.** ``augment_prompt_with_context``'s + default. Held constant deliberately: the evaluation measured no correctness + change between 2,000 and 20,000 characters, so raising it here would add a + variable to a change whose whole purpose is to hold every variable but one. + +The dual-read pilot +------------------- +Also above the seam, and for the same reason: comparing two backends is not a +thing either backend can do. The facade starts the observational managed read +before awaiting legacy and detaches the comparison afterwards, so a piloted turn +waits exactly as long as an unpiloted one (Requirement 18.5). Legacy is always +what is served. See ``kb_backend.dual_read``. + +Score direction +---------------The seam speaks **relevance** (higher is better). This facade still emits a +``distance`` key (lower is better), derived by exact negation, because +``app_api/assistants/routes.py`` puts that value in an HTTP response body that a +client already reads. The rename stops at the seam; no caller has to change. """ import logging import os -from typing import Any, Dict, List, Set +import time +from typing import Any, Dict, List, Optional, Set import boto3 -from apis.shared.embeddings.bedrock_embeddings import search_assistant_knowledgebase +from apis.shared.assistants.kb_access import KbAccess +from apis.shared.kb_backend.dual_read import schedule_observation, start_managed_read +from apis.shared.kb_backend.idleness import schedule_activity_touch +from apis.shared.kb_backend.metrics import ( + METRIC_ACCESS_DENIED, + METRIC_STATUS_FILTER_FAIL_CLOSED, + emit_count, +) +from apis.shared.kb_backend.protocol import DEFAULT_TOP_K, Chunk, distance_from_relevance +from apis.shared.kb_backend.query_guard import clamp_query +from apis.shared.kb_backend.resolver import load_record, resolve_backend logger = logging.getLogger(__name__) +#: Parity contract (Requirement 3.2): the cap is 2,000 characters on every +#: backend, unchanged from the value the legacy path has always used. Named so +#: that a change to it is a visible change to a constant rather than an edit to a +#: default argument. +MAX_CONTEXT_CHARS = 2000 -async def search_assistant_knowledgebase_with_formatting(assistant_id: str, query: str, top_k: int = 5) -> List[Dict[str, Any]]: + +async def search_assistant_knowledgebase_with_formatting( + assistant_id: str, + query: str, + top_k: int = DEFAULT_TOP_K, + *, + access: Optional[KbAccess], +) -> List[Dict[str, Any]]: """ Search assistant knowledge base and return formatted results + Resolves the knowledge base's backend, delegates the search across the seam, + then applies the parity rules that must hold on every backend: the document + status filter, and ``top_k`` narrowing after it. + Args: assistant_id: Assistant identifier to filter vectors query: User query text top_k: Number of top results to return (default: 5) + access: The invoking user's resolved grant, or ``None`` if they have none. + Required and keyword-only (Requirement 25.1): a caller that forgets it + fails loudly at the call site, while a caller that genuinely has no + grant passes ``None`` and gets nothing. Build one with + ``kb_access.granted`` if the permission is already in hand, or + ``kb_access.resolve_kb_access`` if it is not. Returns: List of dictionaries containing: @@ -30,32 +101,96 @@ async def search_assistant_knowledgebase_with_formatting(assistant_id: str, quer - distance: Similarity distance (lower = more similar) - metadata: Original metadata from vector store - key: Vector key/ID + + Empty when the caller has no grant — no backend is contacted at all. """ - try: - # Call the bedrock_embeddings search function - response = await search_assistant_knowledgebase(assistant_id, query) + # Authorization first, before the backend resolution, the query clamp, and + # any AWS call (Requirement 25.1). Ordering is the requirement: a check that + # runs after retrieval has already read the corpus is an audit log, not an + # access control. + if access is None or not access.may_read: + logger.error( + f"refusing knowledge base retrieval for assistant {assistant_id}: " + f"no resolved access grant" + ) + emit_count(METRIC_ACCESS_DENIED, dimensions={"reason": "no_grant"}) + return [] - # Extract vectors from response - vectors = response.get("vectors", []) + if access.assistant_id != assistant_id: + # A grant for a different assistant is not a grant for this one. This is + # the shape a copy-paste bug takes when a route resolves permission for + # one id and retrieves with another, and while the 1:1 binding holds it is + # the only way the two could disagree. + logger.error( + f"refusing knowledge base retrieval: grant is for assistant " + f"{access.assistant_id}, not {assistant_id}" + ) + emit_count(METRIC_ACCESS_DENIED, dimensions={"reason": "grant_mismatch"}) + return [] - if not vectors: + managed_task = None + try: + # One record read serves both questions: which backend to use, and whether + # this knowledge base is in the dual-read pilot. Reading it here rather + # than letting the resolver read it internally is what keeps the pilot + # from costing an extra DynamoDB round trip on every turn. + record = load_record(assistant_id) + backend = resolve_backend(assistant_id, record=record) + + # Clamp before dispatch, so both backends receive an identically-shaped + # query (Requirement 4.2). Managed KB rejects anything over 10,000 + # characters outright and the quota is not adjustable, so clamping only + # the managed path would make the two backends answer different + # questions and invalidate the dual-read comparison. + query, _ = clamp_query(query) + + # Start the observational read *before* awaiting legacy (Requirement + # 18.5). Nothing is awaited here, so a piloted turn does the same waiting + # as an unpiloted one; managed Retrieve measured 662–695 ms p50 against + # legacy's 257 ms, so awaiting both would nearly triple this leg. + # ``None`` whenever there is no comparison to make. + managed_task = start_managed_read(record, assistant_id, query, top_k) + + started = time.perf_counter() + chunks = await backend.search(assistant_id, query, top_k) + legacy_ms = (time.perf_counter() - started) * 1000.0 + + # Detach the comparison. Legacy is what gets served either way — including + # when it is empty, which is a finding rather than a reason to reach for + # the other engine's answer (Requirement 18.2). + schedule_observation(assistant_id, query, top_k, list(chunks), legacy_ms, managed_task) + managed_task = None + + # Record that this knowledge base was needed (Requirement 22.5), for the + # idleness signal the follow-up spec's eviction threshold has to be chosen + # from — data that cannot be backfilled later. + # + # Only for knowledge bases that have a record. A legacy knowledge base has + # none, and creating one here would break the migration's zero-backfill + # property across 1,692 existing rows for the sake of a metric. Detached and + # throttled, so retrieval waits for neither the write nor its rejection + # (Requirement 22.6). + if record: + schedule_activity_touch(assistant_id, assistant_id) + + if not chunks: logger.info(f"No vectors found for assistant {assistant_id} with query: {query[:50]}...") return [] # Filter out chunks from documents that are not in "complete" status - vectors = _filter_vectors_by_document_status(vectors, assistant_id) + chunks = _filter_chunks_by_document_status(chunks, assistant_id) # Format results - return document_id for on-demand download URL generation formatted_results = [] - for vector in vectors[:top_k]: - metadata = vector.get("metadata", {}) - + for chunk in chunks[:top_k]: formatted_results.append( { - "text": metadata.get("text", ""), - "distance": vector.get("distance"), - "metadata": metadata, - "key": vector.get("key", ""), + "text": chunk.text, + # Derived from relevance by exact negation, so the value a + # caller reads is the one it has always read. + "distance": distance_from_relevance(chunk.relevance), + "metadata": chunk.metadata, + "key": chunk.key, } ) @@ -64,10 +199,41 @@ async def search_assistant_knowledgebase_with_formatting(assistant_id: str, quer except Exception as e: logger.error(f"Error searching knowledge base for assistant {assistant_id}: {e}", exc_info=True) + if managed_task is not None: + # The legacy search raised before the comparison was detached, so + # nothing will ever await this task. Left alone it would run to + # completion, pay for a Retrieve, and be reported as a task whose + # exception was never retrieved. + managed_task.cancel() # Return empty list on error (graceful degradation) return [] +def _filter_chunks_by_document_status(chunks: List[Chunk], assistant_id: str) -> List[Chunk]: + """ + Apply the document status filter to protocol chunks, on any backend. + + Delegates to :func:`_filter_vectors_by_document_status` rather than + reimplementing the lookup, so both backends share one set of DynamoDB + semantics — including its fallback behaviour, which task group 6 changes in + exactly one place. + + Each chunk is presented to the filter as a minimal view carrying only what + the filter reads (``metadata.document_id``) plus its index, and survivors are + mapped back by that index. Order and duplicates are preserved. + + Args: + chunks: Chunks returned by a backend, in backend ranking order + assistant_id: Assistant identifier for DynamoDB key construction + + Returns: + The subset of chunks whose parent document is 'complete' + """ + views = [{"metadata": chunk.metadata, "_chunk_index": index} for index, chunk in enumerate(chunks)] + surviving = _filter_vectors_by_document_status(views, assistant_id) + return [chunks[view["_chunk_index"]] for view in surviving] + + def _filter_vectors_by_document_status(vectors: List[Dict[str, Any]], assistant_id: str) -> List[Dict[str, Any]]: """ Filter vector results to only include chunks from documents with status='complete'. @@ -75,7 +241,14 @@ def _filter_vectors_by_document_status(vectors: List[Dict[str, Any]], assistant_ Extracts unique document_ids from vector metadata, looks up each document's status in DynamoDB, and removes chunks from documents that are not 'complete' or don't exist. - On any DynamoDB failure, falls back to returning unfiltered results (graceful degradation). + Fails CLOSED (Requirement 5): if status cannot be confirmed — no table + configured, or the lookup errors — every chunk is dropped and an empty list is + returned. This deliberately supersedes `reliable-document-deletion` + Requirement 3.4, which specified the opposite. The reasoning changed because + the fail-open path was measured: 936 retrievals in a trailing 30-day window had + chunks dropped by this filter, so the documents it guards against are real, and + serving a user content they believe they deleted is worse than serving nothing. + A per-document lookup failure still skips only that document. Args: vectors: List of vector results from S3 Vectors search @@ -119,12 +292,31 @@ def _filter_vectors_by_document_status(vectors: List[Dict[str, Any]], assistant_ logger.warning(f"Failed to look up document {doc_id}: {e}") # Skip individual lookup failures else: - # No table configured — fall back to unfiltered - logger.warning("DYNAMODB_ASSISTANTS_TABLE_NAME not configured, returning unfiltered results") - valid_doc_ids = doc_ids + # FAIL CLOSED (Requirement 5.2). Previously this returned everything + # unfiltered. Without a table there is no way to confirm that a + # document is still `complete`, and the chunks in question may belong + # to documents a user has deleted. Serving unverifiable content is a + # worse outcome than serving none: the user sees material they believe + # they removed, and nothing in the response signals that the check was + # skipped. + logger.error( + "DYNAMODB_ASSISTANTS_TABLE_NAME not configured; dropping all " + "chunks because document status cannot be confirmed" + ) + emit_count(METRIC_STATUS_FILTER_FAIL_CLOSED) + return [] except Exception as e: - logger.warning(f"DynamoDB lookup failed, returning unfiltered results: {e}") - valid_doc_ids = doc_ids # Graceful degradation + # FAIL CLOSED (Requirement 5.1). Same reasoning as above. Logged at ERROR, + # not WARNING: an empty result from this path is a degradation, and it must + # be distinguishable from the ordinary "corpus had no match" case, which is + # logged at INFO below. + logger.error( + f"Document status lookup failed; dropping all chunks because status " + f"cannot be confirmed: {e}", + exc_info=True, + ) + emit_count(METRIC_STATUS_FILTER_FAIL_CLOSED) + return [] # Filter vectors to only include chunks from valid documents filtered = [v for v in vectors if v.get("metadata", {}).get("document_id") in valid_doc_ids] @@ -136,13 +328,16 @@ def _filter_vectors_by_document_status(vectors: List[Dict[str, Any]], assistant_ return filtered -def augment_prompt_with_context(user_message: str, context_chunks: List[Dict[str, Any]], max_context_length: int = 2000) -> str: +def augment_prompt_with_context(user_message: str, context_chunks: List[Dict[str, Any]], max_context_length: int = MAX_CONTEXT_CHARS) -> str: """ Augment user message with retrieved context chunks The context is prepended to the user message with clear delimiters. This allows the LLM to use the retrieved knowledge when generating responses. + Applies on both backends: the cap lives here, above the seam, so neither + adapter can widen it independently. + Args: user_message: Original user message context_chunks: List of context chunks from vector search diff --git a/backend/src/apis/shared/assistants/service.py b/backend/src/apis/shared/assistants/service.py index 8849971f6..2e05fbe1c 100644 --- a/backend/src/apis/shared/assistants/service.py +++ b/backend/src/apis/shared/assistants/service.py @@ -588,9 +588,17 @@ async def _update_assistant_cloud(assistant: Assistant, table_name: str) -> None # edit would re-write a stale GSI5 key onto a delisted agent and silently put it # back in the store. ``listing`` is excluded for the same reason in the other # direction: a stale in-memory copy must not clobber a concurrent review decision. + # + # GSI7_* belongs to the managed-KB migration write path + # (``apis.shared.kb_backend.records``). Unlike GSI5, those keys live on a + # separate item (``SK = KB#{app_kb_id}``) rather than on this METADATA item, so + # this path cannot reach them today — the entry is here so the invariant "GSI + # keys are never written from the generic update" holds uniformly, and a reader + # does not have to know which index lives on which item to trust it. immutable_fields = { "PK", "SK", "GSI_PK", "GSI_SK", "GSI2_PK", "GSI2_SK", "GSI5_PK", "GSI5_SK", + "GSI7_PK", "GSI7_SK", "assistantId", "createdAt", "ownerId", "listing", } diff --git a/backend/src/apis/shared/embeddings/bedrock_embeddings.py b/backend/src/apis/shared/embeddings/bedrock_embeddings.py index 22d1ac3dd..0e74bc7ea 100644 --- a/backend/src/apis/shared/embeddings/bedrock_embeddings.py +++ b/backend/src/apis/shared/embeddings/bedrock_embeddings.py @@ -57,7 +57,13 @@ async def generate_embeddings(chunks: List[str]) -> List[List[float]]: IMPORTANT: This function does NOT validate token counts. Callers that process large documents should validate/split chunks before calling this. - For search queries (short strings), no validation is needed. + + Search queries are length-capped by the caller, not here: the facade applies + `apis.shared.kb_backend.query_guard.clamp_query` before dispatch. This used to + say no validation was needed for queries, which was true only because Titan v2 + tolerates roughly 32,000 characters. Managed Knowledge Base caps `Retrieve` + input at 10,000 and that quota is not adjustable, so the assumption no longer + holds for every backend. Args: chunks: List of text chunks to embed @@ -145,7 +151,8 @@ async def search_assistant_knowledgebase(assistant_id: str, query: str): """Search the S3 vector store for chunks relevant to the query.""" client = boto3.client("s3vectors", region_name=AWS_REGION) - # Generate vector for the query (short string, no token validation needed) + # Generate vector for the query. Length is already capped upstream by the + # facade's query clamp (10,000 chars, the Managed KB Retrieve limit). query_embedding = await generate_embeddings([query]) # Query the Global Index with a STRICT Filter diff --git a/backend/src/apis/shared/kb_backend/__init__.py b/backend/src/apis/shared/kb_backend/__init__.py new file mode 100644 index 000000000..e69de29bb diff --git a/backend/src/apis/shared/kb_backend/byte_cap.py b/backend/src/apis/shared/kb_backend/byte_cap.py new file mode 100644 index 000000000..dd5e532a4 --- /dev/null +++ b/backend/src/apis/shared/kb_backend/byte_cap.py @@ -0,0 +1,258 @@ +"""Per-owner byte cap accounting for managed knowledge bases. + +Managed storage is billed at $5.00/GB-month, roughly 35x what S3 Vectors costs +today. At the measured average of 1.13 MB per user that is about $169/month across +the fleet — but nothing structural stops one user uploading far more, and 30,000 +users at 100 MB each would be 3 TB, or about $15,000/month. The cap is what turns +"unlikely" into "impossible". + +Why an accumulator instead of the obvious condition +--------------------------------------------------- +The natural way to express this is:: + + ConditionExpression="storedBytes + reservedBytes + :n <= :cap" + +**DynamoDB rejects that.** Condition expressions compare operands; they cannot do +arithmetic. Verified directly: the parser fails with ``Cannot parse condition +starting at:+ reserved <= :cap``. + +So the arithmetic is moved to the client, where it is free. A single +``totalBytes`` accumulator is maintained as the invariant +``totalBytes == storedBytes + reservedBytes``, and the guard compares it against a +**literal computed before the call**:: + + ADD totalBytes :n, reservedBytes :n + CONDITION totalBytes <= :max_before where :max_before = cap - n + +That is a single atomic conditional update, so N concurrent reservations cannot +collectively overshoot. The alternative — read, compute, write — has a window +between the read and the write in which another writer commits, which is exactly +the race a cap exists to prevent. + +Reserve / commit / release, not just "add" +----------------------------------------- +Ingestion is not instantaneous: a 50 KiB PDF measured 68-264 seconds. Counting +bytes only on success would let a user start unlimited concurrent uploads that are +each individually under the cap and collectively far over it. So bytes are reserved +up front, converted to stored on success, and returned on failure. A crash between +reserve and commit leaks a reservation, which is the safe direction — it +under-permits rather than over-permits, and the reconciler can recover it. + +Sizing +------ +Size always comes from an S3 ``HEAD`` on the stored object, never from a +client-reported value: a client that under-reports its own size would defeat the +cap entirely. Bedrock's ``RawDataSize`` metric is deliberately **not** used for +enforcement — it returned 0 datapoints for a directly-ingested document during +evaluation and remains unconfirmed. Enforcing against a metric that is sometimes +absent would fail open. + +Import weight +------------- +Module-level imports are stdlib only; ``boto3`` is function-local, so this module +can be imported into a size-constrained Lambda image for free. +""" + +from __future__ import annotations + +import logging +import os +from decimal import Decimal +from typing import Optional + +from apis.shared.kb_backend.metrics import emit_count + +logger = logging.getLogger(__name__) + +METRIC_BYTE_CAP_REJECTED = "KbByteCapRejected" + +#: Defaults mirror the CDK config (Requirement 12.2). Both are read from the +#: environment so an operator can tune them without a code change; the fallbacks +#: keep local runs working. +#: +#: 100 MB is deliberately BELOW the existing 1 GB user-files precedent. At $5.00 +#: per GB-month that precedent would permit roughly $150,000/month across the +#: fleet, which is not a limit so much as a formality. +DEFAULT_PER_OWNER_BYTES = 100 * 1024 * 1024 +DEFAULT_PER_OWNER_ELEVATED_BYTES = 1024 * 1024 * 1024 +DEFAULT_PER_KB_CEILING_BYTES = 500 * 1024 * 1024 + + +class ByteCapExceeded(Exception): + """A reservation would take the owner over their cap. + + Carries the numbers so the caller can render a plain-language message with the + option to request an elevated tier, rather than a bare failure (Requirement + 12.12). A user who cannot see how far over they are cannot act on it. + """ + + def __init__(self, requested: int, cap: int, already_used: Optional[int] = None) -> None: + self.requested = requested + self.cap = cap + self.already_used = already_used + super().__init__( + f"reserving {requested} bytes would exceed the {cap}-byte cap" + + (f" (already using {already_used})" if already_used is not None else "") + ) + + +def _env_int(name: str, default: int) -> int: + raw = os.environ.get(name) + if not raw: + return default + try: + return int(raw) + except ValueError: + logger.warning(f"{name}={raw!r} is not an integer; falling back to {default}") + return default + + +def per_owner_cap(elevated: bool = False) -> int: + """The owner's total allowance in bytes. + + Which tier a user belongs to is the caller's decision: RBAC already owns role + resolution and this module should not grow a second opinion about it. + """ + if elevated: + return _env_int("MANAGED_KB_PER_OWNER_ELEVATED_BYTES", DEFAULT_PER_OWNER_ELEVATED_BYTES) + return _env_int("MANAGED_KB_PER_OWNER_DEFAULT_BYTES", DEFAULT_PER_OWNER_BYTES) + + +def per_kb_ceiling() -> int: + """Ceiling for a single knowledge base, independent of the owner's total. + + Stops one knowledge base consuming an entire elevated allowance and starving + the owner's others. + """ + return _env_int("MANAGED_KB_PER_KB_CEILING_BYTES", DEFAULT_PER_KB_CEILING_BYTES) + + +def _table(): + import boto3 + + return boto3.resource("dynamodb").Table(os.environ["DYNAMODB_ASSISTANTS_TABLE_NAME"]) + + +def object_size_bytes(bucket: str, key: str) -> int: + """Authoritative size, from S3 rather than from the client. + + A client-reported size is an input, and an input that can lower its own cost is + not a measurement. + """ + import boto3 + + response = boto3.client("s3").head_object(Bucket=bucket, Key=key) + return int(response["ContentLength"]) + + +def reserve( + assistant_id: str, + app_kb_id: str, + n_bytes: int, + cap: int, +) -> None: + """Reserve ``n_bytes`` against the cap, atomically. + + Raises :class:`ByteCapExceeded` if the reservation would breach the cap. The + comparison is against ``cap - n_bytes``, computed here, because DynamoDB cannot + add inside a condition — see the module docstring. + + ``attribute_not_exists`` covers the first reservation on a record that has + never held bytes, so a fresh knowledge base does not need initialising. + """ + from botocore.exceptions import ClientError + + from apis.shared.kb_backend.records import kb_pk, kb_sk + + if n_bytes < 0: + raise ValueError("n_bytes must not be negative") + if n_bytes == 0: + return + if n_bytes > cap: + # Cannot fit even into an empty allowance; no point issuing the write. + emit_count(METRIC_BYTE_CAP_REJECTED) + raise ByteCapExceeded(requested=n_bytes, cap=cap) + + try: + _table().update_item( + Key={"PK": kb_pk(assistant_id), "SK": kb_sk(app_kb_id)}, + UpdateExpression="ADD #total :n, #reserved :n", + ConditionExpression="attribute_not_exists(#total) OR #total <= :max_before", + ExpressionAttributeNames={ + # `total` is a DynamoDB reserved keyword, so these are aliased. + "#total": "totalBytes", + "#reserved": "reservedBytes", + }, + ExpressionAttributeValues={ + ":n": Decimal(n_bytes), + ":max_before": Decimal(cap - n_bytes), + }, + ) + except ClientError as exc: + if exc.response.get("Error", {}).get("Code") == "ConditionalCheckFailedException": + emit_count(METRIC_BYTE_CAP_REJECTED) + raise ByteCapExceeded(requested=n_bytes, cap=cap) from exc + raise + + +def commit(assistant_id: str, app_kb_id: str, n_bytes: int) -> None: + """Convert a reservation into stored bytes. + + ``totalBytes`` is untouched: the bytes were already counted at reserve time. + Adding here as well would double-count and shrink the owner's allowance on + every successful upload. + """ + from apis.shared.kb_backend.records import kb_pk, kb_sk + + if n_bytes == 0: + return + _table().update_item( + Key={"PK": kb_pk(assistant_id), "SK": kb_sk(app_kb_id)}, + UpdateExpression="ADD #reserved :neg, #stored :n", + ExpressionAttributeNames={"#reserved": "reservedBytes", "#stored": "storedBytes"}, + ExpressionAttributeValues={":neg": Decimal(-n_bytes), ":n": Decimal(n_bytes)}, + ) + + +def release(assistant_id: str, app_kb_id: str, n_bytes: int) -> None: + """Return a reservation after a failed ingestion. + + Decrements both the reservation and the accumulator, restoring the allowance + exactly. Not releasing would silently shrink the owner's cap with every failed + upload until they could not upload at all — a leak that presents as "the + product stopped working" long after the failures that caused it. + """ + from apis.shared.kb_backend.records import kb_pk, kb_sk + + if n_bytes == 0: + return + _table().update_item( + Key={"PK": kb_pk(assistant_id), "SK": kb_sk(app_kb_id)}, + UpdateExpression="ADD #reserved :neg, #total :neg", + ExpressionAttributeNames={"#reserved": "reservedBytes", "#total": "totalBytes"}, + ExpressionAttributeValues={":neg": Decimal(-n_bytes)}, + ) + + +def reserve_snapshot( + assistant_id: str, + app_kb_id: str, + total_bytes: int, + cap: int, +) -> None: + """Reserve a whole migration corpus up front (Requirement 12.11/12.12). + + Migration is the largest byte-adding operation in the system and the only one + that runs unattended, which makes it both the easiest place to forget the check + and the worst. Reserving per-document as the worker progresses would let a + migration run for an hour and then stop halfway, leaving a half-populated + managed knowledge base and an owner over their cap with no way back. + + So the entire snapshot is reserved *before* the migration enters ``shadow``. A + corpus that cannot fit fails immediately, with numbers the caller can turn into + "this needs an elevated tier" rather than a stack trace. + + Deliberately the same conditional write as :func:`reserve`; the distinction is + the caller's contract, not the mechanism. + """ + reserve(assistant_id, app_kb_id, total_bytes, cap) diff --git a/backend/src/apis/shared/kb_backend/dual_read.py b/backend/src/apis/shared/kb_backend/dual_read.py new file mode 100644 index 000000000..03f4ae977 --- /dev/null +++ b/backend/src/apis/shared/kb_backend/dual_read.py @@ -0,0 +1,347 @@ +"""The dual-read pilot: measure the managed backend against real traffic. + +Requirement 18. The rollout should rest on evidence from *our* corpus and *our* +users, not solely on a three-document benchmark. So an opted-in knowledge base +can have both backends answer the same query, with **legacy always served** and +the managed result kept purely as an observation. + +Three rules make this safe to leave switched on, and each is a property of the +code rather than an intention: + +* **Legacy is what is served.** The managed result never reaches the caller. It + is not blended, not preferred when it looks better, not used as a fallback when + legacy is empty — an empty legacy result is a *finding*, and substituting the + other engine's answer would destroy the measurement and change what users see + in the same move. +* **The managed call cannot fail the turn.** It runs as a detached task whose + exceptions are logged and dropped. :func:`observe` has no failure mode that + propagates. +* **It cannot add user-visible latency (18.5).** Both searches start together and + the caller is handed the legacy result the moment it resolves; the managed call + keeps running afterwards on the event loop. This matters concretely: managed + ``Retrieve`` measured a 662–695 ms p50 against legacy's 257 ms, so anything that + awaited both would nearly triple the retrieval leg of every piloted turn. + +Why the task needs a strong reference +------------------------------------- +``asyncio.create_task`` returns the only strong reference to the task. Drop it and +the task becomes eligible for garbage collection mid-flight, which surfaces as +comparisons that silently stop being recorded under load — the failure mode that +looks like "the pilot found nothing interesting". Hence :data:`_IN_FLIGHT` and the +done-callback that discards from it, which is the documented CPython pattern. + +Why the comparison is a pure function +------------------------------------- +:func:`compare` takes two chunk lists and two durations and returns a value. It +touches no clock, no client and no environment, so the ranking mathematics can be +tested without any of the machinery around it — and the machinery can be tested +without asserting on arithmetic. + +Feature: managed-kb-migration +Requirements: 18.1, 18.2, 18.3, 18.4, 18.5 +""" + +from __future__ import annotations + +import asyncio +import logging +import time +from dataclasses import dataclass +from typing import Any, Dict, List, Mapping, Optional, Sequence, Set + +from apis.shared.kb_backend.metrics import ( + METRIC_DUAL_READ_FAILED, + METRIC_DUAL_READ_LATENCY, + METRIC_DUAL_READ_OVERLAP, + METRIC_DUAL_READ_RANK_CORRELATION, + emit_count, + emit_value, +) +from apis.shared.kb_backend.protocol import DEFAULT_TOP_K, Chunk +from apis.shared.kb_backend.records import ENGINE_MANAGED + +logger = logging.getLogger(__name__) + +#: KB_Record attribute that opts one knowledge base into the pilot. Absence means +#: off (Requirement 18.4), the same convention ``retrievalEngine`` uses: the +#: default costs nothing to express and nothing to revert. +DUAL_READ_ATTR = "dualReadPilot" + +#: Strong references to detached comparison tasks. See the module docstring. +_IN_FLIGHT: Set["asyncio.Task[None]"] = set() + + +def is_pilot_enabled(record: Optional[Mapping[str, Any]]) -> bool: + """Whether this knowledge base is opted into the pilot. + + Strictly ``is True``: a truthy string left behind by a hand-edited record must + not enrol a knowledge base into paying for a second retrieval on every turn. + The same reasoning armed the reconciler's flag, where a permissive read of an + event field turned a report-only job into a deleting one. + """ + if not record: + return False + return record.get(DUAL_READ_ATTR) is True + + +@dataclass(frozen=True) +class Comparison: + """One dual read's observation. Serves nothing; describes everything.""" + + legacy_count: int + managed_count: int + overlap_count: int + overlap_ratio: float + rank_correlation: Optional[float] + legacy_ms: float + managed_ms: float + + def as_log_fields(self) -> Dict[str, Any]: + return { + "legacyCount": self.legacy_count, + "managedCount": self.managed_count, + "overlapCount": self.overlap_count, + "overlapRatio": round(self.overlap_ratio, 4), + "rankCorrelation": ( + None if self.rank_correlation is None else round(self.rank_correlation, 4) + ), + "legacyMs": round(self.legacy_ms, 1), + "managedMs": round(self.managed_ms, 1), + } + + +def _first_positions(chunks: Sequence[Chunk]) -> Dict[str, int]: + """Each ``document_id``'s best rank in a result list, 0-based. + + A document can contribute several chunks, so "the document's rank" is the rank + of its best chunk. Using every chunk instead would let a document with four + passages dominate a correlation over one with a single passage, which measures + chunking rather than agreement. + """ + positions: Dict[str, int] = {} + for index, chunk in enumerate(chunks): + document_id = (chunk.metadata or {}).get("document_id") or chunk.document_id + if document_id and document_id not in positions: + positions[document_id] = index + return positions + + +def _spearman(left: Sequence[float], right: Sequence[float]) -> Optional[float]: + """Pearson correlation of two rank vectors — Spearman, computed by hand. + + Written out rather than pulled from scipy: this package is bundled into + size-constrained Lambda images, and a numerical stack is a large dependency to + add for one dot product. + + ``None`` when fewer than two documents are shared (a correlation over one point + is undefined, not 1.0) or when either vector has zero variance, which is what + happens when both backends return the same single document. + """ + n = len(left) + if n < 2 or n != len(right): + return None + + mean_left = sum(left) / n + mean_right = sum(right) / n + d_left = [value - mean_left for value in left] + d_right = [value - mean_right for value in right] + + covariance = sum(a * b for a, b in zip(d_left, d_right)) + variance_left = sum(a * a for a in d_left) + variance_right = sum(b * b for b in d_right) + + if variance_left == 0 or variance_right == 0: + return None + + return covariance / ((variance_left**0.5) * (variance_right**0.5)) + + +def compare( + legacy: Sequence[Chunk], + managed: Sequence[Chunk], + legacy_ms: float, + managed_ms: float, +) -> Comparison: + """The observation for one dual read (Requirement 18.3). Pure. + + ``overlap_ratio`` is Jaccard — shared documents over the union — chosen because + it is symmetric. A ratio against one side's length would read as agreement when + one backend simply returned fewer documents, which is the case most likely to + occur while the managed corpus is still catching up. + """ + legacy_positions = _first_positions(legacy) + managed_positions = _first_positions(managed) + + legacy_ids = set(legacy_positions) + managed_ids = set(managed_positions) + shared = legacy_ids & managed_ids + union = legacy_ids | managed_ids + + ordered = sorted(shared, key=lambda doc_id: legacy_positions[doc_id]) + correlation = _spearman( + [float(legacy_positions[doc_id]) for doc_id in ordered], + [float(managed_positions[doc_id]) for doc_id in ordered], + ) + + return Comparison( + legacy_count=len(legacy), + managed_count=len(managed), + overlap_count=len(shared), + overlap_ratio=(len(shared) / len(union)) if union else 0.0, + rank_correlation=correlation, + legacy_ms=legacy_ms, + managed_ms=managed_ms, + ) + + +async def _publish(comparison: Comparison) -> None: + """Record the observation. Metrics go to a thread; they are boto3 calls. + + Off the critical path already, but the event loop is shared with every other + in-flight turn, so four synchronous HTTP calls would still be four pauses + everybody pays for. + """ + logger.info(f"dual read comparison: {comparison.as_log_fields()}") + + def _emit() -> None: + emit_value(METRIC_DUAL_READ_OVERLAP, comparison.overlap_ratio, unit="Percent") + if comparison.rank_correlation is not None: + emit_value(METRIC_DUAL_READ_RANK_CORRELATION, comparison.rank_correlation) + emit_value( + METRIC_DUAL_READ_LATENCY, + comparison.legacy_ms, + unit="Milliseconds", + dimensions={"backend": "s3vectors"}, + ) + emit_value( + METRIC_DUAL_READ_LATENCY, + comparison.managed_ms, + unit="Milliseconds", + dimensions={"backend": ENGINE_MANAGED}, + ) + + await asyncio.to_thread(_emit) + + +async def observe( + assistant_id: str, + query: str, + top_k: int, + legacy_chunks: List[Chunk], + legacy_ms: float, + managed_task: "Optional[asyncio.Task[List[Chunk]]]", +) -> None: + """Await the already-running managed search and record the comparison. + + Never raises, and never returns anything a caller could serve. ``managed_task`` + is awaited here rather than started here, so that by the time this runs the + managed call has been in flight for as long as the legacy one took — which is + what makes the two latencies comparable and the pilot non-additive. + """ + if managed_task is None: + return + + started = time.perf_counter() + try: + managed_chunks = await managed_task + except asyncio.CancelledError: + raise + except Exception as exc: + # A managed-side failure is a finding, not an incident: the turn was + # served from legacy before this coroutine ran. + logger.warning( + f"dual read: the managed backend failed for assistant {assistant_id}; " + f"the turn was already served from legacy: {exc}" + ) + await asyncio.to_thread(emit_count, METRIC_DUAL_READ_FAILED) + return + + managed_ms = legacy_ms + (time.perf_counter() - started) * 1000.0 + + try: + await _publish(compare(legacy_chunks, managed_chunks, legacy_ms, managed_ms)) + except Exception as exc: # noqa: BLE001 - observation must not escape + logger.warning(f"dual read: could not record the comparison: {exc}") + + +def start_managed_read( + record: Optional[Mapping[str, Any]], + assistant_id: str, + query: str, + top_k: int = DEFAULT_TOP_K, +) -> "Optional[asyncio.Task[List[Chunk]]]": + """Launch the observational managed search, or return ``None``. + + ``None`` — meaning "no dual read this turn" — for every one of: the knowledge + base is not opted in, this build has no managed backend registered, the record + already names managed as its engine (there would be nothing to compare + against), or the task could not be created. Each is an ordinary state, so none + of them logs at error level or raises. + + Called *before* the legacy search is awaited, which is the whole basis of + Requirement 18.5. + """ + if not is_pilot_enabled(record): + return None + + from apis.shared.kb_backend.records import resolve_engine + from apis.shared.kb_backend.resolver import backend_for_engine + + if resolve_engine(record) == ENGINE_MANAGED: + # Already promoted: the managed backend is the one being served, so a + # "comparison" would be the same call twice at twice the price. + return None + + managed = backend_for_engine(ENGINE_MANAGED) + if managed is None: + return None + + try: + task = asyncio.create_task(managed.search(assistant_id, query, top_k)) + except RuntimeError as exc: + logger.warning(f"dual read: could not start the managed search: {exc}") + return None + + _IN_FLIGHT.add(task) + task.add_done_callback(_IN_FLIGHT.discard) + return task + + +def schedule_observation( + assistant_id: str, + query: str, + top_k: int, + legacy_chunks: List[Chunk], + legacy_ms: float, + managed_task: "Optional[asyncio.Task[List[Chunk]]]", +) -> None: + """Detach :func:`observe` so the caller can return immediately. + + The point of Requirement 18.5 in one function: nothing after this line is + awaited before the user gets their answer. + """ + if managed_task is None: + return + + try: + observer = asyncio.create_task( + observe(assistant_id, query, top_k, legacy_chunks, legacy_ms, managed_task) + ) + except RuntimeError as exc: + logger.warning(f"dual read: could not schedule the comparison: {exc}") + managed_task.cancel() + return + + _IN_FLIGHT.add(observer) + observer.add_done_callback(_IN_FLIGHT.discard) + + +__all__ = [ + "DUAL_READ_ATTR", + "Comparison", + "compare", + "is_pilot_enabled", + "observe", + "schedule_observation", + "start_managed_read", +] diff --git a/backend/src/apis/shared/kb_backend/idleness.py b/backend/src/apis/shared/kb_backend/idleness.py new file mode 100644 index 000000000..50813652d --- /dev/null +++ b/backend/src/apis/shared/kb_backend/idleness.py @@ -0,0 +1,281 @@ +"""When was this knowledge base last actually needed? + +Requirements 22.5, 22.6. Two rules, and both exist because the obvious answer is +wrong in a way that destroys data. + +**Idleness is not retrieval (22.5).** A knowledge base is idle when *nothing* has +needed it, and retrieval is only one of the ways it gets needed. An agent can be +invoked hundreds of times a day and retrieve nothing from its own corpus, because +retrieval only fires when the query matches — so a corpus judged by retrieval alone +looks abandoned precisely when its agent is busiest with questions the documents do +not answer. The follow-up spec's eviction pass would then delete the documents +behind a live agent. So idleness is the maximum of the knowledge base's own +``lastRetrievedAt`` and the ``lastUsedAt`` of any agent bound to it. + +While this phase holds ``App_KB_Id == assistant_id`` there is exactly one bound +agent and it is the assistant itself, so "any bound agent" is one ``METADATA`` +read. That is deliberately written as a maximum over a set rather than a single +lookup: F4 makes the set larger, and a maximum over one element is the same code. + +**Never write a timestamp per retrieval (22.6).** Retrieval is the hot path. The +write is therefore conditional on the stored value being older than a throttle +window, so at most one write lands per window no matter how many turns race — the +same shape as ``assistants.service.bump_last_used_at``, which solved this for +``lastUsedAt`` and is the precedent being followed rather than a second invention. +A conditional write that loses is not a write; it is a rejected update, which is +why calling this on every retrieval is consistent with the requirement. + +Nothing here reclaims anything +------------------------------ +``reclaim`` is reserved in the migration state enum and never entered in this +phase. This module exists now anyway, because the eviction threshold the follow-up +spec has to choose can only be chosen from historical idleness data, and that data +cannot be backfilled — a timestamp nobody recorded in August is not available in +November. + +Feature: managed-kb-migration +Requirements: 22.1, 22.5, 22.6 +""" + +from __future__ import annotations + +import logging +import os +from typing import Any, Iterable, Mapping, Optional + +logger = logging.getLogger(__name__) + +#: One write per knowledge base per day at most. Chosen to match +#: ``bump_last_used_at``'s default: idleness is measured in days, so a finer +#: resolution buys nothing and costs a write per turn. +THROTTLE_HOURS = 24 + +#: Attribute on the KB_Record. +LAST_RETRIEVED_ATTR = "lastRetrievedAt" + +#: Strong references to detached touch tasks. Without this the only reference is +#: the one ``create_task`` returns, and a dropped task can be collected mid-flight. +_IN_FLIGHT: set = set() + + +def _table(): + import boto3 + + return boto3.resource("dynamodb").Table(os.environ["DYNAMODB_ASSISTANTS_TABLE_NAME"]) + + +def _now(): + from datetime import datetime, timezone + + return datetime.now(timezone.utc) + + +def _iso(moment) -> str: + return moment.strftime("%Y-%m-%dT%H:%M:%SZ") + + +def throttle_hours() -> int: + """Resolved at call time, never as a default argument. + + A module constant bound into a signature is captured once at import, so a test + overriding it silently gets the production value. That cost this feature a + 33-second test that ignored its own override. + """ + raw = os.environ.get("KB_LAST_RETRIEVED_THROTTLE_HOURS") + try: + value = int(raw) if raw else THROTTLE_HOURS + except ValueError: + return THROTTLE_HOURS + return max(value, 1) + + +def touch_last_retrieved(assistant_id: str, app_kb_id: str) -> bool: + """Record that this knowledge base served a retrieval. Never raises. + + Returns ``True`` only for the caller whose write actually landed — at most one + per throttle window. Callers do not need the result; it is returned because a + boolean that names the winner is what let ``bump_last_used_at`` hang + resume-on-first-use off the same write, and the reconciler may want the same + hook later. + + Guarded on ``attribute_exists(SK)`` as well as the freshness floor, so this + cannot bring a KB_Record into existence. A legacy knowledge base has no record + and must keep having none: the migration's zero-backfill property is that + nothing writes to these 1,692 rows until their owner opts in, and a metrics + side effect that created rows would break it while looking harmless. + """ + if not os.environ.get("DYNAMODB_ASSISTANTS_TABLE_NAME"): + return False + + try: + from datetime import timedelta + + from apis.shared.kb_backend.records import kb_pk, kb_sk + + now = _now() + floor = _iso(now - timedelta(hours=throttle_hours())) + _table().update_item( + Key={"PK": kb_pk(assistant_id), "SK": kb_sk(app_kb_id)}, + UpdateExpression=f"SET {LAST_RETRIEVED_ATTR} = :now", + ConditionExpression=( + f"attribute_exists(SK) AND (attribute_not_exists({LAST_RETRIEVED_ATTR}) " + f"OR {LAST_RETRIEVED_ATTR} < :floor)" + ), + ExpressionAttributeValues={":now": _iso(now), ":floor": floor}, + ) + return True + except Exception as exc: + code = getattr(exc, "response", {}).get("Error", {}).get("Code") + if code == "ConditionalCheckFailedException": + # Fresh enough, or there is no KB_Record. Both are ordinary. + return False + logger.warning(f"could not record lastRetrievedAt for kb {app_kb_id}: {exc}") + return False + + +def schedule_activity_touch(assistant_id: str, app_kb_id: str) -> None: + """Run :func:`touch_last_retrieved` off the request path. Never raises. + + Retrieval must not wait for a bookkeeping write, nor for its rejection — and + rejection is the *common* case, since at most one write per throttle window + lands. The write goes to a thread because boto3 is synchronous and blocking the + event loop would make every other in-flight turn pay for it. + + Fire-and-forget with a strong reference held until completion, the same pattern + and the same reason as the dual-read pilot: ``create_task`` returns the only + reference, and dropping it lets the task be collected mid-flight, which shows up + as timestamps that silently stop being recorded under load. + + Falls back to doing nothing at all when there is no running loop. A missing + idleness sample is a gap in a baseline metric; an exception here would be a + failed retrieval. + """ + import asyncio + + async def _touch() -> None: + try: + await asyncio.to_thread(touch_last_retrieved, assistant_id, app_kb_id) + except Exception as exc: # noqa: BLE001 - observability only + logger.debug(f"lastRetrievedAt touch skipped for {app_kb_id}: {exc}") + + try: + task = asyncio.create_task(_touch()) + except RuntimeError: + return + + _IN_FLIGHT.add(task) + task.add_done_callback(_IN_FLIGHT.discard) + + +def bound_agent_ids(assistant_id: str, record: Optional[Mapping[str, Any]] = None) -> list: + """The agents bound to this knowledge base. + + One, this phase, and it is the assistant itself (Requirement 6.5). Written as a + list so that the caller below is a maximum over a set today and stays one when + F4 makes the set bigger — the alternative is a single lookup that has to be + rewritten, in the module whose whole point is not to under-report activity. + """ + return [assistant_id] + + +def agent_last_used_at(assistant_id: str) -> Optional[str]: + """The assistant's ``lastUsedAt``, read from its ``METADATA`` row. + + Raw table access rather than the assistants service, for this package's usual + reason: importing ``apis.shared.assistants`` pulls the embeddings stack into a + size-constrained Lambda image. + """ + try: + response = _table().get_item(Key={"PK": f"AST#{assistant_id}", "SK": "METADATA"}) + except Exception as exc: + logger.warning(f"could not read lastUsedAt for assistant {assistant_id}: {exc}") + return None + item = response.get("Item") or {} + for key in ("lastUsedAt", "updatedAt", "createdAt"): + value = item.get(key) + if value: + return str(value) + return None + + +def last_activity_at( + assistant_id: str, + record: Optional[Mapping[str, Any]] = None, + agent_timestamps: Optional[Iterable[Optional[str]]] = None, +) -> Optional[str]: + """The most recent sign of life: retrieval **or** agent use (Requirement 22.5). + + ``agent_timestamps`` lets a caller sweeping many knowledge bases supply values + it has already read instead of paying a ``get_item`` per knowledge base. When + omitted, the bound agents are read here. + + ``None`` means nothing is known — no retrieval recorded and no agent timestamp. + That is **not** the same as "idle since the beginning of time", and callers must + not treat it as such: it is what a knowledge base provisioned an hour ago looks + like. :func:`idle_days` returns ``None`` for it rather than a large number. + """ + candidates = [str((record or {}).get(LAST_RETRIEVED_ATTR) or "") or None] + + if agent_timestamps is None: + candidates.extend( + agent_last_used_at(agent_id) + for agent_id in bound_agent_ids(assistant_id, record) + ) + else: + candidates.extend(agent_timestamps) + + known = [value for value in candidates if value] + if not known: + return None + # ISO-8601 UTC strings compare correctly lexicographically, which is why every + # timestamp in this feature is written in that exact form. + return max(known) + + +def idle_days( + assistant_id: str, + record: Optional[Mapping[str, Any]] = None, + agent_timestamps: Optional[Iterable[Optional[str]]] = None, + now: Optional[str] = None, +) -> Optional[float]: + """Days since the last sign of life, or ``None`` if nothing is known. + + ``None`` rather than a default is the whole point: a knowledge base with no + recorded activity is unmeasured, not maximally idle, and a metric that reported + "very idle" for every freshly provisioned corpus would be exactly the training + signal that makes operators stop reading it. + """ + from datetime import datetime, timezone + + latest = last_activity_at(assistant_id, record, agent_timestamps) + if not latest: + return None + + try: + parsed = datetime.strptime(latest, "%Y-%m-%dT%H:%M:%SZ").replace(tzinfo=timezone.utc) + except ValueError: + try: + parsed = datetime.fromisoformat(latest.replace("Z", "+00:00")) + except ValueError: + logger.warning(f"unparseable activity timestamp {latest!r}; treating as unknown") + return None + + reference = ( + datetime.strptime(now, "%Y-%m-%dT%H:%M:%SZ").replace(tzinfo=timezone.utc) + if now + else _now() + ) + return max((reference - parsed).total_seconds() / 86400.0, 0.0) + + +__all__ = [ + "LAST_RETRIEVED_ATTR", + "THROTTLE_HOURS", + "agent_last_used_at", + "bound_agent_ids", + "idle_days", + "last_activity_at", + "schedule_activity_touch", + "throttle_hours", + "touch_last_retrieved", +] diff --git a/backend/src/apis/shared/kb_backend/managed_backend.py b/backend/src/apis/shared/kb_backend/managed_backend.py new file mode 100644 index 000000000..5d7f6f39b --- /dev/null +++ b/backend/src/apis/shared/kb_backend/managed_backend.py @@ -0,0 +1,635 @@ +"""The Amazon Bedrock Managed Knowledge Base backend, behind the common protocol. + +Retrieval, direct ingestion and document deletion for a managed knowledge base. +Everything here is reachable only for a KB_Record that names +``retrievalEngine == "managed"``, which happens exactly once per knowledge base, +at promotion. + +Scores need no conversion — and must not get one +------------------------------------------------ +``Retrieve`` returns ``score`` as **relevance**: higher is more relevant, which is +already the protocol's canonical direction. So unlike +:mod:`~apis.shared.kb_backend.s3vectors_backend`, this adapter passes the score +through untouched. Applying the legacy adapter's ``relevance_from_distance`` +negation here would invert the ranking, and an inverted ranking raises nothing, +logs nothing and alarms nothing: retrieval keeps returning five chunks and the +answers quietly get worse. The guard is +``tests/property/test_pbt_kb_score_direction.py``. + +``managedSearchConfiguration``, never ``vectorSearchConfiguration`` +------------------------------------------------------------------ +Requirement 11.1. ``vectorSearchConfiguration`` is a real member of +``KnowledgeBaseRetrievalConfiguration`` in the service model, so it passes +client-side validation and then fails at the service with "not supported for +managed knowledge bases". Every retrieval, including the canary the ingestion +consumer runs, would fail together — loudly, but only after deploy. + +Reranking is ``MANAGED``, not ``NONE`` (Requirement 11.2). It measurably separates +scores (0.89/0.38/0.25/0.21/0.19 versus a nearly flat 1.00/0.84/0.78/0.77/0.77 +without it), and that separation is what makes the 2,000-character context cap +defensible: with a flat distribution the cap truncates chunks that were barely +distinguishable from the best one. + +Hybrid search is not configured, and no attempt is made to (Requirement 11.3). +There is no toggle; it is simply how managed search works. + +Document identity: one id, no chunk keys +---------------------------------------- +``customDocumentIdentifier`` is the platform ``document_id`` (Requirement 9.4). +That 1:1 mapping is what lets the status filter above the seam join on a +``document_id`` per chunk, and it retires the whole ``{doc_id}#{chunk_index}`` +scheme on this path — including ``delete_vector_tail`` and the chunk-shrinkage +stash (Requirement 9.6). Deletion is by document id, so there is no tail to +shrink and nothing to stash. + +Two hard limits, both server-enforced +------------------------------------- +* **10 documents per call.** The packaged service model's ``KnowledgeBaseDocuments`` + list carries ``max: 10``, and the same for ``DocumentIdentifiers``. AWS's user + guide claims 25; that claim does not apply to managed knowledge bases and was + disproven server-side. A batch of 11 fails the whole call, so the batch size is + a constant here rather than a caller's choice. +* **10 concurrent document operations per account.** Ingests and deletes share + that budget, so both go through one semaphore rather than each keeping its own. + +``StartIngestionJob`` is never called (Requirement 9.2): it is 0.1 RPS +account-wide and not adjustable, which for a bulk upload means one document every +ten seconds for the entire account. + +Import boundary +--------------- +Module-level imports are stdlib plus this package's own stdlib-only modules; +``boto3`` is imported inside the client factories. See +``tests/architecture/test_kb_backend_boundary.py``. +""" + +from __future__ import annotations + +import asyncio +import logging +import os +import weakref +from typing import Any, Dict, Iterable, List, Mapping, Optional, Sequence, Tuple + +from apis.shared.kb_backend.protocol import DEFAULT_TOP_K, Chunk, DocumentSource + +logger = logging.getLogger(__name__) + +#: Requirement 9.3. Server-enforced; the model's list shape has ``max: 10``. +MAX_DOCUMENTS_PER_CALL = 10 + +#: Requirement 9.5. Ingest and delete operations share one account-wide budget of +#: 10 concurrent operations, so one semaphore covers both. +MAX_CONCURRENT_DOCUMENT_OPERATIONS = 10 + +#: Requirement 11.2. ``MANAGED`` uses the service's reranker; ``NONE`` disables +#: reranking and flattens the score distribution. +RERANKING_MODEL_TYPE = "MANAGED" + +#: The connector all of this platform's managed documents arrive through. +CONTENT_DATA_SOURCE_TYPE = "CUSTOM" + +#: Requirement 11.5. Isolation-critical filters are restricted to exact-match +#: operators. ``startsWith`` and ``stringContains`` are prefix/substring matches: +#: a filter written to isolate ``ast-1`` would also admit ``ast-10``, and the +#: over-match is invisible because the extra results look like ordinary hits. +ISOLATION_SAFE_FILTER_OPERATORS = frozenset({"equals", "in"}) + +#: Where ``customDocumentIdentifier`` surfaces on a retrieval result, in the order +#: tried. ``location.customDocumentLocation.id`` is the authoritative one for a +#: CUSTOM connector; the metadata key is a documented mirror of it. +CUSTOM_IDENTIFIER_METADATA_KEY = "x-amz-bedrock-kb-custom-document-identifier" + + +class ManagedKbError(RuntimeError): + """A managed knowledge base operation could not be performed.""" + + +class ManagedKbNotProvisioned(ManagedKbError): + """The KB_Record carries no AWS identifiers yet. + + Raised rather than provisioning inline: provisioning takes 47–124 s and this + may be a retrieval on a user's turn. The caller decides whether to wait. + """ + + +class UnsafeFilterOperator(ManagedKbError): + """A filter used an operator that cannot be trusted for isolation. + + Raised rather than silently downgraded to ``equals``, which would change the + caller's meaning, or passed through, which would widen the boundary. + """ + + +# ── Clients ────────────────────────────────────────────────────────────────── +def _region() -> str: + return os.environ.get("AWS_REGION") or os.environ.get("AWS_DEFAULT_REGION") or "us-west-2" + + +def bedrock_agent_runtime_client(): + """The data-plane client (``Retrieve``). Imported lazily, deliberately.""" + import boto3 + + return boto3.client("bedrock-agent-runtime", region_name=_region()) + + +def bedrock_agent_client(): + """The control-plane client (ingest/delete documents). Imported lazily.""" + import boto3 + + return boto3.client("bedrock-agent", region_name=_region()) + + +# ── Concurrency bound ──────────────────────────────────────────────────────── +# +# Keyed by event loop rather than module-global, because an ``asyncio.Semaphore`` +# is bound to the loop it is first awaited on: a single module-level instance +# would break the second test (or the second worker) to use a fresh loop. A weak +# key means a finished loop's semaphore is collected with it. +_SEMAPHORES: "weakref.WeakKeyDictionary[Any, asyncio.Semaphore]" = ( + weakref.WeakKeyDictionary() +) + + +def document_operation_semaphore() -> asyncio.Semaphore: + """The shared bound on concurrent ingest/delete operations (Requirement 9.5).""" + loop = asyncio.get_running_loop() + semaphore = _SEMAPHORES.get(loop) + if semaphore is None: + semaphore = asyncio.Semaphore(MAX_CONCURRENT_DOCUMENT_OPERATIONS) + _SEMAPHORES[loop] = semaphore + return semaphore + + +# ── Payload builders ───────────────────────────────────────────────────────── +def validate_isolation_filter(retrieval_filter: Optional[Mapping[str, Any]]) -> None: + """Refuse any filter operator that is not exact-match (Requirement 11.5). + + Recurses through ``andAll`` / ``orAll`` because a compound filter is only as + safe as its least safe leaf, and a ``stringContains`` buried three levels down + is exactly the kind of thing that survives review. + """ + if not retrieval_filter: + return + + for operator, operand in retrieval_filter.items(): + if operator in ("andAll", "orAll"): + for nested in operand or []: + validate_isolation_filter(nested) + continue + if operator not in ISOLATION_SAFE_FILTER_OPERATORS: + raise UnsafeFilterOperator( + f"filter operator {operator!r} is not permitted: an " + f"isolation-critical filter must use one of " + f"{sorted(ISOLATION_SAFE_FILTER_OPERATORS)}. Prefix and substring " + f"operators over-match silently — a filter isolating 'ast-1' also " + f"admits 'ast-10', and the extra results are indistinguishable " + f"from legitimate hits." + ) + + +def retrieval_configuration( + top_k: int = DEFAULT_TOP_K, + retrieval_filter: Optional[Mapping[str, Any]] = None, +) -> Dict[str, Any]: + """Build ``retrievalConfiguration`` for a managed knowledge base. + + ``managedSearchConfiguration`` only. There is no branch that could produce + ``vectorSearchConfiguration``, so it cannot be reintroduced by a stray + condition — only by editing this function, which + ``tests/shared/test_managed_kb_backend.py`` notices. + """ + validate_isolation_filter(retrieval_filter) + + managed: Dict[str, Any] = { + # Requirement 3.1 parity: both backends are asked for the same number. + "numberOfResults": top_k, + "rerankingModelType": RERANKING_MODEL_TYPE, + } + if retrieval_filter: + managed["filter"] = dict(retrieval_filter) + + # No hybrid-search key: it is not configurable and not attempted (Req 11.3). + return {"managedSearchConfiguration": managed} + + +#: Bedrock caps ``DocumentMetadata.inlineAttributes`` at 50 entries (verified in the +#: packaged service model: ``{'min': 1, 'max': 50}``). Exceeding it fails the whole +#: ``IngestKnowledgeBaseDocuments`` call, so one document with chatty metadata would +#: take its entire batch of ten down with it. +MAX_INLINE_ATTRIBUTES = 50 + +#: Keys the platform depends on, kept in preference to caller-supplied ones when +#: truncating. ``document_id`` is load-bearing: the facade's status filter joins on +#: it, so a chunk that arrives without it cannot be matched to its document and +#: would be dropped as unverifiable. +_RESERVED_METADATA_KEYS = ("document_id", "filename") + + +def _inline_attributes(metadata: Mapping[str, Any]) -> List[Dict[str, Any]]: + """Metadata as ``IN_LINE_ATTRIBUTE`` entries, string-valued and bounded. + + Only strings are emitted. Mixed attribute types are a per-key commitment on + Bedrock's side, and the platform's metadata is loosely typed, so coercing + everything to a string keeps a stray ``None`` or ``int`` from poisoning a key + for every future document. + + The list is capped at :data:`MAX_INLINE_ATTRIBUTES`. ``source.metadata`` is + caller-supplied and unbounded, so without this a caller could fail an entire + ten-document batch with one over-decorated document. Reserved keys are emitted + first so truncation cannot drop them — a plain ``sorted()`` would drop by + alphabet, and ``document_id`` sorts after several plausible caller keys. + """ + ordered: List[tuple] = [] + seen = set() + + for key in _RESERVED_METADATA_KEYS: + if key in metadata and metadata[key] is not None: + ordered.append((key, metadata[key])) + seen.add(key) + + for key, value in sorted(metadata.items()): + if key in seen or value is None: + continue + ordered.append((key, value)) + + if len(ordered) > MAX_INLINE_ATTRIBUTES: + dropped = len(ordered) - MAX_INLINE_ATTRIBUTES + logger.warning( + f"document metadata has {len(ordered)} attributes; keeping the first " + f"{MAX_INLINE_ATTRIBUTES} and dropping {dropped} " + f"(Bedrock's inlineAttributes limit)" + ) + ordered = ordered[:MAX_INLINE_ATTRIBUTES] + + return [ + {"key": key, "value": {"type": "STRING", "stringValue": str(value)}} + for key, value in ordered + ] + + +def document_payload( + source: DocumentSource, + *, + bucket: Optional[str] = None, +) -> Dict[str, Any]: + """One entry of the ``documents`` array. + + ``customDocumentIdentifier`` is the platform ``document_id`` verbatim + (Requirement 9.4) — not a derived or prefixed form. It is the join key the + status filter needs and the handle deletion uses, so any transformation here + would have to be reversed in two other places. + + Prefers the S3 location when the source has one: the object is already in the + documents bucket, and Bedrock reading it directly avoids pulling the bytes + through this process. Falls back to inline text for a source that only has + chunks, joining them back into a document because managed ingestion does its + own chunking and pre-chunked input would be re-chunked anyway. + """ + identifier = {"id": source.document_id} + custom: Dict[str, Any] = { + "customDocumentIdentifier": identifier, + } + + if source.s3_key: + resolved = bucket or os.environ.get("S3_ASSISTANTS_DOCUMENTS_BUCKET_NAME") + if not resolved: + raise ManagedKbError( + f"document {source.document_id} has an S3 key but no bucket: pass " + f"bucket= or set S3_ASSISTANTS_DOCUMENTS_BUCKET_NAME" + ) + custom["sourceType"] = "S3_LOCATION" + custom["s3Location"] = {"uri": f"s3://{resolved}/{source.s3_key}"} + elif source.chunks: + custom["sourceType"] = "IN_LINE" + custom["inlineContent"] = { + "type": "TEXT", + "textContent": {"data": "\n\n".join(source.chunks)}, + } + else: + raise ManagedKbError( + f"document {source.document_id} has neither an s3_key nor chunks; " + f"there is nothing to ingest" + ) + + document: Dict[str, Any] = { + "content": {"dataSourceType": CONTENT_DATA_SOURCE_TYPE, "custom": custom} + } + + metadata = {"document_id": source.document_id, "filename": source.filename} + metadata.update({k: v for k, v in source.metadata.items() if k not in metadata}) + attributes = _inline_attributes(metadata) + if attributes: + document["metadata"] = { + "type": "IN_LINE_ATTRIBUTE", + "inlineAttributes": attributes, + } + return document + + +def document_identifier(document_id: str) -> Dict[str, Any]: + """One entry of ``documentIdentifiers`` for a delete. + + Deletion is by the platform document id, full stop. There is no chunk tail to + enumerate and no shrinkage case to handle, because one document is one + document (Requirement 9.6). + """ + return { + "dataSourceType": CONTENT_DATA_SOURCE_TYPE, + "custom": {"id": document_id}, + } + + +def batched(items: Sequence[Any], size: int = MAX_DOCUMENTS_PER_CALL) -> List[List[Any]]: + """Split ``items`` into batches of at most ``size``. + + ``size`` is validated against the server limit rather than trusted. A caller + passing 25 — the number AWS's user guide gives, which does not apply to managed + knowledge bases — would otherwise produce a request that fails as a whole, + losing the other 24 documents along with the 25th. + """ + if size < 1: + raise ValueError("batch size must be at least 1") + if size > MAX_DOCUMENTS_PER_CALL: + raise ValueError( + f"batch size {size} exceeds the server-enforced maximum of " + f"{MAX_DOCUMENTS_PER_CALL} documents per call. AWS's user guide claims " + f"25; that does not apply to managed knowledge bases and an " + f"11-document call fails entirely." + ) + return [list(items[i : i + size]) for i in range(0, len(items), size)] + + +# ── The backend ────────────────────────────────────────────────────────────── +class ManagedKbBackend: + """Retrieval and direct ingestion against a Managed Knowledge Base. + + ``kb_ref`` is the ``App_KB_Id`` (equal to the ``assistant_id`` in this phase), + never an AWS ``knowledgeBaseId``. The AWS identifiers are resolved internally + from the KB_Record on each operation, because a dormancy/rehydration cycle + replaces them and a caller holding one would keep querying a knowledge base + that no longer exists. + + Clients are injectable so tests can stub them; they are created lazily so + constructing this class costs nothing at import time. + """ + + def __init__( + self, + *, + runtime_client=None, + agent_client=None, + locator=None, + bucket: Optional[str] = None, + ) -> None: + self._runtime_client = runtime_client + self._agent_client = agent_client + self._locator = locator + self._bucket = bucket + + # ── plumbing ──────────────────────────────────────────────────────────── + def _runtime(self): + if self._runtime_client is None: + self._runtime_client = bedrock_agent_runtime_client() + return self._runtime_client + + def _agent(self): + if self._agent_client is None: + self._agent_client = bedrock_agent_client() + return self._agent_client + + def _locate(self, kb_ref: str) -> Tuple[str, str]: + """Resolve ``kb_ref`` to ``(awsKbId, awsDataSourceId)``.""" + if self._locator is not None: + located = self._locator(kb_ref) + else: + from apis.shared.kb_backend.records import get_kb_record + + # App_KB_Id == assistant_id in this phase, so one value serves both. + item = get_kb_record(kb_ref, kb_ref) + located = ( + (item.get("awsKbId"), item.get("awsDataSourceId")) if item else (None, None) + ) + + aws_kb_id, aws_data_source_id = located + if not aws_kb_id: + raise ManagedKbNotProvisioned( + f"knowledge base {kb_ref} has no awsKbId: it is not provisioned yet" + ) + return aws_kb_id, aws_data_source_id + + async def _locate_async(self, kb_ref: str) -> Tuple[str, str]: + return await asyncio.to_thread(self._locate, kb_ref) + + # ── retrieval ─────────────────────────────────────────────────────────── + async def search( + self, + kb_ref: str, + query: str, + top_k: int = DEFAULT_TOP_K, + retrieval_filter: Optional[Mapping[str, Any]] = None, + ) -> List[Chunk]: + """Retrieve up to ``top_k`` chunks, best first. + + Ordering is the service's. ``Retrieve`` returns results ranked best-first + and this returns them in that order, with the reported ``score`` as the + canonical ``relevance`` — unconverted, because the directions already + agree. + + The synchronous ``retrieve`` call runs off the event loop (Requirement + 20.7): it was measured at 662–695 ms p50, which is long enough to matter + to every other coroutine sharing the loop. + """ + aws_kb_id, _ = await self._locate_async(kb_ref) + client = self._runtime() + + payload = { + "knowledgeBaseId": aws_kb_id, + "retrievalQuery": {"text": query}, + "retrievalConfiguration": retrieval_configuration(top_k, retrieval_filter), + } + response = await asyncio.to_thread(lambda: client.retrieve(**payload)) + return [self._to_chunk(result) for result in response.get("retrievalResults", [])] + + @staticmethod + def _to_chunk(result: Mapping[str, Any]) -> Chunk: + """Adapt one ``Retrieve`` result. **No score conversion.** + + ``score`` is relevance already: higher is better, which is the protocol's + direction. A missing score stays ``None`` rather than becoming ``0.0``, + for the same reason as in the legacy adapter — a fabricated ``0.0`` would + be indistinguishable from a real score, and on this backend it would rank + the chunk *last* while on the other it would rank first. + """ + metadata = dict(result.get("metadata") or {}) + text = (result.get("content") or {}).get("text", "") + document_id = ManagedKbBackend._document_id(result, metadata) + + # The status filter and the citation formatter above the seam both read + # `metadata["text"]`, which is where the legacy path put it. + metadata.setdefault("text", text) + metadata.setdefault("document_id", document_id) + + return Chunk( + text=text, + relevance=result.get("score"), + document_id=document_id, + metadata=metadata, + key=document_id, + ) + + @staticmethod + def _document_id(result: Mapping[str, Any], metadata: Mapping[str, Any]) -> str: + """Recover the platform ``document_id`` from a retrieval result. + + Deliberately does **not** fall back to the result's own ``documentId``: + that is a service-assigned handle for ``GetDocumentContent``, not the + platform id, and returning it would produce a chunk whose ``document_id`` + looks plausible, joins against no ``DOC#`` record, and is dropped by the + fail-closed status filter — a disappearing-results bug two layers away + from its cause. + """ + location = result.get("location") or {} + custom = location.get("customDocumentLocation") or {} + for candidate in ( + custom.get("id"), + metadata.get(CUSTOM_IDENTIFIER_METADATA_KEY), + metadata.get("document_id"), + ): + if candidate: + return str(candidate) + return "" + + # ── ingestion ─────────────────────────────────────────────────────────── + async def ingest(self, kb_ref: str, source: DocumentSource) -> None: + """Index one document.""" + await self.ingest_documents(kb_ref, [source]) + + async def ingest_documents( + self, + kb_ref: str, + sources: Iterable[DocumentSource], + *, + batch_size: int = MAX_DOCUMENTS_PER_CALL, + ) -> None: + """Index documents with ``IngestKnowledgeBaseDocuments``. + + Batched at 10 and concurrency-bounded at 10 (Requirements 9.3, 9.5). + ``StartIngestionJob`` is never involved (Requirement 9.2). + + No ``clientToken`` is sent, on purpose. Idempotency here comes from + ``customDocumentIdentifier`` being 1:1 with the platform document id, so + re-ingesting a document replaces it. A token derived from the document ids + would look like extra safety and instead swallow a legitimate re-upload of + the same document — silently, since a deduplicated request returns + success. + """ + documents = list(sources) + if not documents: + return + + aws_kb_id, aws_data_source_id = await self._locate_async(kb_ref) + if not aws_data_source_id: + raise ManagedKbNotProvisioned( + f"knowledge base {kb_ref} has no awsDataSourceId: its CUSTOM " + f"connector is not created yet" + ) + + client = self._agent() + payloads = [document_payload(source, bucket=self._bucket) for source in documents] + + await self._run_bounded( + [ + { + "knowledgeBaseId": aws_kb_id, + "dataSourceId": aws_data_source_id, + "documents": batch, + } + for batch in batched(payloads, batch_size) + ], + client.ingest_knowledge_base_documents, + what="IngestKnowledgeBaseDocuments", + ) + + # ── deletion ──────────────────────────────────────────────────────────── + async def delete_document(self, kb_ref: str, document_id: str) -> None: + """Remove one document by its platform id.""" + await self.delete_documents(kb_ref, [document_id]) + + async def delete_documents( + self, + kb_ref: str, + document_ids: Iterable[str], + *, + batch_size: int = MAX_DOCUMENTS_PER_CALL, + ) -> None: + """Remove documents with ``DeleteKnowledgeBaseDocuments``. + + Same batch limit and same shared concurrency budget as ingestion: the + account's 10-concurrent-operation limit counts both together, so deletes + issued alongside ingests must not each get their own allowance. + """ + ids = [document_id for document_id in document_ids if document_id] + if not ids: + return + + aws_kb_id, aws_data_source_id = await self._locate_async(kb_ref) + if not aws_data_source_id: + raise ManagedKbNotProvisioned( + f"knowledge base {kb_ref} has no awsDataSourceId; nothing to delete from" + ) + + client = self._agent() + await self._run_bounded( + [ + { + "knowledgeBaseId": aws_kb_id, + "dataSourceId": aws_data_source_id, + "documentIdentifiers": batch, + } + for batch in batched([document_identifier(i) for i in ids], batch_size) + ], + client.delete_knowledge_base_documents, + what="DeleteKnowledgeBaseDocuments", + ) + + @staticmethod + async def _run_bounded(payloads: Sequence[Mapping[str, Any]], operation, *, what: str) -> None: + """Issue each payload off the event loop, at most 10 in flight. + + The semaphore is acquired *around* the ``to_thread`` call so the bound + counts operations in flight at AWS, not coroutines created here — which is + the number the account limit is expressed in. + """ + semaphore = document_operation_semaphore() + + async def _one(payload: Mapping[str, Any]) -> None: + async with semaphore: + await asyncio.to_thread(lambda: operation(**payload)) + + results = await asyncio.gather( + *(_one(payload) for payload in payloads), return_exceptions=True + ) + failures = [outcome for outcome in results if isinstance(outcome, BaseException)] + if failures: + logger.error(f"{what}: {len(failures)}/{len(payloads)} batches failed") + raise failures[0] + + +__all__ = [ + "CONTENT_DATA_SOURCE_TYPE", + "ISOLATION_SAFE_FILTER_OPERATORS", + "MAX_CONCURRENT_DOCUMENT_OPERATIONS", + "MAX_DOCUMENTS_PER_CALL", + "RERANKING_MODEL_TYPE", + "ManagedKbBackend", + "ManagedKbError", + "ManagedKbNotProvisioned", + "UnsafeFilterOperator", + "batched", + "document_identifier", + "document_operation_semaphore", + "document_payload", + "retrieval_configuration", + "validate_isolation_filter", +] diff --git a/backend/src/apis/shared/kb_backend/metrics.py b/backend/src/apis/shared/kb_backend/metrics.py new file mode 100644 index 000000000..a8da087d8 --- /dev/null +++ b/backend/src/apis/shared/kb_backend/metrics.py @@ -0,0 +1,197 @@ +"""Best-effort custom metrics for the managed knowledge base seam. + +Metric publishing is *observability*, never control flow. Every function here +swallows its own failures: a knowledge base search must not fail because +CloudWatch was briefly unavailable, and a missing metric is a monitoring gap, not +a user-facing error. This matches the convention in +``apis/app_api/kb_sync/dispatcher.py``. + +Namespace +--------- +The namespace must match what the IAM grant allows, or every publish is silently +denied. ``ManagedKbRoleConstruct`` conditions ``cloudwatch:PutMetricData`` on +``cloudwatch:namespace`` equal to ``{projectPrefix}/ManagedKb``, and derives it +from the same ``projectPrefix`` that CDK injects here as ``PROJECT_PREFIX``. The +two must therefore agree; :func:`metric_namespace` is the single reader of that +variable so there is one place to look when they do not. + +Not an ``AWS/...`` namespace, deliberately: CloudWatch reserves every namespace +beginning with ``AWS`` for its own services and rejects writes to them, so such a +grant would authorize nothing while looking correct. Bedrock's own +``AWS/Bedrock/KnowledgeBases`` metrics are a read source, never a write target. + +Import weight +------------- +``boto3`` is imported inside :func:`emit_count`, so importing this module costs +nothing. The seam is imported by a size-constrained Lambda image. +""" + +from __future__ import annotations + +import logging +import os +from typing import Mapping, Optional + +logger = logging.getLogger(__name__) + +#: Emitted when a query was longer than the hard cap and had to be truncated +#: (Requirement 22.3). A non-zero value means users are sending queries that the +#: managed backend would reject outright, which is worth knowing before the +#: engine is switched under them. +METRIC_QUERY_CLAMPED = "KbQueryClamped" + +#: Emitted when the document-status filter could not confirm status and therefore +#: dropped chunks (Requirement 22.4). Distinct from an ordinary empty result: this +#: one means retrieval degraded, not that the corpus had no match. +METRIC_STATUS_FILTER_FAIL_CLOSED = "KbStatusFilterFailClosed" + +#: Emitted when retrieval was refused because the invoking user's access could not +#: be established (Requirement 25.1). Counts both honest denials and +#: check-failed-so-denied, dimensioned by ``reason`` to keep them apart: the first +#: is the system working, the second is a degradation worth alarming on. +METRIC_ACCESS_DENIED = "KbAccessDenied" + +#: Dual-read pilot observations (Requirement 18.3). Values, not counts: the +#: question each answers is "how much do the two backends agree, and at what +#: cost", and a count cannot answer either. +METRIC_DUAL_READ_OVERLAP = "KbDualReadOverlap" +METRIC_DUAL_READ_RANK_CORRELATION = "KbDualReadRankCorrelation" +METRIC_DUAL_READ_LATENCY = "KbDualReadLatency" + +#: Emitted when the observational managed read failed. Never a user-facing +#: failure — the turn was served from legacy before the comparison ran — but a +#: sustained non-zero value is the pilot telling us the engine is not ready. +METRIC_DUAL_READ_FAILED = "KbDualReadFailed" + +#: Fleet gauges (Requirement 22.1), emitted once per reconciler pass rather than +#: per event, because each is a statement about the whole account. +METRIC_KB_COUNT = "KbCount" +METRIC_KB_STORAGE_GB = "KbStorageGB" +METRIC_KB_IDLE_GB = "KbIdleGB" + +#: Days without a sign of life before a knowledge base's bytes count toward +#: :data:`METRIC_KB_IDLE_GB`. A reporting threshold only: nothing reclaims in this +#: phase, and the number the follow-up spec eventually evicts on should be chosen +#: from the distribution this metric records, not inherited from this guess. +IDLE_THRESHOLD_DAYS = 30 + +#: Bytes per gigabyte, decimal — matching how AWS bills storage ($5.00/GB-month), +#: so a dashboard number and an invoice line can be compared without a conversion +#: nobody remembers to apply. +BYTES_PER_GB = 1_000_000_000 + + +def emit_fleet_gauges( + kb_count: int, + stored_bytes: int, + idle_bytes: int, + *, + unmeasured: int = 0, + idle_threshold_days: Optional[int] = None, +) -> None: + """Publish the account-wide knowledge base gauges. Never raises. + + Requirement 22.1. Emitted through EMF rather than ``PutMetricData`` because the + caller is a Lambda whose stdout already reaches CloudWatch Logs, so this needs + no client, no batching and no IAM — and because these are gauges published once + per pass, which is exactly the shape EMF is good at. The namespace is the same + :func:`metric_namespace` the ``PutMetricData`` grant is conditioned on, so both + mechanisms land in one place and a dashboard does not have to know which code + path produced a number. + + ``unmeasured`` rides along as a log property, not a metric: it is the count of + knowledge bases with no recorded activity at all, which is context for reading + ``KbIdleGB`` rather than something to alarm on. Emitting it as a metric would + invite an alarm on a number that is legitimately large the day this ships and + legitimately near zero a month later. + """ + try: + from apis.shared.observability.emf import emit_emf_metrics + + emit_emf_metrics( + metric_namespace(), + { + METRIC_KB_COUNT: int(kb_count), + METRIC_KB_STORAGE_GB: round(stored_bytes / BYTES_PER_GB, 6), + METRIC_KB_IDLE_GB: round(idle_bytes / BYTES_PER_GB, 6), + }, + properties={ + "unmeasuredKnowledgeBases": int(unmeasured), + "idleThresholdDays": int( + IDLE_THRESHOLD_DAYS if idle_threshold_days is None else idle_threshold_days + ), + }, + units={ + METRIC_KB_COUNT: "Count", + METRIC_KB_STORAGE_GB: "Gigabytes", + METRIC_KB_IDLE_GB: "Gigabytes", + }, + ) + except Exception as exc: # noqa: BLE001 - observability must not break a sweep + logger.warning(f"Failed to emit knowledge base fleet gauges: {exc}") + + +def metric_namespace() -> str: + """The custom namespace this feature publishes into. + + Prefers ``MANAGED_KB_METRIC_NAMESPACE``, which the CDK construct sets from the + *same* helper that builds the IAM condition — so where that variable is present + the grant and the publish cannot disagree. Falls back to deriving from + ``PROJECT_PREFIX`` for services that do not receive it and for local runs. + """ + explicit = os.environ.get("MANAGED_KB_METRIC_NAMESPACE") + if explicit: + return explicit + prefix = os.environ.get("PROJECT_PREFIX", "agentcore") + return f"{prefix}/ManagedKb" + + +def emit_count( + metric_name: str, + value: int = 1, + dimensions: Optional[Mapping[str, str]] = None, +) -> None: + """Publish a single count metric. Never raises. + + Swallowing the failure is the point: the caller is on a request path, and a + metric that cannot be published is strictly less important than the answer the + user is waiting for. + """ + _publish(metric_name, value, "Count", dimensions) + + +def emit_value( + metric_name: str, + value: float, + unit: str = "None", + dimensions: Optional[Mapping[str, str]] = None, +) -> None: + """Publish a measurement rather than an occurrence. Never raises. + + Separate from :func:`emit_count` so the unit is a decision at the call site. + A latency published as ``Count`` is not merely mislabelled — CloudWatch will + graph and alarm on it as a rate, and the mistake is invisible until somebody + tries to read the dashboard. + """ + _publish(metric_name, value, unit, dimensions) + + +def _publish( + metric_name: str, + value: float, + unit: str, + dimensions: Optional[Mapping[str, str]], +) -> None: + try: + import boto3 + + datum: dict = {"MetricName": metric_name, "Value": value, "Unit": unit} + if dimensions: + datum["Dimensions"] = [ + {"Name": k, "Value": v} for k, v in sorted(dimensions.items()) + ] + boto3.client("cloudwatch").put_metric_data( + Namespace=metric_namespace(), MetricData=[datum] + ) + except Exception as exc: # noqa: BLE001 - observability must not break retrieval + logger.warning(f"Failed to emit {metric_name} metric: {exc}") diff --git a/backend/src/apis/shared/kb_backend/protocol.py b/backend/src/apis/shared/kb_backend/protocol.py new file mode 100644 index 000000000..9f2955c32 --- /dev/null +++ b/backend/src/apis/shared/kb_backend/protocol.py @@ -0,0 +1,155 @@ +"""The one seam every knowledge base read and write passes through. + +Two backends sit behind :class:`KnowledgeBaseBackend`: the legacy Amazon S3 +Vectors implementation and, from task 8, Amazon Bedrock Managed Knowledge Base. +Callers never learn which one they got. + +Score direction +--------------- +This module's single most important decision is the name of one field. + +The two backends disagree about which direction is better: + +* S3 Vectors returns cosine **distance** — *lower* is more similar. +* Managed KB returns **relevance** — *higher* is more relevant. + +Get that backwards and nothing raises. No log line, no alarm, no failing +request: retrieval simply serves the least relevant chunks it can find, and the +only symptom is that answers get worse. There is no error path to catch because +there is no error. That is why the canonical field is named ``relevance`` and +documented here rather than left implicit, why the legacy adapter converts +*inside itself* rather than at some call site, and why +``tests/property/test_pbt_kb_score_direction.py`` exists at all. + +Exact round-trip +---------------- +:func:`relevance_from_distance` and :func:`distance_from_relevance` are exact +inverses under IEEE-754, not approximate ones, because the facade still emits a +``distance`` key derived from ``relevance`` and +``app_api/assistants/routes.py`` puts that value in an HTTP response body. A +``1.0 - x`` conversion would round-trip ``0.1`` to ``0.09999999999999998`` and +change a value a client already reads. Negation is exact for every finite float, +so the derived value is the same value, not a very close one. + +Import boundary +--------------- +Module-level imports here are **stdlib only**. ``apis.shared.kb_backend`` is +bundled into size-constrained Lambda images and must not drag in +``apis.shared.assistants`` (whose ``__init__`` imports the embeddings stack) or +``boto3``. Enforced by ``tests/architecture/test_kb_backend_boundary.py``. +""" + +from __future__ import annotations + +from dataclasses import dataclass, field +from typing import Any, Dict, List, Optional, Protocol, runtime_checkable + +#: Parity contract, Requirement 3.1: both backends are asked for five chunks. +#: Named here, above the seam, so neither adapter can drift from the other. +DEFAULT_TOP_K = 5 + + +def relevance_from_distance(distance: Optional[float]) -> Optional[float]: + """Convert a cosine **distance** (lower is better) to **relevance**. + + Negation, deliberately, rather than ``1.0 - distance``: it inverts the + direction — which is the entire job — while being exactly reversible by + :func:`distance_from_relevance` for every finite float. Nothing compares + relevance values *across* backends (the dual-read pilot compares rank order, + not magnitude), so the absolute range is free and exactness is not. + + ``None`` passes through as ``None``. The S3 Vectors query always asks for + distances, so a missing one means a malformed response; the facade's + long-standing behaviour is to emit ``distance: None`` rather than invent a + score, and fabricating ``0.0`` here would promote such a chunk to + best-in-class. + """ + if distance is None: + return None + return -distance + + +def distance_from_relevance(relevance: Optional[float]) -> Optional[float]: + """Recover the original distance from a relevance. Exact inverse of above.""" + if relevance is None: + return None + return -relevance + + +@dataclass(frozen=True) +class Chunk: + """One retrieved passage, in the shape every backend must produce. + + Frozen because a chunk crosses the seam as a value: an adapter that returned + something a caller could mutate would let ranking be edited after the + backend had decided it. + """ + + text: str + + #: Canonical score. **HIGHER IS MORE RELEVANT**, on both backends, always. + #: The legacy adapter has already converted S3 Vectors' inverted distance by + #: the time a chunk exists. ``None`` means the backend reported no score + #: (see :func:`relevance_from_distance`); it is preserved, never defaulted, + #: because a default would be indistinguishable from a real score. + relevance: Optional[float] + + #: Platform document id. The status filter joins on this, and on the managed + #: path it is the ``customDocumentIdentifier`` (task 8.4). + document_id: str + + metadata: Dict[str, Any] = field(default_factory=dict) + + #: Backend-native identifier: ``{document_id}#{chunk_index}`` on legacy. + key: str = "" + + +@dataclass(frozen=True) +class DocumentSource: + """A document to ingest, in the least-common-denominator form. + + The two backends want different things — legacy needs text already chunked, + managed needs the source bytes or an S3 location — so both are optional here + and each adapter validates what it needs. Tasks 8.4 and 9.1 extend this; + it exists now only so the protocol's ``ingest`` signature is real rather + than ``Any``. + """ + + document_id: str + filename: str + chunks: Optional[List[str]] = None + s3_key: Optional[str] = None + metadata: Dict[str, Any] = field(default_factory=dict) + + +@runtime_checkable +class KnowledgeBaseBackend(Protocol): + """What both backends implement, and all a caller may assume. + + ``kb_ref`` is the application-owned reference to the knowledge base — the + ``App_KB_Id``, which equals the ``assistant_id`` in this phase. Never an AWS + ``knowledgeBaseId``: those are replaceable across a dormancy/rehydration + cycle, so an adapter resolves one internally and no caller holds one. + + ``runtime_checkable`` supports ``isinstance`` structural assertions in the + tests. It checks method *presence* only, never signatures, so it is a + guard against a missing method, not a substitute for reading the protocol. + """ + + async def search(self, kb_ref: str, query: str, top_k: int = DEFAULT_TOP_K) -> List[Chunk]: + """Return up to ``top_k`` chunks, best first, scored by relevance. + + Ordering is the backend's: both underlying APIs return results ranked + best-first, and an adapter re-sorting them would be inventing a ranking + rather than reporting one. What an adapter *must* guarantee is that its + ``relevance`` values agree with the order it returns. + """ + ... + + async def ingest(self, kb_ref: str, source: DocumentSource) -> None: + """Index ``source`` into the knowledge base.""" + ... + + async def delete_document(self, kb_ref: str, document_id: str) -> None: + """Remove every trace of ``document_id`` from the knowledge base.""" + ... diff --git a/backend/src/apis/shared/kb_backend/provisioning.py b/backend/src/apis/shared/kb_backend/provisioning.py new file mode 100644 index 000000000..4bdda009f --- /dev/null +++ b/backend/src/apis/shared/kb_backend/provisioning.py @@ -0,0 +1,662 @@ +"""Lazy provisioning of a Managed Knowledge Base, and its CUSTOM connector. + +A knowledge base is created in AWS the first time a document is actually ready to +be indexed — never when an assistant is created (Requirement 7.1). Creation was +measured at 47–124 s to ``ACTIVE`` (n=7), so this never sits on an interactive +path, and every call here is issued off the event loop (Requirement 20.7). + +Order of operations, which is the whole design +---------------------------------------------- +The KB_Record is written in ``provisioning`` **before** the first AWS call and the +returned identifiers are attached afterwards with a conditional update +(Requirement 7.3). Reversing those two steps looks harmless and is not: a crash +between ``CreateKnowledgeBase`` returning and the record being written would leave +a billed AWS resource that no record points at and no code can find. Nothing +raises, nothing alarms, and the only evidence is the invoice. + +Written record-first, the same crash leaves a ``provisioning`` record that is a +durable **retry anchor** (Requirement 7.8): the Reconciler can match it against +the orphan and adopt it, and a plain retry of this function reuses the persisted +``clientToken`` so AWS deduplicates rather than creating a second knowledge base. + +Four details that are each a defect if omitted +---------------------------------------------- +1. **The ``clientToken`` is built, not interpolated.** The API's minimum is **33 + characters** — verified in the packaged service model, ``ClientToken`` has + ``min: 33``, ``max: 256``, ``pattern: [a-zA-Z0-9](-*[a-zA-Z0-9]){0,256}``. The + natural ``{id}-{variant}-kb`` template is 31 characters and fails *client-side* + validation, before a request is ever sent. :func:`build_client_token` + constructs one that cannot be too short, and :func:`validate_client_token` + refuses one that is. + +2. **"Unable to verify the specified embedding model" is retryable.** It was + observed against a model confirmed ``ACTIVE`` and directly invokable: it is IAM + eventual consistency, not a configuration error. Treated as fatal, lazy + provisioning fails intermittently while pointing at the wrong cause — an + operator reads the message and goes to check the model. + +3. **``dataDeletionPolicy: RETAIN`` at creation.** The documented remedy for the + ``DELETE_UNSUCCESSFUL`` state, which the dev account has already been sitting + in since 2025-11-24. Set deliberately up front, not as incident response. + +4. **``imageExtractionStatus: ENABLED``.** Opt-in. Left at its default, chart and + image content is never described and never indexed — a silent loss of a + capability being paid for, with no error anywhere. + +Why ``storageConfiguration`` is absent +-------------------------------------- +There is no vector store to provision: that is the point of a managed knowledge +base. Sending ``storageConfiguration`` at all is rejected (Requirement 8.2), so it +is omitted entirely rather than passed empty. + +Reconciling "``managedKnowledgeBaseConfiguration={}``" with the embedding pin +---------------------------------------------------------------------------- +Requirement 8.1 records that ``managedKnowledgeBaseConfiguration`` has **no +required members**, so ``{}`` is a valid value. Requirement 8.5 separately pins +``embeddingModelType: CUSTOM`` to ``amazon.titan-embed-text-v2:0`` at float32 and +1024 dimensions. Those are not in conflict: the packaged service model puts +``embeddingModelType`` / ``embeddingModelArn`` / +``embeddingModelConfiguration.bedrockEmbeddingModelConfiguration`` inside +``managedKnowledgeBaseConfiguration`` as optional members, so the pin goes there. +It is pinned rather than left to the service default because the choice is +**immutable after creation** (Requirement 8.8) and because keeping today's Titan +v2 embedding preserves continuity with the legacy corpus. + +Import boundary +--------------- +Module-level imports are stdlib plus this package's own stdlib-only modules. +``boto3`` is imported inside the function that builds a client, so importing this +module into a size-constrained Lambda image costs nothing. See +``tests/architecture/test_kb_backend_boundary.py``. +""" + +from __future__ import annotations + +import asyncio +import hashlib +import logging +import os +import re +from dataclasses import dataclass +from datetime import datetime, timezone +from typing import Any, Awaitable, Callable, Dict, Mapping, Optional + +from apis.shared.kb_backend.metrics import emit_count + +logger = logging.getLogger(__name__) + +# ── Immutable embedding configuration (Requirement 8.5, 8.8) ───────────────── +# +# Immutable after creation: Bedrock rejects a change, so a drift here is not a +# migration but a rebuild. Recorded on the KB_Record too, so a mismatch is +# detectable rather than mysterious. +EMBEDDING_MODEL_ID = "amazon.titan-embed-text-v2:0" +EMBEDDING_DIMENSIONS = 1024 + +#: The service model's ``EmbeddingDataType`` enum is ``['FLOAT32', 'BINARY']`` — +#: upper case. "float32" is rejected. +EMBEDDING_DATA_TYPE = "FLOAT32" +EMBEDDING_MODEL_TYPE = "CUSTOM" + +# ── Knowledge base and data source shapes ──────────────────────────────────── +KNOWLEDGE_BASE_TYPE = "MANAGED" + +#: Classic ``CUSTOM`` / ``S3`` / ``WEB`` at the top level are rejected with +#: "Unsupported data source type for MANAGED knowledge base type". The real +#: connector type nests one level down, in ``connectorParameters``. +DATA_SOURCE_TYPE = "MANAGED_KNOWLEDGE_BASE_CONNECTOR" +CONNECTOR_TYPE = "CUSTOM" +CONNECTOR_VERSION = "1" + +#: Requirement 8.7. ``RETAIN`` at creation time, not later. +DATA_DELETION_POLICY = "RETAIN" + +#: Requirement 8.6. Opt-in; the default indexes no image or chart content. +IMAGE_EXTRACTION_STATUS = "ENABLED" + +# ── clientToken (Requirement 7.5, 7.6) ────────────────────────────────────── +CLIENT_TOKEN_MIN_LENGTH = 33 +CLIENT_TOKEN_MAX_LENGTH = 256 + +#: Anchored copy of the service model's pattern. Note what it permits and does +#: not: the first and last characters must be alphanumeric, so a token may not +#: begin or end with a hyphen, though hyphens may run consecutively inside. +CLIENT_TOKEN_PATTERN = re.compile(r"^[a-zA-Z0-9](-*[a-zA-Z0-9]){0,256}$") + +#: Length of the deterministic digest suffix. 40 hex characters alone clears the +#: 33-character minimum, so a token is long enough even when the caller's parts +#: sanitize away to nothing. +_DIGEST_CHARS = 40 + +# ── Retry classification (Requirement 7.7) ────────────────────────────────── +# +# Matched against the *message*, lower-cased, because this failure arrives as an +# ordinary validation-shaped error and is indistinguishable from a real +# misconfiguration by error code alone. +RETRYABLE_MESSAGE_FRAGMENTS = ( + "unable to verify the specified embedding model", + "unable to verify the embedding model", +) + +#: Transport-level errors that are retryable for the usual reasons. Deliberately +#: narrow: a genuine ``ValidationException`` must fail fast and loudly, because +#: retrying a malformed request just delays the report by a minute. +RETRYABLE_ERROR_CODES = ( + "ThrottlingException", + "TooManyRequestsException", + "InternalServerException", + "ServiceUnavailableException", +) + +MAX_PROVISION_ATTEMPTS = 5 +_MAX_BACKOFF_SECONDS = 30.0 + +METRIC_PROVISION_RETRIED = "KbProvisionRetried" +METRIC_PROVISION_ADOPTED = "KbProvisionAdopted" + + +class ProvisioningError(RuntimeError): + """Provisioning could not complete.""" + + +class RetryableProvisioningError(ProvisioningError): + """Provisioning failed for a reason that will plausibly clear on its own.""" + + +class ProvisioningInProgress(RetryableProvisioningError): + """Another worker owns this provisioning and has not finished. + + Retryable rather than fatal, and deliberately *not* an attempt to provision + anyway: two workers creating one knowledge base each is exactly the + duplication Requirement 7.4 forbids. + """ + + +@dataclass(frozen=True) +class ProvisionedKnowledgeBase: + """The identifiers a caller needs, plus how they were obtained. + + ``created`` distinguishes "this call made the AWS resource" from "this call + found one already recorded", which is what makes idempotency assertable + instead of assumed. + """ + + aws_kb_id: str + aws_data_source_id: str + client_token: str + created: bool = False + + +# ── Payload builders ───────────────────────────────────────────────────────── +def build_client_token(*parts: Any) -> str: + """Build a ``clientToken`` that satisfies the API's constraints by construction. + + Deterministic in its inputs, so a retry of the same provisioning produces the + same token and AWS deduplicates the create rather than making a second + knowledge base. That property is the reason the token is also persisted on the + KB_Record: a *later* process, with no memory of this one, must be able to + reproduce the retry. + + The caller's parts are sanitized to the permitted alphabet and a digest of the + original seed is appended. The digest is not decoration — it is what + guarantees the 33-character minimum regardless of how short the inputs are, + which is the failure the natural ``{id}-{variant}-kb`` template walks into at + 31 characters. + """ + seed = "-".join(str(part) for part in parts if str(part) != "") + digest = hashlib.sha256(seed.encode("utf-8")).hexdigest()[:_DIGEST_CHARS] + + sanitized = re.sub(r"[^a-zA-Z0-9-]", "-", seed).strip("-") + token = f"{sanitized}-{digest}" if sanitized else digest + + if len(token) > CLIENT_TOKEN_MAX_LENGTH: + # Truncate from the left of the *prefix*, never the digest: the digest is + # what carries the uniqueness, so a token trimmed to its prefix could + # collide with a different knowledge base's. + keep = CLIENT_TOKEN_MAX_LENGTH - len(digest) - 1 + token = f"{sanitized[:keep].rstrip('-')}-{digest}" + + validate_client_token(token) + return token + + +def validate_client_token(token: str) -> None: + """Refuse a token the API would refuse, with a message saying which rule. + + Checked locally because botocore validates ``min``/``max``/``pattern`` + client-side: a short token never reaches AWS, so there is no service error to + read and no request id to quote. Raising here, naming the length, is the + difference between a one-line fix and an afternoon. + """ + if not isinstance(token, str): + raise ValueError(f"clientToken must be a string, got {type(token).__name__}") + if len(token) < CLIENT_TOKEN_MIN_LENGTH: + raise ValueError( + f"clientToken must be at least {CLIENT_TOKEN_MIN_LENGTH} characters " + f"(the API's documented minimum); got {len(token)}: {token!r}. Build " + f"tokens with build_client_token() rather than interpolating a " + f"template — the natural '{{id}}-{{variant}}-kb' form is 31 characters " + f"and fails botocore's client-side validation before any request." + ) + if len(token) > CLIENT_TOKEN_MAX_LENGTH: + raise ValueError( + f"clientToken must be at most {CLIENT_TOKEN_MAX_LENGTH} characters; " + f"got {len(token)}" + ) + if not CLIENT_TOKEN_PATTERN.match(token): + raise ValueError( + f"clientToken {token!r} does not match the API's pattern " + f"{CLIENT_TOKEN_PATTERN.pattern}: it must begin and end with an " + f"alphanumeric character" + ) + + +def embedding_model_arn(region: Optional[str] = None) -> str: + """ARN of the pinned embedding model. + + Foundation-model ARNs carry no account, hence the empty account segment. + """ + return f"arn:aws:bedrock:{region or _region()}::foundation-model/{EMBEDDING_MODEL_ID}" + + +def build_tags( + app_kb_id: str, + owner_user_id: str, + project_prefix: Optional[str] = None, + environment: Optional[str] = None, +) -> Dict[str, str]: + """Tags the Reconciler and the teardown script both read (Requirement 20.11). + + Delegates to :mod:`apis.shared.kb_backend.tags`, which owns the key names and + the value resolution. Kept as a thin wrapper because the provisioning saga is + the only caller and this is where a reader looks for it. + + ⚠️ This function used to build the tags itself, with keys ``prefix``/``env`` + and values from ``PROJECT_PREFIX``/``ENVIRONMENT`` — neither of which the + provisioning Lambda receives. Every knowledge base would have been tagged with + the hardcoded defaults, the teardown script (which read a different pair of + variables) would have matched nothing, and two environments in one account + would have claimed each other's corpora. See the ``tags`` module docstring. + """ + from apis.shared.kb_backend.tags import build_tags as _canonical + + return _canonical(app_kb_id, owner_user_id, project_prefix, environment) + + +def knowledge_base_payload( + name: str, + role_arn: str, + client_token: str, + *, + description: Optional[str] = None, + tags: Optional[Mapping[str, str]] = None, + region: Optional[str] = None, + kms_key_arn: Optional[str] = None, +) -> Dict[str, Any]: + """The exact ``CreateKnowledgeBase`` request (Requirements 8.1, 8.2, 8.5). + + ``storageConfiguration`` is absent by construction — there is no key to + accidentally set to ``None``, because a managed knowledge base has no vector + store and sending one is rejected. + """ + validate_client_token(client_token) + + managed: Dict[str, Any] = { + # Requirement 8.5: pinned, and immutable from here on. + "embeddingModelType": EMBEDDING_MODEL_TYPE, + "embeddingModelArn": embedding_model_arn(region), + "embeddingModelConfiguration": { + "bedrockEmbeddingModelConfiguration": { + "dimensions": EMBEDDING_DIMENSIONS, + "embeddingDataType": EMBEDDING_DATA_TYPE, + } + }, + } + if kms_key_arn: + # Requirement 20.5, only where customer-managed encryption is required. + managed["serverSideEncryptionConfiguration"] = {"kmsKeyArn": kms_key_arn} + + payload: Dict[str, Any] = { + "name": name, + "roleArn": role_arn, + "clientToken": client_token, + "knowledgeBaseConfiguration": { + "type": KNOWLEDGE_BASE_TYPE, + "managedKnowledgeBaseConfiguration": managed, + }, + } + if description: + payload["description"] = description + if tags: + payload["tags"] = dict(tags) + return payload + + +def data_source_payload( + knowledge_base_id: str, + name: str, + client_token: str, + *, + description: Optional[str] = None, +) -> Dict[str, Any]: + """The exact ``CreateDataSource`` request (Requirements 8.3, 8.4, 8.6, 8.7). + + Note the two-level nesting of the connector type. Putting ``CUSTOM`` at the + top level — which is what every pre-managed example does — is rejected with + "Unsupported data source type for MANAGED knowledge base type". + """ + validate_client_token(client_token) + + payload: Dict[str, Any] = { + "knowledgeBaseId": knowledge_base_id, + "name": name, + "clientToken": client_token, + # Requirement 8.7 — at creation, because it cannot rescue a knowledge base + # that is already stuck in DELETE_UNSUCCESSFUL. + "dataDeletionPolicy": DATA_DELETION_POLICY, + "dataSourceConfiguration": { + "type": DATA_SOURCE_TYPE, + "managedKnowledgeBaseConnectorConfiguration": { + "connectorParameters": { + "type": CONNECTOR_TYPE, + "version": CONNECTOR_VERSION, + }, + # Requirement 8.6 — opt-in, and silent when omitted. + "mediaExtractionConfiguration": { + "imageExtractionConfiguration": { + "imageExtractionStatus": IMAGE_EXTRACTION_STATUS + } + }, + }, + }, + } + if description: + payload["description"] = description + return payload + + +# ── Retry classification ───────────────────────────────────────────────────── +def is_retryable_error(exc: BaseException) -> bool: + """Whether ``exc`` should be retried rather than surfaced (Requirement 7.7). + + The embedding-verification message is the interesting case. It arrives looking + like a configuration error, which is why a first reading of it produces code + that fails the whole provisioning and tells the operator to check a model that + is demonstrably fine. It is IAM eventual consistency: the role's + ``bedrock:InvokeModel`` grant has not propagated yet. + """ + message = str(exc).lower() + if any(fragment in message for fragment in RETRYABLE_MESSAGE_FRAGMENTS): + return True + + response = getattr(exc, "response", None) + if isinstance(response, Mapping): + code = response.get("Error", {}).get("Code") + if code in RETRYABLE_ERROR_CODES: + return True + return False + + +# ── Clients and clocks ─────────────────────────────────────────────────────── +def _region() -> str: + return os.environ.get("AWS_REGION") or os.environ.get("AWS_DEFAULT_REGION") or "us-west-2" + + +def bedrock_agent_client(): + """A ``bedrock-agent`` control-plane client. Imported lazily, deliberately.""" + import boto3 + + return boto3.client("bedrock-agent", region_name=_region()) + + +def _now_iso() -> str: + return datetime.now(timezone.utc).isoformat(timespec="seconds").replace("+00:00", "Z") + + +async def _call( + operation: Callable[..., Any], + payload: Mapping[str, Any], + *, + what: str, + max_attempts: int = MAX_PROVISION_ATTEMPTS, + sleep: Callable[[float], Awaitable[None]] = asyncio.sleep, +) -> Any: + """Invoke a synchronous boto3 operation off the event loop, with retries. + + ``asyncio.to_thread`` rather than a direct call because this runs inside the + async request path and ``CreateKnowledgeBase`` blocks for 47–124 s + (Requirement 20.7). Called directly it would stall every other coroutine on + the loop — including the health check that decides whether the task is alive. + """ + for attempt in range(1, max_attempts + 1): + try: + return await asyncio.to_thread(lambda: operation(**payload)) + except Exception as exc: + if attempt >= max_attempts or not is_retryable_error(exc): + raise + delay = min(2.0**attempt, _MAX_BACKOFF_SECONDS) + logger.warning( + f"{what} failed with a retryable error (attempt {attempt}/" + f"{max_attempts}, retrying in {delay}s): {exc}" + ) + emit_count(METRIC_PROVISION_RETRIED, dimensions={"operation": what}) + await sleep(delay) + + # Unreachable: the loop either returns or raises. + raise ProvisioningError(f"{what} exhausted {max_attempts} attempts") + + +# ── The saga ───────────────────────────────────────────────────────────────── +def _complete(item: Mapping[str, Any]) -> bool: + return bool(item.get("awsKbId")) and bool(item.get("awsDataSourceId")) + + +def _resource_name(app_kb_id: str, project_prefix: Optional[str] = None) -> str: + """The knowledge base's AWS name. + + Resolved through the same helper as the tags, so a knowledge base's name and + its ``ManagedKbPrefix`` tag can never disagree. The name is only a convention — + every filter in this feature matches on tags — but a name that says ``prod`` + while the tag says ``dev`` is the kind of thing an operator reads once and + trusts. + """ + from apis.shared.kb_backend.tags import tag_prefix + + return f"{tag_prefix(project_prefix)}-kb-{app_kb_id}" + + +async def provision_managed_kb( + assistant_id: str, + app_kb_id: Optional[str] = None, + owner_user_id: str = "", + *, + role_arn: Optional[str] = None, + client=None, + region: Optional[str] = None, + kms_key_arn: Optional[str] = None, + project_prefix: Optional[str] = None, + environment: Optional[str] = None, + max_attempts: int = MAX_PROVISION_ATTEMPTS, + sleep: Callable[[float], Awaitable[None]] = asyncio.sleep, +) -> ProvisionedKnowledgeBase: + """Provision, or adopt, the managed knowledge base for ``app_kb_id``. + + Safe to call repeatedly and concurrently. Three paths, in the order they are + tried: + + * **Already provisioned** — the record carries both identifiers, so this + returns them and calls nothing. + * **Resuming** — a record exists in ``provisioning``, which is what a crash + between the AWS create and the conditional update leaves behind. Its + persisted ``clientToken`` is reused, so the retried ``CreateKnowledgeBase`` + is deduplicated by AWS and no second knowledge base appears. + * **Fresh** — the record is written first, then AWS is called. + + Losing the ``create_provisioning`` race raises + :class:`ProvisioningInProgress` rather than proceeding. The winner is already + creating the knowledge base; a loser that pressed on with its own token would + create a second one and only one of them could ever be recorded. + """ + from apis.shared.kb_backend import records as r + + app_kb_id = app_kb_id or assistant_id + role_arn = role_arn or os.environ.get("MANAGED_KB_SERVICE_ROLE_ARN") + if not role_arn: + raise ProvisioningError( + "no Bedrock knowledge base service role: pass role_arn or set " + "MANAGED_KB_SERVICE_ROLE_ARN" + ) + + client = client or bedrock_agent_client() + kb_token = build_client_token(app_kb_id, "knowledge-base") + ds_token = build_client_token(app_kb_id, "data-source") + + existing = await asyncio.to_thread(r.get_kb_record, assistant_id, app_kb_id) + + if existing and _complete(existing): + # Idempotent: nothing to create, and nothing to write. + return ProvisionedKnowledgeBase( + aws_kb_id=existing["awsKbId"], + aws_data_source_id=existing["awsDataSourceId"], + client_token=existing.get("clientToken") or kb_token, + created=False, + ) + + if existing: + # The retry-anchor path. Reuse the persisted token: it is the only thing + # that makes the re-create idempotent on AWS's side. + kb_token = existing.get("clientToken") or kb_token + emit_count(METRIC_PROVISION_ADOPTED, dimensions={"appKbId": app_kb_id}) + logger.info( + f"resuming provisioning for kb {app_kb_id} from its existing " + f"{existing.get('provisioningState')} record" + ) + else: + record = r.KbRecord( + app_kb_id=app_kb_id, + owner_user_id=owner_user_id, + provisioning_state=r.PROVISIONING, + client_token=kb_token, + embedding_model_id=EMBEDDING_MODEL_ID, + embedding_dimensions=EMBEDDING_DIMENSIONS, + image_extraction=True, + parser_config={ + "imageExtractionStatus": IMAGE_EXTRACTION_STATUS, + "connectorType": CONNECTOR_TYPE, + "embeddingDataType": EMBEDDING_DATA_TYPE, + }, + ) + try: + # DDB before AWS. See the module docstring; this ordering is the + # difference between a retry anchor and an untraceable paying resource. + await asyncio.to_thread(r.create_provisioning, assistant_id, record) + except r.TransitionLost as exc: + other = await asyncio.to_thread(r.get_kb_record, assistant_id, app_kb_id) + if other and _complete(other): + return ProvisionedKnowledgeBase( + aws_kb_id=other["awsKbId"], + aws_data_source_id=other["awsDataSourceId"], + client_token=other.get("clientToken") or kb_token, + created=False, + ) + raise ProvisioningInProgress( + f"another worker is provisioning kb {app_kb_id}; retry later " + f"rather than creating a second knowledge base" + ) from exc + + name = _resource_name(app_kb_id, project_prefix) + + aws_kb_id = (existing or {}).get("awsKbId") + if not aws_kb_id: + response = await _call( + client.create_knowledge_base, + knowledge_base_payload( + name=name, + role_arn=role_arn, + client_token=kb_token, + description=f"Managed knowledge base for {app_kb_id}", + tags=build_tags(app_kb_id, owner_user_id, project_prefix, environment), + region=region, + kms_key_arn=kms_key_arn, + ), + what="CreateKnowledgeBase", + max_attempts=max_attempts, + sleep=sleep, + ) + aws_kb_id = response["knowledgeBase"]["knowledgeBaseId"] + + aws_data_source_id = (existing or {}).get("awsDataSourceId") + if not aws_data_source_id: + ds_response = await _call( + client.create_data_source, + data_source_payload( + knowledge_base_id=aws_kb_id, + name=name, + client_token=ds_token, + description=f"CUSTOM connector for {app_kb_id}", + ), + what="CreateDataSource", + max_attempts=max_attempts, + sleep=sleep, + ) + aws_data_source_id = ds_response["dataSource"]["dataSourceId"] + + try: + await asyncio.to_thread( + r.attach_aws_ids, + assistant_id, + app_kb_id, + aws_kb_id, + aws_data_source_id, + _now_iso(), + ) + except r.TransitionLost: + # Another worker attached first, or the record has already left + # `provisioning`. Its identifiers win: they are the ones every reader + # will see, so returning our own would hand back a knowledge base that + # no record points at. + current = await asyncio.to_thread(r.get_kb_record, assistant_id, app_kb_id) + if current and _complete(current): + return ProvisionedKnowledgeBase( + aws_kb_id=current["awsKbId"], + aws_data_source_id=current["awsDataSourceId"], + client_token=current.get("clientToken") or kb_token, + created=False, + ) + raise + + return ProvisionedKnowledgeBase( + aws_kb_id=aws_kb_id, + aws_data_source_id=aws_data_source_id, + client_token=kb_token, + created=True, + ) + + +__all__ = [ + "CLIENT_TOKEN_MAX_LENGTH", + "CLIENT_TOKEN_MIN_LENGTH", + "CLIENT_TOKEN_PATTERN", + "CONNECTOR_TYPE", + "CONNECTOR_VERSION", + "DATA_DELETION_POLICY", + "DATA_SOURCE_TYPE", + "EMBEDDING_DATA_TYPE", + "EMBEDDING_DIMENSIONS", + "EMBEDDING_MODEL_ID", + "EMBEDDING_MODEL_TYPE", + "IMAGE_EXTRACTION_STATUS", + "KNOWLEDGE_BASE_TYPE", + "ProvisionedKnowledgeBase", + "ProvisioningError", + "ProvisioningInProgress", + "RetryableProvisioningError", + "build_client_token", + "build_tags", + "data_source_payload", + "embedding_model_arn", + "is_retryable_error", + "knowledge_base_payload", + "provision_managed_kb", + "validate_client_token", +] diff --git a/backend/src/apis/shared/kb_backend/query_guard.py b/backend/src/apis/shared/kb_backend/query_guard.py new file mode 100644 index 000000000..36966a320 --- /dev/null +++ b/backend/src/apis/shared/kb_backend/query_guard.py @@ -0,0 +1,74 @@ +"""Hard cap on retrieval query length. + +Amazon Bedrock Managed Knowledge Base caps ``Retrieve`` query input at **10,000 +characters** and that quota is **not adjustable**. Exceeding it is a request +error, not a degraded result — so an unclamped query is the difference between a +slightly-truncated answer and no answer at all. + +Why this lives above the seam +----------------------------- +The clamp is applied in the facade, before backend dispatch, so both backends see +an identically-shaped query. Clamping only the managed path would mean the two +backends answered *different questions* whenever a query ran long, which would +quietly invalidate the dual-read comparison this migration depends on: a rank +disagreement would be indistinguishable from a genuine retrieval difference. + +That does mean the legacy path is now clamped too, where previously it was not. +Titan v2 tolerates roughly 32,000 characters, so queries between 10,000 and that +ceiling used to be embedded whole and now are not. This is deliberate — parity is +worth more than the tail of a pathological query — and it is why the truncation +emits a metric rather than passing silently. + +Why it never raises +------------------- +A query too long is a fixable input, not a failure. Raising would turn a +recoverable situation into a 500 on a chat turn. The function is total: every +input maps to an output of at most :data:`MAX_QUERY_CHARS` characters. +""" + +from __future__ import annotations + +import logging +from typing import Tuple + +from apis.shared.kb_backend.metrics import METRIC_QUERY_CLAMPED, emit_count + +logger = logging.getLogger(__name__) + +#: Managed KB's ``Retrieve`` input limit. Not adjustable — do not raise this +#: hoping for a quota increase; there is not one to request. +MAX_QUERY_CHARS = 10_000 + + +def clamp_query(query: str) -> Tuple[str, bool]: + """Return ``(clamped_query, was_truncated)``. + + Truncates from the end, keeping the head. For a natural-language query the + beginning carries the intent, so a tail-truncated query still retrieves + something sensible; head-truncating would change the question entirely. + + A ``None`` or non-string input is coerced rather than rejected, because the + caller is a request path and the clamp is a guard, not a validator. + """ + if not query: + return "", False + + if not isinstance(query, str): + query = str(query) + + if len(query) <= MAX_QUERY_CHARS: + return query, False + + original_length = len(query) + clamped = query[:MAX_QUERY_CHARS] + + # Error-level would overstate it (the request still succeeds) and debug would + # hide it. A clamped query means someone's answer is based on a partial + # question, which an operator should be able to see without turning on debug. + logger.warning( + f"Query clamped from {original_length} to {MAX_QUERY_CHARS} characters " + f"(Managed KB Retrieve limit, not adjustable)" + ) + emit_count(METRIC_QUERY_CLAMPED) + + return clamped, True diff --git a/backend/src/apis/shared/kb_backend/records.py b/backend/src/apis/shared/kb_backend/records.py new file mode 100644 index 000000000..342566797 --- /dev/null +++ b/backend/src/apis/shared/kb_backend/records.py @@ -0,0 +1,625 @@ +"""KB_Record persistence for the managed knowledge base migration. + +Records live in the **existing** assistants table as siblings of the assistant's +``METADATA`` row, preserving the adjacency-list convention:: + + PK = AST#{assistant_id} + SK = KB#{app_kb_id} # app_kb_id == assistant_id this phase + SK = KBTOMB#{app_kb_id} # whole-KB tombstone + SK = KBTOMB#{app_kb_id}#DOC#{document_id} # per-document tombstone + +Three invariants are load-bearing. Each is enforced here rather than left to +callers, because each fails silently when violated: + +**1. Absence means legacy.** ``retrievalEngine`` is written *only* as +``"managed"``. Nothing here ever writes ``"s3vectors"`` onto a record that did +not already carry it. That is what makes this migration zero-backfill: every +existing knowledge base is already correct by virtue of having no opinion, and +rollback is a single attribute removal rather than a data rewrite. A backfill +that "helpfully" stamped the legacy value on 1,692 records would convert a +pointer flip into a migration of its own. + +**2. Every transition is conditional.** These functions are called from a +dispatcher that fans out to concurrent workers, so a read-then-write would let +two workers both believe they won. Each transition therefore carries a DynamoDB +``ConditionExpression`` and surfaces the loss as :class:`TransitionLost` rather +than an opaque ``ClientError``. + +**3. Sparse work keys are removed, not just ignored.** ``GSI7_PK``/``GSI7_SK`` +exist only while a record is eligible for background work. On reaching a terminal +state they are ``REMOVE``d, so an ineligible knowledge base is invisible to the +dispatcher's query *by physics* rather than by filter. This matters more than the +usual sparse-index argument because the dispatcher creates and deletes billed AWS +resources: a missing key can only ever mean "do nothing", whereas a stale key +means "act on something nobody asked you to act on". + +Import boundary +--------------- +This module deliberately talks to DynamoDB through the raw table resource instead +of importing ``apis.shared.assistants``. That package's ``__init__`` imports +``rag_service``, which imports the embeddings stack at module scope; pulling it +into the migration Lambda image would blow the image-size budget. The same +constraint is why ``apis/app_api/kb_sync/records.py`` is written this way, and +this module follows it: **module-level imports are stdlib only**, and ``boto3`` +is imported inside the functions that need it. ``kb_backend/__init__.py`` is +intentionally empty so importing a submodule pulls in nothing else. +""" + +from __future__ import annotations + +import logging +import os +from dataclasses import dataclass, field +from decimal import Decimal +from typing import Any, Dict, Iterable, Mapping, Optional + +logger = logging.getLogger(__name__) + +# ── Engines ────────────────────────────────────────────────────────────────── +# +# LEGACY is never persisted. It is the value `resolve_engine` returns for a +# record that carries no `retrievalEngine` attribute, which is every record that +# predates this feature. +ENGINE_LEGACY = "s3vectors" +ENGINE_MANAGED = "managed" + +# ── Provisioning ───────────────────────────────────────────────────────────── +PROVISIONING = "provisioning" +ACTIVE = "active" +FAILED = "failed" +DELETING = "deleting" + +# ── Migration states ───────────────────────────────────────────────────────── +SHADOW = "shadow" +VERIFY = "verify" +PROMOTE = "promote" +RETAIN = "retain" +MIGRATION_FAILED = "failed" + +#: Reserved in the enum so a stored value round-trips, but never entered in this +#: phase. Reclaiming legacy vectors is explicitly a follow-up spec; a worker that +#: found itself here would delete data this phase has promised to retain. +RECLAIM = "reclaim" + +#: States that keep a record in the dispatcher's queue. Work keys are written on +#: entering one of these. +WORK_ELIGIBLE_STATES = frozenset({SHADOW, VERIFY, PROMOTE}) + +#: States that take a record out of the queue for good. Work keys are removed on +#: entering one of these. ``RETAIN`` is the terminal state this phase reaches; +#: ``MIGRATION_FAILED`` is terminal too and leaves the record on legacy, which +#: keeps working. +TERMINAL_STATES = frozenset({RETAIN, MIGRATION_FAILED}) + +ALL_MIGRATION_STATES = frozenset( + {SHADOW, VERIFY, PROMOTE, RETAIN, MIGRATION_FAILED, RECLAIM} +) + + +class TransitionLost(Exception): + """A conditional write was rejected because the guard did not hold. + + Raised instead of leaking ``ConditionalCheckFailedException`` so callers can + tell "another worker got there first, do nothing" apart from a real error. + Losing a race is normal and must not be logged as a failure. + """ + + +class ReclaimNotSupported(Exception): + """Refuses an attempt to enter ``reclaim``, which this phase never does.""" + + +# ── Keys ───────────────────────────────────────────────────────────────────── +def kb_pk(assistant_id: str) -> str: + return f"AST#{assistant_id}" + + +def kb_sk(app_kb_id: str) -> str: + return f"KB#{app_kb_id}" + + +def kb_tombstone_sk(app_kb_id: str) -> str: + return f"KBTOMB#{app_kb_id}" + + +def document_tombstone_sk(app_kb_id: str, document_id: str) -> str: + return f"KBTOMB#{app_kb_id}#DOC#{document_id}" + + +def work_pk(state: str) -> str: + return f"KBWORK#{state}" + + +def _table(): + import boto3 + + return boto3.resource("dynamodb").Table(os.environ["DYNAMODB_ASSISTANTS_TABLE_NAME"]) + + +# ── Model ──────────────────────────────────────────────────────────────────── +@dataclass +class KbRecord: + """A knowledge base's control-plane state. + + A dataclass rather than a Pydantic model on purpose: this module is imported + by a size-constrained Lambda image and has no need for validation machinery + it would then have to carry. + """ + + app_kb_id: str + owner_user_id: str + visibility: str = "PRIVATE" + + # Absent means legacy. Only ever ENGINE_MANAGED when present. + retrieval_engine: Optional[str] = None + + provisioning_state: str = PROVISIONING + aws_kb_id: Optional[str] = None + aws_data_source_id: Optional[str] = None + + # Immutable after creation: Bedrock rejects changing either, so they are + # recorded to make a mismatch detectable rather than mysterious. + embedding_model_id: str = "amazon.titan-embed-text-v2:0" + embedding_dimensions: int = 1024 + + # Captured at creation because a corpus indexed without image extraction is + # not comparable to one indexed with it. + parser_config: Dict[str, Any] = field(default_factory=dict) + image_extraction: bool = False + + stored_bytes: int = 0 + reserved_bytes: int = 0 + last_retrieved_at: Optional[str] = None + + migration_state: Optional[str] = None + migration_generation: int = 0 + migration_lease_until: Optional[str] = None + migration_progress: Dict[str, Any] = field(default_factory=dict) + migration_error: Optional[str] = None + + promoted_at: Optional[str] = None + rolled_back_at: Optional[str] = None + retain_until: Optional[str] = None + + pinned: bool = False + exempt_from_reclaim: bool = False + + client_token: Optional[str] = None + + def to_item(self, assistant_id: str) -> Dict[str, Any]: + """Serialize for DynamoDB, omitting absent optionals. + + Optionals are omitted rather than written as ``None`` so that "has no + opinion" stays distinguishable from "explicitly null". ``retrievalEngine`` + depends on that distinction. + """ + item: Dict[str, Any] = { + "PK": kb_pk(assistant_id), + "SK": kb_sk(self.app_kb_id), + "appKbId": self.app_kb_id, + "ownerUserId": self.owner_user_id, + "visibility": self.visibility, + "provisioningState": self.provisioning_state, + "embeddingModelId": self.embedding_model_id, + "embeddingDimensions": Decimal(self.embedding_dimensions), + "parserConfig": self.parser_config, + "imageExtraction": self.image_extraction, + "storedBytes": Decimal(self.stored_bytes), + "reservedBytes": Decimal(self.reserved_bytes), + "migrationGeneration": Decimal(self.migration_generation), + "pinned": self.pinned, + "exemptFromReclaim": self.exempt_from_reclaim, + } + + optional = { + "retrievalEngine": self.retrieval_engine, + "awsKbId": self.aws_kb_id, + "awsDataSourceId": self.aws_data_source_id, + "lastRetrievedAt": self.last_retrieved_at, + "migrationState": self.migration_state, + "migrationLeaseUntil": self.migration_lease_until, + "migrationError": self.migration_error, + "promotedAt": self.promoted_at, + "rolledBackAt": self.rolled_back_at, + "retainUntil": self.retain_until, + "clientToken": self.client_token, + } + item.update({k: v for k, v in optional.items() if v is not None}) + + if self.migration_progress: + item["migrationProgress"] = self.migration_progress + + return item + + +def resolve_engine(item: Optional[Mapping[str, Any]]) -> str: + """Return the backend that should serve this record. + + The whole migration rests on this function's default. A record with no + ``retrievalEngine`` attribute — which is every knowledge base that existed + before this feature — resolves to the legacy backend. Nothing had to be + written to make that true, and nothing has to be unwritten to roll back. + + A missing record resolves to legacy for the same reason: the absence of an + opinion is an answer, not an error. + """ + if not item: + return ENGINE_LEGACY + return ENGINE_MANAGED if item.get("retrievalEngine") == ENGINE_MANAGED else ENGINE_LEGACY + + +# ── Reads ──────────────────────────────────────────────────────────────────── +def get_kb_record(assistant_id: str, app_kb_id: str) -> Optional[Dict[str, Any]]: + response = _table().get_item(Key={"PK": kb_pk(assistant_id), "SK": kb_sk(app_kb_id)}) + return response.get("Item") + + +def query_due_work(state: str, now_iso: str, limit: int = 20) -> list: + """Records in ``state`` whose ``dueAt`` has passed, oldest first. + + Reads the sparse index, so records that have left the queue are not returned + because they have no key — not because they were filtered out. + """ + from boto3.dynamodb.conditions import Key + + response = _table().query( + IndexName="KbWorkIndex", + KeyConditionExpression=Key("GSI7_PK").eq(work_pk(state)) & Key("GSI7_SK").lte(now_iso), + Limit=limit, + ) + return response.get("Items", []) + + +# ── Transitions ────────────────────────────────────────────────────────────── +def _conditional(operation, **kwargs): + """Run a conditional write, translating a failed guard into TransitionLost.""" + from botocore.exceptions import ClientError + + try: + return operation(**kwargs) + except ClientError as exc: + if exc.response.get("Error", {}).get("Code") == "ConditionalCheckFailedException": + raise TransitionLost( + "conditional write rejected; another writer won or the " + "precondition no longer holds" + ) from exc + raise + + +def create_provisioning( + assistant_id: str, + record: KbRecord, +) -> Dict[str, Any]: + """Create the record, exactly once. + + ``attribute_not_exists(PK)`` makes this idempotent under concurrency: two + callers racing to enrol the same knowledge base produce one record and one + :class:`TransitionLost`, rather than one silently overwriting the other's + ``clientToken`` and orphaning a half-created AWS knowledge base. + """ + item = record.to_item(assistant_id) + _conditional( + _table().put_item, + Item=item, + ConditionExpression="attribute_not_exists(PK) AND attribute_not_exists(SK)", + ) + return item + + +def attach_aws_ids( + assistant_id: str, + app_kb_id: str, + aws_kb_id: str, + aws_data_source_id: str, + now_iso: str, +) -> None: + """Record the AWS identifiers and mark the record active. + + Guarded on still being ``provisioning`` so a late-returning create cannot + overwrite identifiers belonging to a newer generation. + """ + _conditional( + _table().update_item, + Key={"PK": kb_pk(assistant_id), "SK": kb_sk(app_kb_id)}, + UpdateExpression=( + "SET awsKbId = :kb, awsDataSourceId = :ds, " + "provisioningState = :active, updatedAt = :now" + ), + ConditionExpression="provisioningState = :provisioning", + ExpressionAttributeValues={ + ":kb": aws_kb_id, + ":ds": aws_data_source_id, + ":active": ACTIVE, + ":provisioning": PROVISIONING, + ":now": now_iso, + }, + ) + + +def set_resource_policy_state( + assistant_id: str, + app_kb_id: str, + aws_kb_id: Optional[str], + revision_id: Optional[str], +) -> None: + """Record which ``awsKbId`` the resource policy is currently attached to. + + Requirement 25.7. This attribute is the whole staleness check: a policy + attaches to an ARN, so once the record's ``awsKbId`` and this value disagree, + the policy is on a resource nobody reads and sharing has silently stopped. + Storing the target rather than a boolean is what turns that from an event + somebody has to remember to fire into a comparison + (``resource_policy.policy_is_stale``). + + Unconditional, deliberately. Every other writer here guards on the state it + expects, because those transitions must not race. This one records what AWS has + just confirmed, and a stale overwrite of the *same* fact is harmless while a + refused write would leave the record claiming a policy target that is no longer + true — the failure mode the attribute exists to prevent. + + Passing ``None`` clears both attributes, for a knowledge base that stopped + being shared. + """ + key = {"PK": kb_pk(assistant_id), "SK": kb_sk(app_kb_id)} + if aws_kb_id is None: + _table().update_item( + Key=key, + UpdateExpression="REMOVE policyAwsKbId, policyRevisionId", + ) + return + + values: Dict[str, Any] = {":kb": aws_kb_id} + expression = "SET policyAwsKbId = :kb" + if revision_id: + expression += ", policyRevisionId = :rev" + values[":rev"] = revision_id + else: + expression += " REMOVE policyRevisionId" + + _table().update_item( + Key=key, + UpdateExpression=expression, + ExpressionAttributeValues=values, + ) + + +def promote_engine( + assistant_id: str, + app_kb_id: str, + generation: int, + now_iso: str, +) -> None: + """Flip the record to the managed backend. The single cutover write. + + Four guards, all necessary: + + * ``attribute_not_exists(retrievalEngine)`` — this knowledge base has not + already been promoted. Without it, the other three guards all remain true + *after* a successful promotion, so a worker that crashed between the + promotion and the state transition promotes a second time on resume: same + value, but a fresh ``promotedAt`` that overwrites the real cutover moment and + a second ``KbMigrationPromoted``. Worse, two concurrent workers would both + succeed, which is precisely what Requirement 15.10 forbids. Found by the + convergence property test, which counted two promotions across a crash at + the state transition. Rollback ``REMOVE``s the attribute, so this does not + block a deliberate re-promotion. + * ``migrationState = promote`` — only a record that reached the cutover step + may cut over. + * ``migrationGeneration = :gen`` — a worker whose lease expired and whose + generation has been superseded cannot promote on stale information. + * ``migrationProgress.migrated = migrationProgress.total`` — the catch-up + pass has converged. Without this, promotion could strand documents written + during migration on a backend nobody reads any more. Comparing two + document paths keeps the check atomic with the write; passing the total in + as a value would let it go stale between read and write. + + ``total`` is a DynamoDB reserved keyword, so the progress paths are aliased + through ``ExpressionAttributeNames``. Without the aliases the whole condition + is rejected as a ``ValidationException`` — loudly, which is the good case, but + only because it never validates at all. + + Because this is one conditional write, rollback is symmetric: see + :func:`rollback_engine`. + """ + _conditional( + _table().update_item, + Key={"PK": kb_pk(assistant_id), "SK": kb_sk(app_kb_id)}, + UpdateExpression="SET retrievalEngine = :managed, promotedAt = :now", + ConditionExpression=( + "attribute_not_exists(retrievalEngine) " + "AND migrationState = :promote " + "AND migrationGeneration = :gen " + "AND #progress.#migrated = #progress.#total" + ), + ExpressionAttributeNames={ + "#progress": "migrationProgress", + "#migrated": "migrated", + "#total": "total", + }, + ExpressionAttributeValues={ + ":managed": ENGINE_MANAGED, + ":promote": PROMOTE, + ":gen": Decimal(generation), + ":now": now_iso, + }, + ) + + +def rollback_engine(assistant_id: str, app_kb_id: str, now_iso: str) -> None: + """Return the record to the legacy backend by REMOVING the engine attribute. + + Note the ``REMOVE``. Rollback restores the original *shape*, not a written + legacy value, so a rolled-back record is byte-indistinguishable from one that + never migrated. Writing ``"s3vectors"`` here would work today and quietly + break the "absence means legacy" invariant that lets this feature ship + without touching 1,692 existing records. + + Guarded on currently being managed so a double rollback is a no-op loss + rather than a spurious ``rolledBackAt`` bump. + """ + _conditional( + _table().update_item, + Key={"PK": kb_pk(assistant_id), "SK": kb_sk(app_kb_id)}, + UpdateExpression="REMOVE retrievalEngine SET rolledBackAt = :now", + ConditionExpression="retrievalEngine = :managed", + ExpressionAttributeValues={":managed": ENGINE_MANAGED, ":now": now_iso}, + ) + + +def set_migration_state( + assistant_id: str, + app_kb_id: str, + new_state: str, + generation: int, + due_at: Optional[str] = None, + expected_states: Optional[Iterable[str]] = None, + error: Optional[str] = None, +) -> None: + """Move to ``new_state``, maintaining the sparse work keys. + + Entering a work-eligible state writes ``GSI7_PK``/``GSI7_SK``; entering a + terminal state ``REMOVE``s them. The removal is the point: it is what takes + the record out of the dispatcher's queue, and skipping it would leave a + finished knowledge base being handed to workers forever. + + ``expected_states`` guards the transition against a concurrent writer that + has already moved the record on. The generation is always guarded. + """ + if new_state == RECLAIM: + raise ReclaimNotSupported( + "reclaim is reserved but never entered in this phase; reclaiming " + "legacy vectors is a follow-up spec" + ) + if new_state not in ALL_MIGRATION_STATES: + raise ValueError(f"unknown migration state: {new_state!r}") + if new_state in WORK_ELIGIBLE_STATES and not due_at: + raise ValueError(f"{new_state} is work-eligible and requires due_at") + + values: Dict[str, Any] = { + ":state": new_state, + ":gen": Decimal(generation), + } + sets = ["migrationState = :state"] + removes = [] + + if new_state in TERMINAL_STATES: + # Leaving the queue: the keys must go, not merely be ignored. + removes.extend(["GSI7_PK", "GSI7_SK"]) + else: + sets.extend(["GSI7_PK = :wpk", "GSI7_SK = :wsk"]) + values[":wpk"] = work_pk(new_state) + values[":wsk"] = due_at + + if error is not None: + sets.append("migrationError = :err") + values[":err"] = error + + expression = f"SET {', '.join(sets)}" + if removes: + expression += f" REMOVE {', '.join(removes)}" + + condition = "migrationGeneration = :gen" + if expected_states is not None: + expected = list(expected_states) + if not expected: + raise ValueError("expected_states must be non-empty when provided") + placeholders = [] + for index, state in enumerate(expected): + placeholder = f":exp{index}" + placeholders.append(placeholder) + values[placeholder] = state + condition += f" AND migrationState IN ({', '.join(placeholders)})" + + _conditional( + _table().update_item, + Key={"PK": kb_pk(assistant_id), "SK": kb_sk(app_kb_id)}, + UpdateExpression=expression, + ConditionExpression=condition, + ExpressionAttributeValues=values, + ) + + +def acquire_lease( + assistant_id: str, + app_kb_id: str, + lease_until: str, + now_iso: str, +) -> None: + """Take the worker lease, or lose the race. + + The guard admits exactly two situations: no lease has ever been taken, or the + existing lease has expired. A live lease held by another worker rejects, + which is what stops two workers migrating the same knowledge base and + double-ingesting its corpus. + + ISO-8601 UTC strings compare correctly lexicographically, so the expiry test + is a plain string comparison and stays atomic with the write. + """ + _conditional( + _table().update_item, + Key={"PK": kb_pk(assistant_id), "SK": kb_sk(app_kb_id)}, + UpdateExpression="SET migrationLeaseUntil = :until", + ConditionExpression=( + "attribute_not_exists(migrationLeaseUntil) OR migrationLeaseUntil < :now" + ), + ExpressionAttributeValues={":until": lease_until, ":now": now_iso}, + ) + + +def retry_from_failed( + assistant_id: str, + app_kb_id: str, + generation: int, + due_at: str, +) -> None: + """Re-enter ``shadow`` from ``failed`` on the next generation, atomically. + + One write, not two. Split into "bump the generation" then + "set_migration_state", a crash between them leaves a record carrying a new + generation while still ``failed`` — and with no work keys, so it is invisible + to the dispatcher while the retry control has already reported success. The + user would wait forever on an upgrade nothing owns. + + Guarded on **both** the old generation and still being ``failed``, so two + concurrent retries yield one new attempt and one :class:`TransitionLost`. The + generation bump is also what fences the abandoned attempt: every conditional + write belonging to it is guarded on the old value, so a straggler worker + cannot land on the new generation. + + ``migrationError`` is removed rather than left behind, so a subsequent + failure's reason cannot be mistaken for this one's. + """ + _conditional( + _table().update_item, + Key={"PK": kb_pk(assistant_id), "SK": kb_sk(app_kb_id)}, + UpdateExpression=( + "SET migrationState = :shadow, migrationGeneration = :next, " + "GSI7_PK = :wpk, GSI7_SK = :wsk REMOVE migrationError" + ), + ConditionExpression="migrationGeneration = :gen AND migrationState = :failed", + ExpressionAttributeValues={ + ":shadow": SHADOW, + ":failed": MIGRATION_FAILED, + ":gen": Decimal(generation), + ":next": Decimal(generation + 1), + ":wpk": work_pk(SHADOW), + ":wsk": due_at, + }, + ) + + +def dismiss_upgrade_notice(assistant_id: str, app_kb_id: str, now_iso: str) -> None: + """Retire the one-time post-upgrade notice. + + Unconditional on purpose: dismissing an already-dismissed notice is not a + race worth losing, and the attribute's only reader treats any value as + "dismissed". Guarded only on the record existing, so a dismissal for a + knowledge base that never had a record cannot conjure one. + """ + _conditional( + _table().update_item, + Key={"PK": kb_pk(assistant_id), "SK": kb_sk(app_kb_id)}, + UpdateExpression="SET upgradeNoticeDismissedAt = :now", + ConditionExpression="attribute_exists(PK) AND attribute_exists(SK)", + ExpressionAttributeValues={":now": now_iso}, + ) diff --git a/backend/src/apis/shared/kb_backend/resolver.py b/backend/src/apis/shared/kb_backend/resolver.py new file mode 100644 index 000000000..05585231c --- /dev/null +++ b/backend/src/apis/shared/kb_backend/resolver.py @@ -0,0 +1,198 @@ +"""Which backend serves this knowledge base. + +One question, answered in one place: read the KB_Record's ``retrievalEngine`` and +hand back the matching implementation. Callers get an object satisfying +:class:`~apis.shared.kb_backend.protocol.KnowledgeBaseBackend` and are given no +way to ask which one it is. + +Absence means legacy +-------------------- +The decision itself is delegated to +:func:`apis.shared.kb_backend.records.resolve_engine` rather than re-derived +here. That function is the one place that knows a missing ``retrievalEngine`` +attribute means the legacy backend, and it is covered by its own property test +(task 3.3). Two implementations of the same default would be two chances to +disagree about the invariant that lets 1,692 existing knowledge bases keep +working with zero backfill writes. + +Resolution never fails a turn +----------------------------- +The KB_Record lookup is a DynamoDB read that today's retrieval path does not +perform, so it is a new way for retrieval to break. It is therefore wrapped: any +failure — unreachable table, unset ``DYNAMODB_ASSISTANTS_TABLE_NAME``, malformed +item — resolves to legacy, which is what every knowledge base in existence +already uses. The failure is logged at warning level. Choosing legacy on an +unreadable record is not a guess; it is the same answer the absent attribute +gives, and the whole migration is built so that answer is always safe. + +Why both backends are registered at import +------------------------------------------ +Registration is not a startup step. Both adapters are installed in +:data:`_BACKENDS` when this module is imported, so there is no sequence anybody +has to remember and no service that can come up half-configured. That matters +because forgetting would not be loud in a useful way: a promoted knowledge base +would raise :class:`BackendUnavailable` on every turn — correct as a fail-safe, +useless as a signal, and only ever seen by the one user whose knowledge base was +migrated. + +It costs nothing. Both adapter modules import stdlib and the protocol only, with +``boto3`` and their clients created lazily inside methods, which +``tests/architecture/test_kb_backend_boundary.py`` asserts in a fresh +interpreter. Registering an object whose constructor does no work is not the same +as connecting to anything. + +Registering the managed backend does **not** make the feature live. Nothing can +resolve to it until a record says ``retrievalEngine == "managed"``, and nothing +writes that value except a promotion, which needs the migration flag on and an +explicit opt-in. Registration only settles what happens *once* a record says so. +""" + +from __future__ import annotations + +import logging +from typing import Any, Dict, Mapping, Optional + +from apis.shared.kb_backend.managed_backend import ManagedKbBackend +from apis.shared.kb_backend.protocol import KnowledgeBaseBackend +from apis.shared.kb_backend.records import ENGINE_LEGACY, ENGINE_MANAGED, resolve_engine +from apis.shared.kb_backend.s3vectors_backend import S3VectorsBackend + +logger = logging.getLogger(__name__) + + +class BackendUnavailable(RuntimeError): + """A record names an engine this build has no implementation for. + + Raised rather than quietly falling back to legacy. A record only ever names + ``managed`` after a successful promotion, and serving legacy for a promoted + knowledge base would read an index that migration has stopped maintaining — + fewer results, silently, with no error to notice. + + Reachable only if a backend is explicitly unregistered (which tests do) or if + a future engine name is written by a newer deployment than the one reading it. + """ + + +# Engine → backend, populated at import. See the module docstring for why this is +# not a startup step. Both constructors are inert: clients are created lazily. +_BACKENDS: Dict[str, KnowledgeBaseBackend] = { + ENGINE_LEGACY: S3VectorsBackend(), + ENGINE_MANAGED: ManagedKbBackend(), +} + + +def register_backend(engine: str, backend: KnowledgeBaseBackend) -> None: + """Install the implementation for ``engine``, replacing any previous one.""" + _BACKENDS[engine] = backend + + +def unregister_backend(engine: str) -> None: + """Remove ``engine``'s implementation. Absent engines are ignored.""" + _BACKENDS.pop(engine, None) + + +def registered_engines() -> frozenset: + """Engines this build can serve. Introspection for tests and diagnostics.""" + return frozenset(_BACKENDS) + + +def load_record( + assistant_id: str, + app_kb_id: Optional[str] = None, +) -> Dict[str, Any]: + """The KB_Record, or an empty mapping if there is none to be had. + + For callers that need more from the record than which backend serves it — the + dual-read pilot flag, the byte cap, the migration state — and would otherwise + read it a second time. + + Returns ``{}`` rather than ``None`` for both "no such record" and "the read + failed", because those two cases have the same answer everywhere in this + feature: an absent opinion is the legacy opinion. Collapsing them here means + no caller has to remember to handle ``None`` and every caller can pass the + result straight to :func:`resolve_backend` as ``record=``, which is what makes + one read enough. + """ + from apis.shared.kb_backend.records import get_kb_record + + try: + return dict(get_kb_record(assistant_id, app_kb_id or assistant_id) or {}) + except Exception as exc: + logger.warning( + f"KB_Record lookup failed for assistant {assistant_id}; treating it as " + f"absent, which resolves to {ENGINE_LEGACY}: {exc}" + ) + return {} + + +def backend_for_engine(engine: str) -> Optional[KnowledgeBaseBackend]: + """The implementation for ``engine``, or ``None`` if this build has none. + + Unlike :func:`resolve_backend` this does not raise, because its callers are + asking a different question. The dual-read pilot wants "is there a managed + backend I could compare against?", and ``None`` is an ordinary answer for it — + a test that unregistered one, or a deployment older than the engine name it was + handed — not the fail-safe emergency that an unservable *promoted* record is. + """ + return _BACKENDS.get(engine) + + +def resolve_engine_for( + assistant_id: str, + app_kb_id: Optional[str] = None, + record: Optional[Mapping[str, Any]] = None, +) -> str: + """Return the engine name for a knowledge base. + + Pass ``record`` when the caller already holds the KB_Record to skip the + read. ``app_kb_id`` defaults to ``assistant_id``, which is the 1:1 binding + this phase deliberately preserves. + """ + if record is not None: + return resolve_engine(record) + + from apis.shared.kb_backend.records import get_kb_record + + try: + item = get_kb_record(assistant_id, app_kb_id or assistant_id) + except Exception as exc: + # Unreadable record ⇒ legacy, the same answer absence gives. + logger.warning( + f"KB_Record lookup failed for assistant {assistant_id}, " + f"resolving to {ENGINE_LEGACY}: {exc}" + ) + return ENGINE_LEGACY + + return resolve_engine(item) + + +def resolve_backend( + assistant_id: str, + app_kb_id: Optional[str] = None, + record: Optional[Mapping[str, Any]] = None, +) -> KnowledgeBaseBackend: + """Return the backend instance that should serve this knowledge base.""" + engine = resolve_engine_for(assistant_id, app_kb_id, record) + try: + return _BACKENDS[engine] + except KeyError: + raise BackendUnavailable( + f"knowledge base {app_kb_id or assistant_id} names engine {engine!r}, " + f"which this build cannot serve (have: {sorted(_BACKENDS)}). " + f"Refusing to substitute {ENGINE_LEGACY}: a promoted knowledge base's " + f"legacy index is no longer maintained." + ) from None + + +__all__ = [ + "BackendUnavailable", + "ENGINE_LEGACY", + "ENGINE_MANAGED", + "backend_for_engine", + "load_record", + "register_backend", + "registered_engines", + "resolve_backend", + "resolve_engine_for", + "unregister_backend", +] diff --git a/backend/src/apis/shared/kb_backend/resource_policy.py b/backend/src/apis/shared/kb_backend/resource_policy.py new file mode 100644 index 000000000..0d5211d4b --- /dev/null +++ b/backend/src/apis/shared/kb_backend/resource_policy.py @@ -0,0 +1,329 @@ +"""IAM-enforced retrieval on a shared managed knowledge base. + +Requirements 25.6, 25.7. Resource policies are MANAGED-only and are the only +mechanism in this design that offers *infrastructure* isolation rather than +filter-level isolation. A policy attached to a knowledge base ARN restricts +``bedrock:Retrieve`` and ``bedrock:GetDocumentContent`` to the principals it +names, which matters because the platform's own identity grant +(``grantManagedKbRetrieval``) is written against ``knowledge-base/*`` — every +knowledge base in the account, present and future. Without a policy, any +principal in the account holding a similar grant can read a shared corpus. + +What this is NOT +---------------- +It is **not** per-user authorization, and the temptation to describe it that way +is the reason this paragraph exists. Every user of this platform retrieves through +the same infrastructure identity — the AgentCore runtime role — so no resource +policy can distinguish user A from user B. Per-user authorization is, and remains, +the application's job (Requirement 25.3, ``apis.shared.assistants.kb_access``). +What a policy buys is a narrower blast radius for a corpus that belongs to more +than one person: the set of *infrastructure* identities able to reach it shrinks +from "anything in the account with a wildcard grant" to an explicit list. + +Applied only where a knowledge base is shared beyond its owner, because a policy +on a single-owner knowledge base would restrict nothing that the assistant's own +access check does not already restrict, while adding a control-plane call and a +piece of state to keep in step. + +Why staleness is state rather than an event +------------------------------------------- +A policy attaches to the AWS knowledge base ARN, so any cycle producing a new +``awsKbId`` silently drops sharing — the call succeeds, the policy is simply on a +resource nobody reads any more. The obvious fix is to re-apply from wherever a new +identifier is created. That fix is only as good as the completeness of the list of +such places, and this phase already has two (fresh provisioning, resumed +provisioning) with dormancy/rehydration a known future third. + +So the record stores the identifier the policy was last applied *to*, and +:func:`policy_is_stale` compares it against the current one. A path that produces +a new ``awsKbId`` and forgets to re-apply is then not a silent regression: the +next :func:`ensure_retrieve_policy` sees a mismatch and repairs it. The invariant +is checked by comparing two values, which no new code path can bypass by omission +(Requirement 24.12). + +Import weight +------------- +``boto3`` and ``json`` usage stays inside functions where practical; the module +imports stdlib only, per this package's Lambda-image constraint. + +Feature: managed-kb-migration +Requirements: 25.6, 25.7 +""" + +from __future__ import annotations + +import json +import logging +import os +from typing import Any, Dict, Iterable, Mapping, Optional, Sequence, Tuple + +logger = logging.getLogger(__name__) + +#: The actions a shared knowledge base's readers need. ``GetDocumentContent`` is +#: included because a retrieval that returns a citation the caller cannot then +#: fetch is a half-share — the evaluation names both as what resource policies +#: cover. +RETRIEVE_ACTIONS: Tuple[str, ...] = ("bedrock:Retrieve", "bedrock:GetDocumentContent") + +#: Statement id. Fixed so a re-application replaces the platform's own statement +#: rather than accumulating near-duplicates. +POLICY_SID = "PlatformSharedRetrieve" + +POLICY_VERSION = "2012-10-17" + +#: Comma-separated ARNs of the infrastructure identities that retrieve on users' +#: behalf — the AgentCore runtime role, and the App API task role for test-chat. +#: Read at call time, never captured in a default argument. +PRINCIPALS_ENV = "MANAGED_KB_RETRIEVAL_PRINCIPAL_ARNS" + +#: Record attributes tracking what was applied where. +POLICY_KB_ID_ATTR = "policyAwsKbId" +POLICY_REVISION_ATTR = "policyRevisionId" + + +class ResourcePolicyError(RuntimeError): + """A resource policy could not be applied or removed.""" + + +def _region() -> str: + return os.environ.get("AWS_REGION", "us-west-2") + + +def bedrock_agent_client(): + """Control-plane client. ``PutResourcePolicy`` lives on ``bedrock-agent``. + + Verified against the pinned botocore service model: ``PutResourcePolicy`` + takes ``resourceArn`` and ``policy`` (both required) plus an optional + ``expectedRevisionId``, and returns ``resourceArn`` and ``revisionId``. + """ + import boto3 + + return boto3.client("bedrock-agent", region_name=_region()) + + +def knowledge_base_arn( + aws_kb_id: str, + region: Optional[str] = None, + account_id: Optional[str] = None, +) -> str: + """The ARN a policy attaches to. + + Refuses to guess the account. A wrong account in an ARN does not fail loudly — + ``PutResourcePolicy`` would target a resource this caller cannot see, and the + error it raises names a resource the operator did not know existed. Better to + say what is missing. + """ + resolved_account = account_id or os.environ.get("AWS_ACCOUNT_ID") + if not resolved_account: + raise ResourcePolicyError( + f"cannot build a knowledge base ARN for {aws_kb_id} without an account " + f"id: pass account_id or set AWS_ACCOUNT_ID" + ) + return f"arn:aws:bedrock:{region or _region()}:{resolved_account}:knowledge-base/{aws_kb_id}" + + +def retrieval_principals(explicit: Optional[Iterable[str]] = None) -> Tuple[str, ...]: + """The infrastructure identities allowed to retrieve, in a stable order. + + Sorted and de-duplicated so the same configuration always produces the same + policy document — otherwise every call looks like a change and nothing can be + compared. + """ + if explicit is not None: + candidates: Sequence[str] = list(explicit) + else: + candidates = (os.environ.get(PRINCIPALS_ENV) or "").split(",") + return tuple(sorted({arn.strip() for arn in candidates if arn and arn.strip()})) + + +def retrieve_policy_document(kb_arn: str, principals: Sequence[str]) -> Dict[str, Any]: + """The policy granting exactly the shared-read actions to exactly ``principals``. + + No wildcard principal and no wildcard resource: a resource policy whose point + is to narrow access is worse than no policy at all if it widens it instead. + """ + if not principals: + raise ResourcePolicyError( + "refusing to write a resource policy with no principals: an empty " + "principal list is not a narrower grant, it is an unparseable one" + ) + return { + "Version": POLICY_VERSION, + "Statement": [ + { + "Sid": POLICY_SID, + "Effect": "Allow", + "Principal": {"AWS": list(principals)}, + "Action": list(RETRIEVE_ACTIONS), + "Resource": kb_arn, + } + ], + } + + +def policy_is_stale(record: Optional[Mapping[str, Any]]) -> bool: + """Whether the recorded policy target no longer matches the live ``awsKbId``. + + ``True`` when a knowledge base exists in AWS and either no policy target was + ever recorded or the recorded one differs. ``False`` for a record with no + ``awsKbId`` at all: nothing has been provisioned, so there is nothing to be + stale against. + """ + if not record: + return False + aws_kb_id = record.get("awsKbId") + if not aws_kb_id: + return False + return record.get(POLICY_KB_ID_ATTR) != aws_kb_id + + +async def ensure_retrieve_policy( + assistant_id: str, + app_kb_id: str, + *, + shared: bool, + record: Optional[Mapping[str, Any]] = None, + principals: Optional[Iterable[str]] = None, + client=None, + region: Optional[str] = None, + account_id: Optional[str] = None, +) -> Optional[str]: + """Bring the knowledge base's resource policy in line with its sharing state. + + Returns the revision id of a policy that is now in place, or ``None`` when no + policy is wanted or none could be applied. + + Four cases: + + * **Not shared, no policy recorded** — nothing to do. + * **Not shared, policy recorded** — remove it, and forget the target. A + knowledge base that stops being shared should stop carrying the statement + that says it is. + * **Shared, policy current** — nothing to do. This is the common path and it + makes no AWS call, which is what allows callers to invoke this freely. + * **Shared, policy missing or stale** — apply, then record the ``awsKbId`` it + was applied to. + + ``shared`` is supplied by the caller rather than derived here: sharing is an + application fact (visibility plus share records) that lives above this seam, + and this package may not import the assistants package. + """ + from apis.shared.kb_backend import records as r + + if record is None: + import asyncio + + record = await asyncio.to_thread(r.get_kb_record, assistant_id, app_kb_id) + + if not record: + return None + + aws_kb_id = record.get("awsKbId") + recorded_target = record.get(POLICY_KB_ID_ATTR) + + if not shared: + if recorded_target: + await _remove(assistant_id, app_kb_id, recorded_target, client, region, account_id) + return None + + if not aws_kb_id: + # Shared, but nothing provisioned yet. Provisioning is lazy by design, so + # this is ordinary, not an error: the next call after provisioning sees a + # stale (unset) target and applies. + return None + + if not policy_is_stale(record): + return record.get(POLICY_REVISION_ATTR) + + resolved = retrieval_principals(principals) + if not resolved: + # Loud, and not repaired by guessing. A policy with no principals cannot + # be written, and inventing one would either widen access or lock the + # platform out of its own corpus. + logger.error( + f"knowledge base {app_kb_id} is shared but {PRINCIPALS_ENV} names no " + f"principals; no resource policy applied (Requirement 25.6)" + ) + return None + + arn = knowledge_base_arn(aws_kb_id, region, account_id) + document = retrieve_policy_document(arn, resolved) + api = client or bedrock_agent_client() + + try: + response = api.put_resource_policy(resourceArn=arn, policy=json.dumps(document)) + except Exception as exc: + raise ResourcePolicyError( + f"failed to apply the retrieve policy for kb {app_kb_id} on {arn}: {exc}" + ) from exc + + revision_id = response.get("revisionId") + await _record(assistant_id, app_kb_id, aws_kb_id, revision_id) + + if recorded_target and recorded_target != aws_kb_id: + logger.info( + f"re-applied the retrieve policy for kb {app_kb_id}: it was attached " + f"to {recorded_target}, which is no longer this knowledge base's id " + f"(Requirement 25.7)" + ) + return revision_id + + +async def _remove( + assistant_id: str, + app_kb_id: str, + recorded_target: str, + client, + region: Optional[str], + account_id: Optional[str], +) -> None: + """Delete the policy and forget the target, tolerating an absent policy. + + A ``ResourceNotFoundException`` here means the policy or its knowledge base is + already gone, which is the state being asked for. The record is cleared either + way, so a knowledge base cannot be left claiming a policy that does not exist. + """ + api = client or bedrock_agent_client() + arn = knowledge_base_arn(recorded_target, region, account_id) + try: + api.delete_resource_policy(resourceArn=arn) + except Exception as exc: + if type(exc).__name__ not in ("ResourceNotFoundException", "ValidationException"): + raise ResourcePolicyError( + f"failed to remove the retrieve policy for kb {app_kb_id} on {arn}: {exc}" + ) from exc + logger.info( + f"retrieve policy for kb {app_kb_id} was already absent on {arn}; " + f"clearing the record anyway" + ) + await _record(assistant_id, app_kb_id, None, None) + + +async def _record( + assistant_id: str, + app_kb_id: str, + aws_kb_id: Optional[str], + revision_id: Optional[str], +) -> None: + import asyncio + + from apis.shared.kb_backend import records as r + + await asyncio.to_thread( + r.set_resource_policy_state, assistant_id, app_kb_id, aws_kb_id, revision_id + ) + + +__all__ = [ + "POLICY_KB_ID_ATTR", + "POLICY_REVISION_ATTR", + "POLICY_SID", + "PRINCIPALS_ENV", + "RETRIEVE_ACTIONS", + "ResourcePolicyError", + "ensure_retrieve_policy", + "knowledge_base_arn", + "policy_is_stale", + "retrieval_principals", + "retrieve_policy_document", +] diff --git a/backend/src/apis/shared/kb_backend/s3vectors_backend.py b/backend/src/apis/shared/kb_backend/s3vectors_backend.py new file mode 100644 index 000000000..d34e5b73f --- /dev/null +++ b/backend/src/apis/shared/kb_backend/s3vectors_backend.py @@ -0,0 +1,142 @@ +"""The legacy Amazon S3 Vectors backend, behind the common protocol. + +This is the retrieval path every assistant chat has used to date, moved here +unchanged and wrapped in :class:`~apis.shared.kb_backend.protocol.Chunk`. The +only thing this adapter *adds* is the score-direction conversion, and the only +thing it takes away from ``rag_service`` is knowledge of what an S3 Vectors +response looks like. + +Why this delegates instead of copying the query +----------------------------------------------- +``apis.shared.embeddings.bedrock_embeddings.search_assistant_knowledgebase`` +stays where it is and this adapter calls it. It is a published export of two +packages (``apis.shared.embeddings`` and +``apis.app_api.documents.ingestion.embeddings``), and task 5.2 of this spec +still expects to edit it in place. Re-implementing its ``query_vectors`` call +here would mean two copies of the topK/filter/returnDistance construction, which +is precisely the divergence risk this seam exists to remove. What moves here is +everything ``rag_service`` used to know: the response shape, the score +direction, and the parity ``top_k``. + +Score direction — read this before touching :meth:`S3VectorsBackend.search` +--------------------------------------------------------------------------- +S3 Vectors returns cosine **distance**: ``0.0`` is a perfect match and larger is +worse. The protocol canonicalizes on **relevance**, where larger is better. The +conversion happens *here*, once, so that nothing above the seam ever has to know +which direction this particular backend counts in. + +Inverting it raises nothing and logs nothing. Retrieval keeps returning five +chunks, the request keeps succeeding, and the answers quietly get worse. The +guard is ``tests/property/test_pbt_kb_score_direction.py``. + +Ordering is *not* re-sorted here. S3 Vectors already returns results ranked +nearest-first, and today's code passes that order straight through; re-sorting +would be a behaviour change dressed up as a safety measure. The invariant this +adapter owns is that the ``relevance`` values it attaches agree with the order it +returns — descending relevance for ascending distance. + +Import boundary +--------------- +Module-level imports are **stdlib only**; ``boto3`` and the embeddings stack are +imported inside the methods that use them, so importing this module into a +size-constrained Lambda image costs nothing. See +``tests/architecture/test_kb_backend_boundary.py``. +""" + +from __future__ import annotations + +import logging +from typing import Any, Dict, List + +from apis.shared.kb_backend.protocol import ( + DEFAULT_TOP_K, + Chunk, + DocumentSource, + relevance_from_distance, +) + +logger = logging.getLogger(__name__) + + +class S3VectorsBackend: + """Retrieval and ingestion over the S3 Vectors index. + + ``kb_ref`` is the ``App_KB_Id``, which equals the ``assistant_id`` in this + phase; the S3 Vectors index is global and partitioned by an ``assistant_id`` + metadata filter, so the reference is used directly as that filter value. + + Stateless, so a shared instance is safe and no client is held across calls. + """ + + async def search(self, kb_ref: str, query: str, top_k: int = DEFAULT_TOP_K) -> List[Chunk]: + """Query the index and return chunks scored by relevance, best first. + + ``top_k`` is accepted to satisfy the protocol but the underlying query + has always requested a fixed five results (Requirement 3.1), and + narrowing happens above the seam *after* the document-status filter has + run — filtering first and slicing second is what stops a single + incomplete document from silently shrinking a five-chunk answer to four. + Slicing here instead would change that, so this returns what the index + returned. + """ + from apis.shared.embeddings.bedrock_embeddings import search_assistant_knowledgebase + + response = await search_assistant_knowledgebase(kb_ref, query) + vectors = response.get("vectors", []) + return [self._to_chunk(vector) for vector in vectors] + + @staticmethod + def _to_chunk(vector: Dict[str, Any]) -> Chunk: + """Adapt one S3 Vectors hit, converting distance into relevance. + + The ``.get`` defaults mirror the formatting this replaced exactly: a + missing ``text`` or ``key`` became ``""`` and a missing ``distance`` + became ``None``, so they still do. + """ + metadata = vector.get("metadata", {}) + return Chunk( + text=metadata.get("text", ""), + # The conversion. Lower distance ⇒ higher relevance. + relevance=relevance_from_distance(vector.get("distance")), + document_id=metadata.get("document_id", ""), + metadata=metadata, + key=vector.get("key", ""), + ) + + async def ingest(self, kb_ref: str, source: DocumentSource) -> None: + """Embed and store ``source``'s chunks, as the current pipeline does. + + Requires pre-chunked text: splitting is the ingestion pipeline's job + (it owns the tokenizer this package deliberately does not depend on), + so an unchunked source is a programming error rather than something to + paper over with a naive split. + """ + from apis.shared.embeddings.bedrock_embeddings import ( + generate_embeddings, + store_embeddings_in_s3, + ) + + if not source.chunks: + raise ValueError( + f"S3VectorsBackend.ingest requires pre-chunked text for document " + f"{source.document_id}; chunking belongs to the ingestion pipeline" + ) + + embeddings = await generate_embeddings(source.chunks) + await store_embeddings_in_s3( + assistant_id=kb_ref, + document_id=source.document_id, + chunks=source.chunks, + embeddings=embeddings, + metadata={"filename": source.filename, **source.metadata}, + ) + + async def delete_document(self, kb_ref: str, document_id: str) -> None: + """Delete every vector belonging to ``document_id``.""" + from apis.shared.embeddings.bedrock_embeddings import delete_vectors_for_document + + deleted = await delete_vectors_for_document(document_id) + logger.info( + f"S3VectorsBackend: deleted {deleted} vectors for document " + f"{document_id} (kb {kb_ref})" + ) diff --git a/backend/src/apis/shared/kb_backend/tags.py b/backend/src/apis/shared/kb_backend/tags.py new file mode 100644 index 000000000..0c52eab40 --- /dev/null +++ b/backend/src/apis/shared/kb_backend/tags.py @@ -0,0 +1,211 @@ +"""The managed knowledge base tag contract, in one place. + +Requirement 20.11. Tags are not housekeeping here: a tag-filtered +``ListKnowledgeBases`` is how the reconciler tells this platform's knowledge bases +from everything else in the account, and how teardown scopes itself. An untagged — +or mistagged — knowledge base is invisible to both, which means it is never +reclaimed and never deleted, and it keeps billing at $5.00/GB-month with no +CloudFormation console to notice it in. + +Why this module exists +---------------------- +It did not, and the tags drifted three ways: + +* ``provisioning.build_tags`` wrote keys ``prefix``/``env`` with values from + ``PROJECT_PREFIX``/``ENVIRONMENT`` — neither of which the provisioning Lambda is + given, so every knowledge base would have been tagged with the hardcoded + defaults regardless of project or environment. +* ``tombstones.project_tag_filter`` was a hand-written *mirror* of that function, + documented as such. A mirror is a second implementation, and the only thing + keeping two implementations equal is that nobody has edited one of them yet. +* ``kb-migration-construct.ts`` declared a different set of key names entirely + (``ManagedKbPrefix``, …) and exported them plus the correct values as env vars + that **nothing read**. +* ``scripts/teardown/managed-kb.sh`` read a third pair of variables + (``CDK_PROJECT_PREFIX``/``CDK_ENVIRONMENT``) and matched on ``prefix``/``env``. + +Writer and reconciler agreed by luck — both used the same wrong defaults — so the +symptom was not a crash but a teardown that found nothing and reported success. + +So: the keys live here as constants, the value resolution lives here as one +function, and every consumer in every language reads *these* names. +``tests/shared/test_kb_tag_contract.py`` parses the TypeScript and the shell script +and fails if they disagree, because agreement between three languages is not +something a type checker can hold. + +Why the keys are namespaced +--------------------------- +``ManagedKbPrefix`` rather than ``prefix``, and ``ManagedKbEnvironment`` rather +than ``env``. Generic keys collide: many accounts carry an organisation-wide +cost-allocation tag literally called ``env``, and if something else writes it our +filter compares against a value we did not set. The failure mode is a teardown +that skips a knowledge base it owns — the leak this whole contract exists to +prevent. + +Feature: managed-kb-migration +Requirements: 20.11, 20.12, 20.8, 14.1 +""" + +from __future__ import annotations + +import logging +import os +from typing import Any, Dict, Mapping, Optional + +logger = logging.getLogger(__name__) + +# ── Tag keys ───────────────────────────────────────────────────────────────── +# +# Mirrored by `MANAGED_KB_TAG_KEYS` in +# `infrastructure/lib/constructs/managed-kb/kb-migration-construct.ts`, and that +# mirroring is asserted by a test rather than trusted. +TAG_KEY_PREFIX = "ManagedKbPrefix" +TAG_KEY_ENVIRONMENT = "ManagedKbEnvironment" +TAG_KEY_APP_KB_ID = "ManagedKbAppKbId" +TAG_KEY_OWNER_USER_ID = "ManagedKbOwnerUserId" + +#: The two keys that scope a destructive pass. Both are required to match: a +#: knowledge base carrying our project prefix but another environment's tag +#: belongs to that environment, and its name looks exactly like ours. +SCOPE_KEYS = (TAG_KEY_PREFIX, TAG_KEY_ENVIRONMENT) + +# ── Environment variables carrying the values ──────────────────────────────── +# +# Set by the CDK construct that owns the provisioning Lambdas, which is the only +# surface that calls `provision_managed_kb`. Named after the tag rather than after +# the project so it is obvious at the call site that changing one changes what +# gets written into AWS. +ENV_TAG_VALUE_PREFIX = "MANAGED_KB_TAG_VALUE_PREFIX" +ENV_TAG_VALUE_ENVIRONMENT = "MANAGED_KB_TAG_VALUE_ENVIRONMENT" + +#: Fallbacks, in order, for a local run or a service that predates the vars above. +#: Deliberately the *same* chain for writing and for filtering — an asymmetric +#: fallback is how a writer and a reader disagree while both look correct. +FALLBACK_PREFIX_VARS = ("PROJECT_PREFIX", "CDK_PROJECT_PREFIX") +FALLBACK_ENVIRONMENT_VARS = ("ENVIRONMENT", "CDK_ENVIRONMENT") + +#: Last resort. Kept so a local run works without configuration, and logged +#: loudly because two deployments that both fall back to it will claim each +#: other's knowledge bases — they would agree with themselves and delete each +#: other's corpora. +DEFAULT_PREFIX = "agentcore" +DEFAULT_ENVIRONMENT = "dev" + + +def _resolve(explicit: Optional[str], primary: str, fallbacks: tuple, default: str, what: str) -> str: + if explicit: + return explicit + value = os.environ.get(primary) + if value: + return value + for name in fallbacks: + value = os.environ.get(name) + if value: + logger.info( + f"managed KB {what} tag resolved from {name}; {primary} is not set. " + f"This is expected for a local run and unexpected in a deployment." + ) + return value + logger.warning( + f"managed KB {what} tag falling back to {default!r}: none of {primary} or " + f"{fallbacks} is set. Two deployments that both reach this default share a " + f"tag scope and will each treat the other's knowledge bases as their own." + ) + return default + + +def tag_prefix(explicit: Optional[str] = None) -> str: + """The project-prefix tag value.""" + return _resolve(explicit, ENV_TAG_VALUE_PREFIX, FALLBACK_PREFIX_VARS, DEFAULT_PREFIX, "prefix") + + +def tag_environment(explicit: Optional[str] = None) -> str: + """The environment tag value.""" + return _resolve( + explicit, + ENV_TAG_VALUE_ENVIRONMENT, + FALLBACK_ENVIRONMENT_VARS, + DEFAULT_ENVIRONMENT, + "environment", + ) + + +def build_tags( + app_kb_id: str, + owner_user_id: str, + project_prefix: Optional[str] = None, + environment: Optional[str] = None, +) -> Dict[str, str]: + """The complete tag set written at ``CreateKnowledgeBase`` time. + + The owner tag must be opaque (Requirement 20.12). An email address here would + put PII in a field readable by anyone holding + ``bedrock:ListKnowledgeBases``, and unlike a database column a tag cannot be + scrubbed retroactively from the audit trail it has already entered. An + address-shaped value is therefore rejected rather than trimmed: silently + dropping it would hide the caller's mistake. + """ + if "@" in owner_user_id: + raise ValueError( + "ownerUserId tag must be an opaque identifier, never an email address " + "or other personally identifying value (Requirement 20.12)" + ) + return { + TAG_KEY_PREFIX: tag_prefix(project_prefix), + TAG_KEY_ENVIRONMENT: tag_environment(environment), + TAG_KEY_APP_KB_ID: app_kb_id, + TAG_KEY_OWNER_USER_ID: owner_user_id, + } + + +def project_tag_filter( + project_prefix: Optional[str] = None, + environment: Optional[str] = None, +) -> Dict[str, str]: + """The subset of tags a knowledge base must carry to be considered ours. + + Derived from the same resolution :func:`build_tags` uses, not mirrored from + it. That is the entire point of this module: a reader that re-derives what the + writer wrote is a reader that can be wrong on its own. + """ + return { + TAG_KEY_PREFIX: tag_prefix(project_prefix), + TAG_KEY_ENVIRONMENT: tag_environment(environment), + } + + +def matches_project( + tags: Optional[Mapping[str, Any]], + project_prefix: Optional[str] = None, + environment: Optional[str] = None, +) -> bool: + """Whether these tags identify a knowledge base this deployment owns. + + ``False`` for absent or unreadable tags. Unknown ownership is not ownership, + and refusing to act on a resource we cannot attribute is the only safe + direction for a pass that deletes things. + """ + if not tags: + return False + expected = project_tag_filter(project_prefix, environment) + return all(str(tags.get(key, "")) == value for key, value in expected.items()) + + +__all__ = [ + "DEFAULT_ENVIRONMENT", + "DEFAULT_PREFIX", + "ENV_TAG_VALUE_ENVIRONMENT", + "ENV_TAG_VALUE_PREFIX", + "FALLBACK_ENVIRONMENT_VARS", + "FALLBACK_PREFIX_VARS", + "SCOPE_KEYS", + "TAG_KEY_APP_KB_ID", + "TAG_KEY_ENVIRONMENT", + "TAG_KEY_OWNER_USER_ID", + "TAG_KEY_PREFIX", + "build_tags", + "matches_project", + "project_tag_filter", + "tag_environment", + "tag_prefix", +] diff --git a/backend/src/apis/shared/kb_backend/tombstones.py b/backend/src/apis/shared/kb_backend/tombstones.py new file mode 100644 index 000000000..ac44796a4 --- /dev/null +++ b/backend/src/apis/shared/kb_backend/tombstones.py @@ -0,0 +1,921 @@ +"""Tombstoned deletion sagas for managed knowledge bases. + +Every delete here either completes or leaves a durable, retryable work item. That +is the whole requirement (Requirement 13), and it exists because a failed delete +of a managed knowledge base is not a crash — it is a **silent recurring bill**. + +The ordering is the mechanism +----------------------------- +1. Write the Tombstone to DynamoDB. +2. *Then* call AWS. +3. Poll until AWS reports the resource genuinely absent. +4. *Only then* clear the Tombstone. + +Reversing steps 1 and 2 looks equivalent and is not. A crash between the AWS call +and the database write leaves a half-deleted, still-billed resource that no record +points at, nothing alarms on, and no code will ever revisit. Written +tombstone-first, the same crash leaves a row that :func:`iter_tombstones` finds +(Requirement 13.8) and that a later pass can retry. + +Clearing early is the same defect wearing the opposite hat: a tombstone cleared on +the strength of an *accepted* delete call describes a resource AWS may still be +holding, and holding it is what costs money. So :func:`clear_kb_tombstone` is +never called on the accept path — it is reachable only after +:func:`confirm_knowledge_base_absent` has returned true. + +No TTL. Deliberately. +--------------------- +A Tombstone is cleared by confirmed deletion or it stays. Attaching a TTL would +let DynamoDB quietly remove the evidence of a delete that never finished, which +recreates precisely the silent-leak class this module exists to close. The same +reasoning bans a TTL on the KB_Record itself (Requirement 13.6): +:func:`remove_kb_record` refuses outright unless the caller can show confirmation. + +"Accepted" is not "gone" +------------------------ +``DeleteKnowledgeBase`` returns ``status: DELETING`` and the resource lives on for +a measured **2-6 minutes**. There is no waiter, so absence is established by +polling ``ListKnowledgeBases`` until the identifier stops appearing +(Requirement 13.4), with a window comfortably past the observed worst case. + +``DELETE_UNSUCCESSFUL`` is a terminal *operator* state, not a completed delete +(Requirement 13.7). The dev account has contained one since 2025-11-24 that no +reconciler would ever have noticed. Observing it stops the poll, records the state +on the Tombstone, and leaves the Tombstone standing. + +Why the tag filter costs a describe call per knowledge base +----------------------------------------------------------- +``ListKnowledgeBases`` has no tag-filter parameter and its summaries carry neither +``knowledgeBaseArn`` nor ``createdAt`` — verified against the packaged service +model, where ``KnowledgeBaseSummary`` is +``{knowledgeBaseId, name, description, status, updatedAt}``. Both of those are +needed: the ARN to read tags, and ``createdAt`` for the Reconciler's age gate. So +:func:`iter_project_knowledge_bases` pages the list and calls +``GetKnowledgeBase`` per entry. The alternative — synthesizing the ARN from the +region and account — trades a read call for a brittle string, and the caller here +is a daily job. + +Import boundary +--------------- +Module-level imports are stdlib plus this package's own stdlib-only modules; +``boto3`` and ``botocore`` are function-local. Nothing here imports +``apis.shared.assistants``, whose ``__init__`` pulls in the embeddings stack and +would blow the migration Lambda image budget. Enforced by +``tests/architecture/test_kb_backend_boundary.py``. +""" + +from __future__ import annotations + +import logging +import os +import time +from dataclasses import dataclass +from decimal import Decimal +from typing import Any, Callable, Dict, Iterator, List, Mapping, Optional + +from apis.shared.kb_backend.metrics import emit_count +from apis.shared.kb_backend.records import ( + document_tombstone_sk, + kb_pk, + kb_sk, + kb_tombstone_sk, +) + +logger = logging.getLogger(__name__) + +# ── Intents ────────────────────────────────────────────────────────────────── +# +# Recorded on the Tombstone so a retry knows which saga to resume without having +# to infer it from the sort key's shape. +INTENT_DELETE_KB = "delete_kb" +INTENT_DELETE_DOCUMENT = "delete_document" + +#: Attribute name flagging a tombstone whose ``PK`` is *not* a real assistant +#: partition. +#: +#: The reconciler deletes orphans — knowledge bases with no KB_Record — and an +#: orphan by definition carries no assistant id to anchor on, so its tombstone +#: lands in a partition derived from whatever identifier the tags did preserve. +#: That item is a genuine work record and must be kept, but a reader must not +#: mistake the partition for an assistant that exists, and +#: ``iter_tombstones()`` will never return it. This attribute +#: says so on the item, in place of a comment nobody triaging at 3am will read. +SYNTHETIC_PARTITION = "syntheticPartition" + +# ── AWS states ─────────────────────────────────────────────────────────────── +# +# Copied from the packaged service model's ``KnowledgeBaseStatus`` enum: +# ``CREATING | ACTIVE | DELETING | UPDATING | FAILED | DELETE_UNSUCCESSFUL | +# UPDATE_UNSUCCESSFUL``. +KB_STATUS_DELETING = "DELETING" +KB_STATUS_DELETE_UNSUCCESSFUL = "DELETE_UNSUCCESSFUL" + +#: From ``DocumentStatus``. A document AWS reports ``NOT_FOUND`` is gone; anything +#: else — including ``DELETING`` and ``DELETE_IN_PROGRESS`` — is still present. +DOCUMENT_STATUS_NOT_FOUND = "NOT_FOUND" + +# ── Poll windows (Requirement 13.4) ────────────────────────────────────────── +# +# Deletion was measured at 2-6 minutes, so the floor is 360 s and this sits above +# it. These are read *at call time* rather than bound as default arguments, +# because a default argument is evaluated once at import and cannot be patched: +# an earlier version of a sibling poller bound its timeout that way and a test +# that shortened the window had no effect at all, silently waiting the full +# production timeout instead. See `wait_until_retrievable` in the ingestion +# consumer for the same note. +KB_DELETE_POLL_TIMEOUT_SECONDS = 480.0 +KB_DELETE_POLL_INTERVAL_SECONDS = 10.0 + +DOCUMENT_DELETE_POLL_TIMEOUT_SECONDS = 120.0 +DOCUMENT_DELETE_POLL_INTERVAL_SECONDS = 2.0 + +#: ``ListKnowledgeBases`` page size. The list is always paged to exhaustion; this +#: only trades call count against payload size. +LIST_PAGE_SIZE = 100 + +# ── Metrics ────────────────────────────────────────────────────────────────── +METRIC_TOMBSTONE_WRITTEN = "KbTombstoneWritten" +METRIC_TOMBSTONE_CLEARED = "KbTombstoneCleared" + +#: A delete that was accepted but never confirmed. Sustained non-zero is the only +#: signal that the delete saga is leaking paid resources. +METRIC_TOMBSTONE_SURVIVED = "KbTombstoneSurvived" + +#: Requirement 13.7. Needs an alarm, not a dashboard: nothing clears this state +#: on its own. +METRIC_DELETE_UNSUCCESSFUL = "KbDeleteUnsuccessful" + + +class TombstoneError(RuntimeError): + """A tombstoned delete could not be completed.""" + + +class DeleteNotConfirmed(TombstoneError): + """AWS never reported the resource absent within the poll window. + + Retryable. The Tombstone is deliberately left in place, so the work item + outlives this process. + """ + + +class DeleteUnsuccessful(TombstoneError): + """AWS reported ``DELETE_UNSUCCESSFUL`` (Requirement 13.7). + + Distinct from :class:`DeleteNotConfirmed` because it is *not* a matter of + waiting longer. It is an actionable operator state that persists until someone + intervenes, and it must never be mistaken for a completed delete. + """ + + +class ServiceRoleStillInUse(TombstoneError): + """Refuses to delete a service role that still has knowledge bases. + + Requirement 13.5. Removing the role first is a documented route *into* + ``DELETE_UNSUCCESSFUL``: the pending deletion needs the role it was created + with, and without it the knowledge base can be neither deleted nor recovered. + """ + + +class RecordRemovalRefused(TombstoneError): + """Refuses to remove a KB_Record before AWS confirmed the deletion. + + Requirement 13.6. The record is the only pointer to the AWS identifiers, so + dropping it early converts a retryable delete into an untraceable one. + """ + + +@dataclass(frozen=True) +class KnowledgeBaseFacts: + """What AWS says about one knowledge base. + + ``created_at`` is **AWS's own** ``createdAt``, carried through unmodified. The + Reconciler's age gate depends on that provenance: substituting the time this + process happened to look would make a reconciler that was down for a week + treat every knowledge base in the account as brand new (Requirement 14.3). + """ + + kb_id: str + name: str + status: str + arn: Optional[str] = None + created_at: Optional[Any] = None + tags: Mapping[str, str] = None # type: ignore[assignment] + + +@dataclass(frozen=True) +class DeleteOutcome: + """The result of one saga run. + + ``confirmed`` means AWS reported the resource absent — the only condition + under which the Tombstone was cleared. ``tombstone_cleared`` is reported + separately rather than inferred so a test can catch the two drifting apart. + """ + + confirmed: bool + tombstone_cleared: bool + already_absent: bool = False + delete_unsuccessful: bool = False + polls: int = 0 + + +# ── DynamoDB plumbing ──────────────────────────────────────────────────────── +def _table(): + import boto3 + + return boto3.resource("dynamodb").Table(os.environ["DYNAMODB_ASSISTANTS_TABLE_NAME"]) + + +def _now_iso() -> str: + from apis.shared.timestamps import utc_now_iso + + return utc_now_iso() + + +# ── Tombstone writes ───────────────────────────────────────────────────────── +def _write_tombstone( + assistant_id: str, + sort_key: str, + intent: str, + attributes: Mapping[str, Any], +) -> Dict[str, Any]: + """Upsert a Tombstone, preserving the original ``createdAt`` and counting attempts. + + An upsert rather than a ``put_item`` because a retried saga must not restart + the clock. ``createdAt`` is written through ``if_not_exists`` so it records + when the delete was *first* attempted — the number an operator triaging a + stuck tombstone actually wants — while ``attempts`` accumulates with ``ADD``, + which is atomic and needs no read. + + No ``ttl`` attribute is written, and none may be added. See the module + docstring: expiry would silently discard the evidence of an unfinished delete. + """ + now = _now_iso() + values: Dict[str, Any] = { + ":intent": intent, + ":now": now, + ":one": Decimal(1), + } + sets = [ + "intent = :intent", + "createdAt = if_not_exists(createdAt, :now)", + "updatedAt = :now", + ] + for index, (key, value) in enumerate(sorted(attributes.items())): + if value is None: + continue + placeholder = f":a{index}" + sets.append(f"{key} = {placeholder}") + values[placeholder] = value + + _table().update_item( + Key={"PK": kb_pk(assistant_id), "SK": sort_key}, + UpdateExpression=f"SET {', '.join(sets)} ADD attempts :one", + ExpressionAttributeValues=values, + ) + emit_count(METRIC_TOMBSTONE_WRITTEN, dimensions={"intent": intent}) + return {"PK": kb_pk(assistant_id), "SK": sort_key, "intent": intent, "createdAt": now} + + +def write_kb_tombstone( + assistant_id: str, + app_kb_id: str, + aws_kb_id: Optional[str] = None, + aws_data_source_id: Optional[str] = None, + extra_attributes: Optional[Mapping[str, Any]] = None, +) -> Dict[str, Any]: + """Mark a whole-knowledge-base delete as intended. Call this *before* AWS. + + ``extra_attributes`` lets a caller that is not deleting on behalf of a known + assistant say so on the item itself — see ``SYNTHETIC_PARTITION`` below. + """ + attributes: Dict[str, Any] = { + "appKbId": app_kb_id, + "awsKbId": aws_kb_id, + "awsDataSourceId": aws_data_source_id, + } + if extra_attributes: + attributes.update(extra_attributes) + return _write_tombstone( + assistant_id, + kb_tombstone_sk(app_kb_id), + INTENT_DELETE_KB, + attributes, + ) + + +def write_document_tombstone( + assistant_id: str, + app_kb_id: str, + document_id: str, + aws_kb_id: Optional[str] = None, + aws_data_source_id: Optional[str] = None, +) -> Dict[str, Any]: + """Mark a single-document delete as intended. Call this *before* AWS.""" + return _write_tombstone( + assistant_id, + document_tombstone_sk(app_kb_id, document_id), + INTENT_DELETE_DOCUMENT, + { + "appKbId": app_kb_id, + "documentId": document_id, + "awsKbId": aws_kb_id, + "awsDataSourceId": aws_data_source_id, + }, + ) + + +def record_tombstone_error( + assistant_id: str, + sort_key: str, + error: str, + aws_status: Optional[str] = None, +) -> None: + """Annotate a surviving Tombstone with why it survived. + + Never raises. The saga has already failed by the time this is reached, and + losing the annotation is strictly better than replacing a precise failure with + a DynamoDB error from the bookkeeping. + """ + sets = ["lastError = :err", "updatedAt = :now"] + values: Dict[str, Any] = {":err": error[:1024], ":now": _now_iso()} + if aws_status: + sets.append("awsStatus = :status") + values[":status"] = aws_status + + try: + _table().update_item( + Key={"PK": kb_pk(assistant_id), "SK": sort_key}, + UpdateExpression=f"SET {', '.join(sets)}", + ExpressionAttributeValues=values, + ) + except Exception as exc: # noqa: BLE001 - bookkeeping must not mask the real failure + logger.warning(f"could not annotate tombstone {sort_key}: {exc}") + + +def _clear(assistant_id: str, sort_key: str, intent: str) -> bool: + _table().delete_item(Key={"PK": kb_pk(assistant_id), "SK": sort_key}) + emit_count(METRIC_TOMBSTONE_CLEARED, dimensions={"intent": intent}) + return True + + +def clear_kb_tombstone(assistant_id: str, app_kb_id: str, confirmed_absent: bool) -> bool: + """Clear a knowledge-base Tombstone. Refuses unless AWS confirmed absence. + + ``confirmed_absent`` is a required positional argument rather than a keyword + with a convenient default, because the failure mode being guarded against is a + caller who *forgot* the confirmation step. A default of ``True`` would make + the unsafe call the short one; there is no default at all, so the caller has + to state what it knows. + """ + if not confirmed_absent: + raise TombstoneError( + f"refusing to clear the tombstone for kb {app_kb_id}: AWS has not " + f"confirmed the knowledge base is absent. An accepted delete call is " + f"not a completed deletion (Requirement 13.3), and clearing here " + f"would discard the only work item for a resource still being billed." + ) + return _clear(assistant_id, kb_tombstone_sk(app_kb_id), INTENT_DELETE_KB) + + +def clear_document_tombstone( + assistant_id: str, + app_kb_id: str, + document_id: str, + confirmed_absent: bool, +) -> bool: + """Clear a document Tombstone. Refuses unless AWS confirmed absence.""" + if not confirmed_absent: + raise TombstoneError( + f"refusing to clear the tombstone for document {document_id}: AWS has " + f"not confirmed it is absent" + ) + return _clear( + assistant_id, document_tombstone_sk(app_kb_id, document_id), INTENT_DELETE_DOCUMENT + ) + + +def iter_tombstones(assistant_id: str) -> List[Dict[str, Any]]: + """Surviving Tombstones for one assistant, as retryable work items (Req 13.8). + + Keyed on the ``KBTOMB#`` prefix, so a whole-KB tombstone and its documents' + tombstones come back together and in that order — which is the order a retry + wants them. + """ + from boto3.dynamodb.conditions import Key + + response = _table().query( + KeyConditionExpression=Key("PK").eq(kb_pk(assistant_id)) + & Key("SK").begins_with("KBTOMB#") + ) + return response.get("Items", []) + + +def remove_kb_record(assistant_id: str, app_kb_id: str, confirmed_absent: bool) -> None: + """Delete the KB_Record. Refuses unless AWS confirmed the deletion (Req 13.6). + + The record holds the only mapping from ``App_KB_Id`` to the AWS identifiers. + Removing it while AWS still holds the knowledge base turns a resource that a + tombstone could still find into one nothing can address — the exact leak this + module exists to prevent, produced by the cleanup step rather than the crash. + """ + if not confirmed_absent: + raise RecordRemovalRefused( + f"refusing to remove the KB_Record for {app_kb_id} before AWS confirms " + f"deletion (Requirement 13.6); the record is the only pointer to the " + f"AWS identifiers" + ) + _table().delete_item(Key={"PK": kb_pk(assistant_id), "SK": kb_sk(app_kb_id)}) + + +# ── AWS listing, tag-filtered and paginated (Requirement 14.1) ─────────────── +# +# The tag contract itself lives in ``kb_backend.tags``. These two functions are +# thin re-exports kept for their existing callers. +# +# ⚠️ They used to be hand-written *mirrors* of ``provisioning.build_tags``, +# documented as such — and they had drifted: different key names +# (``prefix``/``env`` against the construct's ``ManagedKbPrefix``/ +# ``ManagedKbEnvironment``) and values read from environment variables the +# provisioning Lambda is never given. A mirror is a second implementation, and the +# only thing keeping two implementations equal is that nobody has edited one yet. +def project_tag_filter( + project_prefix: Optional[str] = None, + environment: Optional[str] = None, +) -> Dict[str, str]: + """The tags that identify this platform's knowledge bases. + + Only the two scope keys are matched: the app id and owner id vary per resource + and are identity, not scope. + """ + from apis.shared.kb_backend.tags import project_tag_filter as _canonical + + return _canonical(project_prefix, environment) + + +def matches_project_tags(tags: Optional[Mapping[str, str]], expected: Mapping[str, str]) -> bool: + """True when every expected tag is present with the expected value. + + Absent or empty tags never match. An untagged knowledge base is out of scope + by construction, which is the conservative direction: this predicate gates + deletion, so a false negative leaves a resource alone while a false positive + deletes someone else's. + """ + if not tags: + return False + return all(str(tags.get(key, "")) == value for key, value in expected.items()) + + +def iter_knowledge_base_summaries(client, page_size: Optional[int] = None) -> Iterator[Dict[str, Any]]: + """Every ``KnowledgeBaseSummary`` in the account, paging to exhaustion. + + Hand-rolled paging rather than ``get_paginator`` so that a stubbed client in a + test is a plain object with one method, not something that has to satisfy + botocore's paginator protocol. Reading only the first page would make the + Reconciler's judgement depend on account size: every knowledge base past page + one would look like a missing-vector record and every orphan there would go + unbilled-for-ever. + """ + if page_size is None: + page_size = LIST_PAGE_SIZE + + token: Optional[str] = None + while True: + kwargs: Dict[str, Any] = {"maxResults": page_size} + if token: + kwargs["nextToken"] = token + response = client.list_knowledge_bases(**kwargs) + for summary in response.get("knowledgeBaseSummaries") or []: + yield summary + token = response.get("nextToken") + if not token: + return + + +def describe_knowledge_base(client, kb_id: str) -> Optional[Dict[str, Any]]: + """``GetKnowledgeBase``, or ``None`` if it has already gone. + + A ``ResourceNotFoundException`` between the list and the describe is normal — + something else deleted it, or this saga's own earlier attempt finally landed — + and means exactly what the caller wants to know. + """ + from botocore.exceptions import ClientError + + try: + response = client.get_knowledge_base(knowledgeBaseId=kb_id) + except ClientError as exc: + if exc.response.get("Error", {}).get("Code") == "ResourceNotFoundException": + return None + raise + return response.get("knowledgeBase") or None + + +def knowledge_base_tags(client, arn: str) -> Dict[str, str]: + """Tags for one knowledge base. A read failure yields ``{}``, never a match. + + Failing closed matters here: ``{}`` cannot satisfy + :func:`matches_project_tags`, so a knowledge base whose tags could not be read + is left alone rather than deleted on the strength of a failed lookup. + """ + from botocore.exceptions import ClientError + + try: + return dict((client.list_tags_for_resource(resourceArn=arn) or {}).get("tags") or {}) + except ClientError as exc: + logger.warning(f"could not read tags for {arn}: {exc}") + return {} + + +def iter_project_knowledge_bases( + client, + project_prefix: Optional[str] = None, + environment: Optional[str] = None, + page_size: Optional[int] = None, +) -> Iterator[KnowledgeBaseFacts]: + """This project's knowledge bases, with AWS's ``createdAt`` and status. + + Paginated (Requirement 14.1) and tag-filtered. The filter is applied to tags + read from AWS rather than to the name, because a name is a convention this + code chose and a tag is a fact recorded on the resource: a knowledge base + created by an older naming scheme is still ours, and one that merely happens + to share our prefix is not. + """ + expected = project_tag_filter(project_prefix, environment) + + for summary in iter_knowledge_base_summaries(client, page_size=page_size): + kb_id = summary.get("knowledgeBaseId") + if not kb_id: + continue + + described = describe_knowledge_base(client, kb_id) + if described is None: + continue + + arn = described.get("knowledgeBaseArn") + tags = knowledge_base_tags(client, arn) if arn else {} + if not matches_project_tags(tags, expected): + continue + + yield KnowledgeBaseFacts( + kb_id=kb_id, + name=described.get("name") or summary.get("name") or "", + status=described.get("status") or summary.get("status") or "", + arn=arn, + # AWS's own timestamp, untouched. See KnowledgeBaseFacts. + created_at=described.get("createdAt"), + tags=tags, + ) + + +# ── Confirmation by polling (Requirement 13.3, 13.4) ───────────────────────── +def confirm_knowledge_base_absent( + client, + aws_kb_id: str, + timeout_seconds: Optional[float] = None, + interval_seconds: Optional[float] = None, + sleep: Callable[[float], None] = time.sleep, + monotonic: Callable[[], float] = time.monotonic, +) -> DeleteOutcome: + """Poll ``ListKnowledgeBases`` until ``aws_kb_id`` stops appearing. + + Absence is established from the **list**, not from the delete call's return + value and not from a single ``GetKnowledgeBase``: the delete is asynchronous + and returns ``DELETING`` while the resource is still there and still billed. + + Returns as soon as the identifier is gone. Raises :class:`DeleteUnsuccessful` + the moment ``DELETE_UNSUCCESSFUL`` is observed — that state does not resolve + by waiting, so continuing to poll would burn the window and then report the + wrong reason. Raises :class:`DeleteNotConfirmed` on timeout. + + Both windows resolve from the module constants *at call time*. Bound as + default arguments they would be fixed at import and unpatchable, and a test + that shortened them would sit through the full production wait while + appearing to pass. + """ + if timeout_seconds is None: + timeout_seconds = KB_DELETE_POLL_TIMEOUT_SECONDS + if interval_seconds is None: + interval_seconds = KB_DELETE_POLL_INTERVAL_SECONDS + + deadline = monotonic() + timeout_seconds + polls = 0 + + while True: + polls += 1 + present: Optional[Dict[str, Any]] = None + for summary in iter_knowledge_base_summaries(client): + if summary.get("knowledgeBaseId") == aws_kb_id: + present = summary + break + + if present is None: + return DeleteOutcome(confirmed=True, tombstone_cleared=False, polls=polls) + + status = present.get("status") or "" + if status == KB_STATUS_DELETE_UNSUCCESSFUL: + emit_count(METRIC_DELETE_UNSUCCESSFUL) + raise DeleteUnsuccessful( + f"knowledge base {aws_kb_id} is in {KB_STATUS_DELETE_UNSUCCESSFUL}. " + f"This is an operator state, not a completed delete: it does not " + f"clear on its own and the resource is still billed. The tombstone " + f"is being left in place as the work item." + ) + + if monotonic() >= deadline: + raise DeleteNotConfirmed( + f"knowledge base {aws_kb_id} still present after {timeout_seconds}s " + f"(last status {status!r}) across {polls} polls; leaving the " + f"tombstone as a retryable work item" + ) + + sleep(interval_seconds) + + +def confirm_document_absent( + client, + aws_kb_id: str, + aws_data_source_id: str, + document_id: str, + timeout_seconds: Optional[float] = None, + interval_seconds: Optional[float] = None, + sleep: Callable[[float], None] = time.sleep, + monotonic: Callable[[], float] = time.monotonic, +) -> DeleteOutcome: + """Poll ``GetKnowledgeBaseDocuments`` until the document reports ``NOT_FOUND``. + + Only ``NOT_FOUND`` (or an empty detail list) counts as absent. ``DELETING`` + and ``DELETE_IN_PROGRESS`` are explicitly *present*: treating them as done is + the document-scale version of trusting the accepted delete call. + """ + from apis.shared.kb_backend.managed_backend import document_identifier + + if timeout_seconds is None: + timeout_seconds = DOCUMENT_DELETE_POLL_TIMEOUT_SECONDS + if interval_seconds is None: + interval_seconds = DOCUMENT_DELETE_POLL_INTERVAL_SECONDS + + deadline = monotonic() + timeout_seconds + polls = 0 + + while True: + polls += 1 + response = client.get_knowledge_base_documents( + knowledgeBaseId=aws_kb_id, + dataSourceId=aws_data_source_id, + documentIdentifiers=[document_identifier(document_id)], + ) + details = response.get("documentDetails") or [] + statuses = {detail.get("status") for detail in details} + + if not details or statuses <= {DOCUMENT_STATUS_NOT_FOUND}: + return DeleteOutcome(confirmed=True, tombstone_cleared=False, polls=polls) + + if monotonic() >= deadline: + raise DeleteNotConfirmed( + f"document {document_id} still present in kb {aws_kb_id} after " + f"{timeout_seconds}s (statuses {sorted(s for s in statuses if s)}); " + f"leaving the tombstone as a retryable work item" + ) + + sleep(interval_seconds) + + +# ── Sagas ──────────────────────────────────────────────────────────────────── +def delete_knowledge_base( + assistant_id: str, + app_kb_id: str, + aws_kb_id: str, + aws_data_source_id: Optional[str] = None, + client=None, + remove_record: bool = False, + extra_attributes: Optional[Mapping[str, Any]] = None, + timeout_seconds: Optional[float] = None, + interval_seconds: Optional[float] = None, + sleep: Callable[[float], None] = time.sleep, + monotonic: Callable[[], float] = time.monotonic, +) -> DeleteOutcome: + """Delete a knowledge base under a Tombstone. + + The four steps run strictly in order, and the order is the guarantee: + + 1. **Tombstone first.** Written before any AWS call, so a crash anywhere below + leaves a work item rather than a resource nothing knows about. + 2. **Ask AWS.** ``ResourceNotFoundException`` is success, not failure — an + earlier attempt got there, and the tombstone should still be cleared. + 3. **Confirm by polling.** The accepted call is ignored as evidence. + 4. **Clear the Tombstone**, and only now, optionally, the KB_Record. + + On any failure the Tombstone survives, annotated with the reason, and the + exception propagates so the invocation fails and its retry or DLQ fires. + """ + from apis.shared.kb_backend.managed_backend import bedrock_agent_client + from botocore.exceptions import ClientError + + if client is None: + client = bedrock_agent_client() + + sort_key = kb_tombstone_sk(app_kb_id) + + # Step 1. Before AWS. Always. + write_kb_tombstone( + assistant_id, + app_kb_id, + aws_kb_id, + aws_data_source_id, + extra_attributes=extra_attributes, + ) + + already_absent = False + try: + # Step 2. + try: + client.delete_knowledge_base(knowledgeBaseId=aws_kb_id) + except ClientError as exc: + if exc.response.get("Error", {}).get("Code") != "ResourceNotFoundException": + raise + already_absent = True + logger.info( + f"knowledge base {aws_kb_id} was already absent; treating the " + f"delete as complete and clearing its tombstone" + ) + + # Step 3. "Accepted" is not "gone" — establish absence from the list. + outcome = confirm_knowledge_base_absent( + client, + aws_kb_id, + timeout_seconds=timeout_seconds, + interval_seconds=interval_seconds, + sleep=sleep, + monotonic=monotonic, + ) + except DeleteUnsuccessful as exc: + record_tombstone_error( + assistant_id, sort_key, str(exc), aws_status=KB_STATUS_DELETE_UNSUCCESSFUL + ) + raise + except Exception as exc: + emit_count(METRIC_TOMBSTONE_SURVIVED, dimensions={"intent": INTENT_DELETE_KB}) + record_tombstone_error(assistant_id, sort_key, str(exc)) + raise + + # Step 4. Reachable only with confirmation in hand. + clear_kb_tombstone(assistant_id, app_kb_id, outcome.confirmed) + + if remove_record: + remove_kb_record(assistant_id, app_kb_id, outcome.confirmed) + + return DeleteOutcome( + confirmed=True, + tombstone_cleared=True, + already_absent=already_absent, + polls=outcome.polls, + ) + + +def delete_document( + assistant_id: str, + app_kb_id: str, + document_id: str, + aws_kb_id: str, + aws_data_source_id: str, + client=None, + timeout_seconds: Optional[float] = None, + interval_seconds: Optional[float] = None, + sleep: Callable[[float], None] = time.sleep, + monotonic: Callable[[], float] = time.monotonic, +) -> DeleteOutcome: + """Delete one document under a Tombstone. Same ordering as the KB saga.""" + from apis.shared.kb_backend.managed_backend import ( + bedrock_agent_client, + document_identifier, + ) + + if client is None: + client = bedrock_agent_client() + + sort_key = document_tombstone_sk(app_kb_id, document_id) + + # Step 1. Before AWS. Always. + write_document_tombstone( + assistant_id, app_kb_id, document_id, aws_kb_id, aws_data_source_id + ) + + try: + client.delete_knowledge_base_documents( + knowledgeBaseId=aws_kb_id, + dataSourceId=aws_data_source_id, + documentIdentifiers=[document_identifier(document_id)], + ) + outcome = confirm_document_absent( + client, + aws_kb_id, + aws_data_source_id, + document_id, + timeout_seconds=timeout_seconds, + interval_seconds=interval_seconds, + sleep=sleep, + monotonic=monotonic, + ) + except Exception as exc: + emit_count(METRIC_TOMBSTONE_SURVIVED, dimensions={"intent": INTENT_DELETE_DOCUMENT}) + record_tombstone_error(assistant_id, sort_key, str(exc)) + raise + + clear_document_tombstone(assistant_id, app_kb_id, document_id, outcome.confirmed) + return DeleteOutcome(confirmed=True, tombstone_cleared=True, polls=outcome.polls) + + +# ── Service-role teardown guard (Requirement 13.5) ─────────────────────────── +def knowledge_bases_using_role( + client, + role_arn: str, + project_prefix: Optional[str] = None, + environment: Optional[str] = None, +) -> List[str]: + """Identifiers of this project's knowledge bases still using ``role_arn``. + + Read from ``GetKnowledgeBase``'s ``roleArn`` rather than from our own records, + because the question is what AWS still believes — and a knowledge base our + database has forgotten is exactly the one that makes deleting the role + dangerous. + """ + outstanding: List[str] = [] + for facts in iter_project_knowledge_bases( + client, project_prefix=project_prefix, environment=environment + ): + described = describe_knowledge_base(client, facts.kb_id) + if described is None: + continue + if described.get("roleArn") == role_arn: + outstanding.append(facts.kb_id) + return outstanding + + +def assert_service_role_deletable( + client, + role_arn: str, + project_prefix: Optional[str] = None, + environment: Optional[str] = None, +) -> None: + """Raise unless every knowledge base using ``role_arn`` is confirmed absent. + + Called by teardown before it touches the role. A knowledge base mid-``DELETING`` + still counts as present: it needs the role to finish, and pulling the role out + from under it is a documented route into ``DELETE_UNSUCCESSFUL``, which is + unrecoverable without support. + """ + outstanding = knowledge_bases_using_role( + client, role_arn, project_prefix=project_prefix, environment=environment + ) + if outstanding: + raise ServiceRoleStillInUse( + f"refusing to delete service role {role_arn}: {len(outstanding)} " + f"knowledge base(s) still reference it ({', '.join(sorted(outstanding))}). " + f"Delete them and confirm their absence first (Requirement 13.5); " + f"removing the role while one is still DELETING can strand it in " + f"{KB_STATUS_DELETE_UNSUCCESSFUL}." + ) + + +__all__ = [ + "DOCUMENT_DELETE_POLL_INTERVAL_SECONDS", + "DOCUMENT_DELETE_POLL_TIMEOUT_SECONDS", + "DOCUMENT_STATUS_NOT_FOUND", + "INTENT_DELETE_DOCUMENT", + "INTENT_DELETE_KB", + "KB_DELETE_POLL_INTERVAL_SECONDS", + "KB_DELETE_POLL_TIMEOUT_SECONDS", + "KB_STATUS_DELETE_UNSUCCESSFUL", + "KB_STATUS_DELETING", + "LIST_PAGE_SIZE", + "METRIC_DELETE_UNSUCCESSFUL", + "METRIC_TOMBSTONE_CLEARED", + "METRIC_TOMBSTONE_SURVIVED", + "METRIC_TOMBSTONE_WRITTEN", + "SYNTHETIC_PARTITION", + "DeleteNotConfirmed", + "DeleteOutcome", + "DeleteUnsuccessful", + "KnowledgeBaseFacts", + "RecordRemovalRefused", + "ServiceRoleStillInUse", + "TombstoneError", + "assert_service_role_deletable", + "clear_document_tombstone", + "clear_kb_tombstone", + "confirm_document_absent", + "confirm_knowledge_base_absent", + "delete_document", + "delete_knowledge_base", + "describe_knowledge_base", + "iter_knowledge_base_summaries", + "iter_project_knowledge_bases", + "iter_tombstones", + "knowledge_base_tags", + "knowledge_bases_using_role", + "matches_project_tags", + "project_tag_filter", + "record_tombstone_error", + "remove_kb_record", + "write_document_tombstone", + "write_kb_tombstone", +] diff --git a/backend/src/apis/shared/observability/emf.py b/backend/src/apis/shared/observability/emf.py index 99aeec50a..3844b64cc 100644 --- a/backend/src/apis/shared/observability/emf.py +++ b/backend/src/apis/shared/observability/emf.py @@ -114,6 +114,55 @@ def emit_prompt_cache_metrics( logger.debug("EMF emission skipped: %s", e) +def emit_emf_metrics( + namespace: str, + metrics: dict, + properties: Optional[dict] = None, + units: Optional[dict] = None, +) -> None: + """Emit one EMF record into ``namespace``. Never raises. + + The generic form of the two functions above, for callers whose namespace is not + the prompt-cache one. It lives here rather than being re-implemented per feature + because the parts that are easy to get wrong are not the JSON — they are the + dedicated non-propagating logger and the message-only formatter above. A record + written through the app's normal logger acquires an ``[INFO] name:`` prefix, + CloudWatch Logs silently declines to extract it, and the metric simply never + appears. Nothing errors; there is just no data, which is indistinguishable from + "the thing being measured never happened". + + ``metrics`` maps metric name to numeric value; ``units`` optionally maps the + same names to a CloudWatch unit, defaulting to ``None`` (a bare number). + ``properties`` ride along as queryable log fields and are **not** dimensions — + dimensions multiply metric streams, and every caller here so far wants + fleet-wide aggregates with the detail available in Logs Insights. + """ + try: + units = units or {} + record = { + "_aws": { + "Timestamp": int(time.time() * 1000), + "CloudWatchMetrics": [ + { + "Namespace": namespace, + "Dimensions": [[]], + "Metrics": [ + {"Name": name, "Unit": units.get(name, "None")} + for name in metrics + ], + } + ], + }, + } + record.update({name: value for name, value in metrics.items()}) + for key, value in (properties or {}).items(): + if value is not None: + record[key] = value + _emf_logger.info(json.dumps(record, separators=(",", ":"))) + except Exception as e: # noqa: BLE001 - metrics must never break a caller + logger.debug("EMF emission skipped: %s", e) + + def emit_session_cache_rollup( session_id: str, partial_miss_usd: float, diff --git a/backend/tests/apis/app_api/shares/test_share_s3_offload.py b/backend/tests/apis/app_api/shares/test_share_s3_offload.py index 13a8efe5d..e51d7617b 100644 --- a/backend/tests/apis/app_api/shares/test_share_s3_offload.py +++ b/backend/tests/apis/app_api/shares/test_share_s3_offload.py @@ -136,6 +136,14 @@ async def test_large_conversation_succeeds(self, service): @pytest.mark.asyncio async def test_storage_unavailable_raises(self, monkeypatch): monkeypatch.setenv("SHARED_CONVERSATIONS_TABLE_NAME", "shares-table") + # `bucket_name=None` means "no bucket configured", but the store falls back + # to this variable when the argument is None — so the test only asserted + # what it meant to while the variable happened to be absent from the + # environment. Any developer with a populated `backend/src/.env` (which + # `load_dotenv(override=True)` reads) gave the store a real bucket, and the + # assertion became a live S3 HeadObject against it. Deleted explicitly so + # "unavailable" is a property of the test rather than of the machine. + monkeypatch.delenv("SHARED_CONVERSATIONS_BUCKET_NAME", raising=False) with patch("boto3.resource"): svc = ShareService(snapshot_store=ShareSnapshotStore(bucket_name=None)) svc._table = MagicMock() diff --git a/backend/tests/architecture/test_kb_backend_boundary.py b/backend/tests/architecture/test_kb_backend_boundary.py new file mode 100644 index 000000000..ef347b9a9 --- /dev/null +++ b/backend/tests/architecture/test_kb_backend_boundary.py @@ -0,0 +1,194 @@ +"""Import-boundary enforcement for ``apis.shared.kb_backend``. + +``apis/shared/assistants/__init__.py`` imports ``rag_service``, which imports the +embeddings stack at module scope. So importing anything from the assistants +package pulls in that whole tree — and ``kb_backend`` is bundled into +size-constrained Lambda images (the migration dispatcher, worker, reconciler and +ingestion consumer) that deliberately do not carry it. The same constraint is why +``apis/app_api/kb_sync/records.py`` reaches DynamoDB through the raw table +resource instead of the assistants package. + +The dependency is also the wrong way round architecturally: the facade in +``rag_service`` sits *above* the seam and depends on ``kb_backend``. An import in +the other direction would make the two mutually dependent and the seam +meaningless. + +This is checked two ways, because either alone is insufficient: + +* **Statically**, so a *lazy* import inside a function body is caught. A deferred + import does not fail at module load; it fails at call time, in production, in a + Lambda that has been running fine for a week. +* **At runtime in a fresh interpreter**, so a transitive import through some + innocuous-looking third module is caught too. Static analysis cannot see + through an import chain; a subprocess with an empty ``sys.modules`` can. + +Feature: managed-kb-migration +Requirements: 24.15 +""" + +import ast +import subprocess +import sys +from pathlib import Path +from typing import List, Tuple + +import pytest + +_BACKEND_ROOT = Path(__file__).resolve().parent.parent.parent +_BACKEND_SRC = _BACKEND_ROOT / "src" +_KB_BACKEND = _BACKEND_SRC / "apis" / "shared" / "kb_backend" + +#: Modules whose absence from a fresh import is asserted. ``boto3`` is here +#: because it is the single largest dependency these Lambdas would otherwise +#: pay for, and keeping it function-local is the convention this package follows. +_FORBIDDEN_AT_IMPORT_TIME = ("apis.shared.assistants", "boto3") + + +def _extract_imports(filepath: Path) -> List[Tuple[str, int]]: + """Every imported module path in a file, including imports inside functions.""" + try: + tree = ast.parse(filepath.read_text(encoding="utf-8"), filename=str(filepath)) + except (SyntaxError, UnicodeDecodeError): + return [] + + imports: List[Tuple[str, int]] = [] + for node in ast.walk(tree): + if isinstance(node, ast.Import): + for alias in node.names: + imports.append((alias.name, node.lineno)) + elif isinstance(node, ast.ImportFrom) and node.module: + imports.append((node.module, node.lineno)) + return imports + + +def _kb_backend_files() -> List[Path]: + return sorted(_KB_BACKEND.rglob("*.py")) + + +class TestKbBackendDoesNotImportAssistants: + """No file in kb_backend may import apis.shared.assistants, at any depth.""" + + def test_no_assistants_imports_anywhere(self): + if not _KB_BACKEND.exists(): + pytest.skip("kb_backend package not found") + + violations = [] + for pyfile in _kb_backend_files(): + rel = pyfile.relative_to(_BACKEND_SRC) + for module, lineno in _extract_imports(pyfile): + if module == "apis.shared.assistants" or module.startswith("apis.shared.assistants."): + violations.append(f" {rel}:{lineno} imports '{module}'") + + assert violations == [], ( + "apis.shared.kb_backend must not import apis.shared.assistants " + "(its __init__ pulls in rag_service and the whole embeddings stack, " + "which the migration Lambda images do not carry):\n" + + "\n".join(violations) + + "\n\nNote that a lazy, function-local import does not fix this — it " + "moves the failure from image build to production call time." + ) + + def test_package_init_stays_empty(self): + """An empty ``__init__`` is what makes importing one submodule cheap. + + Re-exporting anything here would mean importing ``kb_backend.records`` + also imports every sibling — including, eventually, the managed backend + and its boto3 client. + """ + init = _KB_BACKEND / "__init__.py" + assert init.exists(), "kb_backend/__init__.py must exist" + assert init.read_text(encoding="utf-8").strip() == "", ( + "kb_backend/__init__.py must stay empty: it is imported by every " + "submodule import, so anything placed here is paid for by all of them" + ) + + +class TestKbBackendFreshImportIsLean: + """Importing a kb_backend submodule must not pull the heavy tree in. + + Each case runs in a fresh interpreter, because by the time this test file + executes, the rest of the suite has already imported both forbidden modules + into ``sys.modules`` — an in-process check would pass no matter what. + """ + + @staticmethod + def _import_and_report(module: str) -> List[str]: + """Import *module* in a subprocess; return which forbidden modules loaded.""" + program = ( + "import sys\n" + f"import {module}\n" + "loaded = [name for name in sys.modules\n" + f" if any(name == f or name.startswith(f + '.') for f in {_FORBIDDEN_AT_IMPORT_TIME!r})]\n" + "print(','.join(sorted(set(loaded))))\n" + ) + result = subprocess.run( + [sys.executable, "-c", program], + capture_output=True, + text=True, + cwd=str(_BACKEND_ROOT), + env={"PYTHONPATH": str(_BACKEND_SRC), "PATH": "/usr/bin:/bin"}, + ) + assert result.returncode == 0, ( + f"importing {module} in a clean interpreter failed:\n{result.stderr}" + ) + return [name for name in result.stdout.strip().split(",") if name] + + def test_records_import_is_stdlib_only(self): + """The constraint as written in task 4.8: records pulls in neither.""" + loaded = self._import_and_report("apis.shared.kb_backend.records") + assert loaded == [], ( + "importing apis.shared.kb_backend.records loaded " + f"{loaded}. Module-level imports in this package must be stdlib " + "only; move boto3 and anything from apis.shared.assistants into the " + "functions that need them." + ) + + @pytest.mark.parametrize( + "module", + [ + "apis.shared.kb_backend.protocol", + "apis.shared.kb_backend.resolver", + "apis.shared.kb_backend.s3vectors_backend", + "apis.shared.kb_backend.managed_backend", + "apis.shared.kb_backend.dual_read", + ], + ) + def test_seam_modules_import_lean(self, module): + """The resolver and both adapters obey the same rule as records. + + The resolver is the one that matters most: the facade imports it on every + retrieval, and it in turn imports every registered backend. If it were + not lean, no submodule of this package could be. + + ``managed_backend`` is on this list because the resolver **registers** it at + import (see the resolver's docstring). That registration is only free while + the adapter's module body stays stdlib-only and its clients stay lazy; the + day someone hoists a ``boto3.client(...)`` to module scope, every Lambda + image carrying any part of this package pays for it. + """ + loaded = self._import_and_report(module) + assert loaded == [], ( + f"importing {module} loaded {loaded}; keep these imports " + "function-local" + ) + + +class TestFacadeDependencyDirectionIsOneWay: + """rag_service depends on kb_backend, never the reverse.""" + + def test_facade_imports_the_seam(self): + """A guard against the facade quietly regrowing its own retrieval path. + + If ``rag_service`` stopped importing the resolver, it would mean the + delegation had been inlined again and the managed backend would be + unreachable — with every legacy test still green. + """ + rag_service = _BACKEND_SRC / "apis" / "shared" / "assistants" / "rag_service.py" + modules = {module for module, _ in _extract_imports(rag_service)} + assert "apis.shared.kb_backend.resolver" in modules, ( + "rag_service must resolve its backend through " + "apis.shared.kb_backend.resolver" + ) + assert "apis.shared.kb_backend.protocol" in modules, ( + "rag_service must use the protocol's canonical chunk shape" + ) diff --git a/backend/tests/lambdas/test_kb_ingestion_consumer.py b/backend/tests/lambdas/test_kb_ingestion_consumer.py new file mode 100644 index 000000000..70d20246b --- /dev/null +++ b/backend/tests/lambdas/test_kb_ingestion_consumer.py @@ -0,0 +1,332 @@ +"""Routing exclusivity for the managed-KB ingestion consumer. + +Feature: managed-kb-migration, task 9.2. + +The failure this file exists to prevent is **double indexing**. The legacy pipeline +is driven by its own pre-existing S3 notification on the same bucket, so for a legacy +document the correct behaviour of this consumer is to do nothing whatsoever. If it +ingested as well, the same bytes would be embedded twice: two sets of vectors, +doubled ingestion cost, and duplicate chunks competing inside one result list. None +of that raises an error, which is exactly why it needs a test. + +The routing is therefore deliberately asymmetric, and both halves are asserted: +legacy must ingest NOTHING here, managed must ingest here and NOT fall back. +""" + +from unittest.mock import MagicMock, patch + +import boto3 +import pytest +from moto import mock_aws + +from apis.app_api.kb_migration import ingestion_consumer as ic + +REGION = "us-east-1" +TABLE = "test-ingestion-consumer" +ASSISTANT_ID = "ast-ing01" +DOCUMENT_ID = "doc-ing01" +BUCKET = "docs-bucket" +KEY = f"assistants/{ASSISTANT_ID}/documents/{DOCUMENT_ID}/report.pdf" + + +@pytest.fixture() +def table(monkeypatch): + monkeypatch.setenv("AWS_DEFAULT_REGION", REGION) + monkeypatch.setenv("AWS_ACCESS_KEY_ID", "testing") + monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "testing") + monkeypatch.setenv("AWS_SESSION_TOKEN", "testing") + monkeypatch.setenv("DYNAMODB_ASSISTANTS_TABLE_NAME", TABLE) + + with mock_aws(): + ddb = boto3.client("dynamodb", region_name=REGION) + ddb.create_table( + TableName=TABLE, + KeySchema=[ + {"AttributeName": "PK", "KeyType": "HASH"}, + {"AttributeName": "SK", "KeyType": "RANGE"}, + ], + AttributeDefinitions=[ + {"AttributeName": "PK", "AttributeType": "S"}, + {"AttributeName": "SK", "AttributeType": "S"}, + ], + BillingMode="PAY_PER_REQUEST", + ) + t = boto3.resource("dynamodb", region_name=REGION).Table(TABLE) + t.put_item( + Item={ + "PK": f"AST#{ASSISTANT_ID}", + "SK": f"DOC#{DOCUMENT_ID}", + "status": "uploading", + } + ) + yield t + + +def _seed_kb(table, **overrides): + """A KB_Record for this assistant. No retrievalEngine unless asked.""" + item = {"PK": f"AST#{ASSISTANT_ID}", "SK": f"KB#{ASSISTANT_ID}", "appKbId": ASSISTANT_ID} + item.update(overrides) + table.put_item(Item=item) + + +def _doc(table): + return table.get_item( + Key={"PK": f"AST#{ASSISTANT_ID}", "SK": f"DOC#{DOCUMENT_ID}"} + )["Item"] + + +def _eventbridge_event(key=KEY): + return {"detail": {"bucket": {"name": BUCKET}, "object": {"key": key}}} + + +class _FakeBackend: + """Records ingest calls; reports the document retrievable immediately.""" + + def __init__(self): + self.ingested = [] + + async def ingest(self, kb_ref, source): + self.ingested.append(source.document_id) + return None + + async def search(self, kb_ref, query, top_k=5): + chunk = MagicMock() + chunk.metadata = {"document_id": DOCUMENT_ID} + return [chunk] + + +# --------------------------------------------------------------------------- +# Legacy must not be touched +# --------------------------------------------------------------------------- +class TestLegacyRouting: + def test_a_legacy_document_is_not_ingested_here(self, table): + """No retrievalEngine means legacy, and legacy is somebody else's job.""" + _seed_kb(table) + fake = _FakeBackend() + + with patch("apis.shared.kb_backend.managed_backend.ManagedKbBackend", return_value=fake): + result = ic.handle_object(BUCKET, KEY) + + assert result["routed"] == "legacy" + assert result["ingested"] is False + assert fake.ingested == [], "a legacy document was ingested into the managed backend" + + def test_a_document_with_no_kb_record_at_all_is_legacy(self, table): + """The overwhelmingly common case today: no record has ever been written.""" + result = ic.handle_object(BUCKET, KEY) + assert result["routed"] == "legacy" + assert result["ingested"] is False + + def test_a_legacy_document_status_is_left_alone(self, table): + """The legacy pipeline owns the terminal transition for its documents. + + Writing `complete` here would race the other Lambda and could mark a + document ready before its vectors exist. + """ + _seed_kb(table) + ic.handle_object(BUCKET, KEY) + assert _doc(table)["status"] == "uploading" + + @pytest.mark.parametrize("engine", ["s3vectors", "S3Vectors", "MANAGED", "managed ", "", "wat"]) + def test_only_the_exact_managed_literal_routes_to_managed(self, table, engine): + """Exact-match, so a casing slip fails safe. + + Failing safe matters asymmetrically: routing to legacy when it should be + managed leaves the existing pipeline handling it correctly, while routing to + managed when the record is not really migrated ingests into a knowledge base + that may not exist. + """ + _seed_kb(table, retrievalEngine=engine) + fake = _FakeBackend() + with patch("apis.shared.kb_backend.managed_backend.ManagedKbBackend", return_value=fake): + result = ic.handle_object(BUCKET, KEY) + assert result["routed"] == "legacy" + assert fake.ingested == [] + + +# --------------------------------------------------------------------------- +# Managed must be ingested here, exactly once +# --------------------------------------------------------------------------- +class TestManagedRouting: + def _seed_managed(self, table): + _seed_kb( + table, + retrievalEngine="managed", + awsKbId="KB123", + awsDataSourceId="DS456", + ) + + def test_a_managed_document_is_ingested_directly(self, table): + self._seed_managed(table) + fake = _FakeBackend() + + with patch( + "apis.shared.kb_backend.managed_backend.ManagedKbBackend", return_value=fake + ): + result = ic.handle_object(BUCKET, KEY) + + assert result["routed"] == "managed" + assert result["ingested"] is True + assert fake.ingested == [DOCUMENT_ID] + + def test_a_managed_document_is_ingested_exactly_once(self, table): + """One invocation, one ingest. Duplicate chunks would compete in retrieval.""" + self._seed_managed(table) + fake = _FakeBackend() + + with patch( + "apis.shared.kb_backend.managed_backend.ManagedKbBackend", return_value=fake + ): + ic.lambda_handler(_eventbridge_event(), None) + + assert fake.ingested == [DOCUMENT_ID] + + def test_the_document_reaches_complete(self, table): + self._seed_managed(table) + with patch( + "apis.shared.kb_backend.managed_backend.ManagedKbBackend", + return_value=_FakeBackend(), + ): + ic.handle_object(BUCKET, KEY) + + assert _doc(table)["status"] == "complete" + + def test_indexed_and_retrievable_are_recorded_separately(self, table): + """Two timestamps, not one. + + Bedrock reports INDEXED up to a second before a document can actually be + retrieved (measured 0.75-1.03 s). Collapsing them would erase the only + evidence of that gap, which is what makes "my upload finished but the + assistant cannot see it" diagnosable. + """ + self._seed_managed(table) + with patch( + "apis.shared.kb_backend.managed_backend.ManagedKbBackend", + return_value=_FakeBackend(), + ): + result = ic.handle_object(BUCKET, KEY) + + item = _doc(table) + assert "indexedAt" in item + assert "retrievableAt" in item + assert result["indexedAt"] and result["retrievableAt"] + + def test_a_managed_document_never_falls_back_to_legacy(self, table): + """Managed engine but unprovisioned must FAIL, not silently degrade. + + A quiet fallback would hand the document to the legacy pipeline as well, + producing the dual index this whole file guards against. + """ + _seed_kb(table, retrievalEngine="managed") # no awsKbId / awsDataSourceId + + with pytest.raises(ic.IngestionRoutingError, match="not provisioned"): + ic.handle_object(BUCKET, KEY) + + def test_a_failed_ingestion_marks_the_document_failed_and_raises(self, table): + """The record is the retry anchor, so a failure must be visible in both + places: on the document and to the event source.""" + self._seed_managed(table) + + class _Failing(_FakeBackend): + async def ingest(self, kb_ref, source): + raise RuntimeError("bedrock unavailable") + + with patch( + "apis.shared.kb_backend.managed_backend.ManagedKbBackend", return_value=_Failing() + ): + with pytest.raises(RuntimeError): + ic.handle_object(BUCKET, KEY) + + item = _doc(table) + assert item["status"] == "failed" + assert "bedrock unavailable" in item["ingestionError"] + + def test_a_document_that_never_becomes_retrievable_is_not_marked_complete(self, table): + """Indexed is not retrievable. Claiming success here is the bug.""" + self._seed_managed(table) + + class _NeverRetrievable(_FakeBackend): + async def search(self, kb_ref, query, top_k=5): + return [] + + # Shrink the poll window: the real 30s default is correct in production + # (the observed gap is ~1s and waiting is cheap) but would add 30s to every + # run of this suite. + with patch( + "apis.shared.kb_backend.managed_backend.ManagedKbBackend", + return_value=_NeverRetrievable(), + ), patch.object(ic, "RETRIEVABLE_POLL_TIMEOUT_SECONDS", 0.05), patch.object( + ic, "RETRIEVABLE_POLL_INTERVAL_SECONDS", 0.01 + ): + with pytest.raises(ic.IngestionRoutingError, match="not retrievable"): + ic.handle_object(BUCKET, KEY) + + assert _doc(table)["status"] != "complete" + + +# --------------------------------------------------------------------------- +# Event parsing +# --------------------------------------------------------------------------- +class TestEventParsing: + def test_eventbridge_shape_is_understood(self): + records = ic.extract_records(_eventbridge_event()) + assert records == [{"bucket": BUCKET, "key": KEY}] + + def test_raw_s3_notification_shape_is_understood(self): + """Both shapes are accepted so a wiring change cannot silently stop + ingestion — the bucket carries two producers.""" + event = {"Records": [{"s3": {"bucket": {"name": BUCKET}, "object": {"key": KEY}}}]} + assert ic.extract_records(event) == [{"bucket": BUCKET, "key": KEY}] + + def test_an_empty_event_is_a_no_op(self): + assert ic.lambda_handler({}, None)["processed"] == 0 + + def test_a_url_encoded_key_is_decoded(self): + a, d, f = ic.parse_object_key( + "assistants/ast-1/documents/doc-2/my+report+%282024%29.pdf" + ) + assert (a, d) == ("ast-1", "doc-2") + assert f == "my report (2024).pdf" + + def test_a_filename_containing_slashes_is_preserved(self): + _, _, f = ic.parse_object_key("assistants/a/documents/d/sub/dir/file.pdf") + assert f == "sub/dir/file.pdf" + + @pytest.mark.parametrize( + "key", + [ + "wrong/ast-1/documents/doc-2/f.pdf", + "assistants/ast-1/wrong/doc-2/f.pdf", + "assistants/ast-1/documents/doc-2", + "", + ], + ) + def test_a_malformed_key_is_refused(self, key): + """Guessing at a malformed key could ingest one assistant's document into + another's knowledge base.""" + with pytest.raises(ic.IngestionRoutingError): + ic.parse_object_key(key) + + +# --------------------------------------------------------------------------- +# Structural guarantees +# --------------------------------------------------------------------------- +class TestNoInProcessOrchestration: + def test_the_module_does_not_use_ensure_future(self): + """Requirement 10.8. A background task is killed when the Lambda handler + returns, converting a reported success into a half-finished ingestion.""" + import ast + import inspect + + # Parsed, not grepped. A substring check trips on this module's own + # docstring, which explains at length WHY it does not orchestrate in + # process — the first version of this test failed on the prose describing + # the very thing it was verifying the absence of. + tree = ast.parse(inspect.getsource(ic)) + called = { + node.func.attr + for node in ast.walk(tree) + if isinstance(node, ast.Call) and isinstance(node.func, ast.Attribute) + } + assert "ensure_future" not in called + assert "create_task" not in called diff --git a/backend/tests/lambdas/test_kb_migration_worker.py b/backend/tests/lambdas/test_kb_migration_worker.py new file mode 100644 index 000000000..879651472 --- /dev/null +++ b/backend/tests/lambdas/test_kb_migration_worker.py @@ -0,0 +1,1022 @@ +""" +Migration dispatcher and worker: bounded, leased, and safe to interrupt. + +Requirements 15, 16, 19.6. The tests here concentrate on the things that are +invisible when they break: + +* The dispatcher **no-ops when the flag is off**, and "off" includes present but + empty. This is the reconciler-arming defect's shape, and it is worth re-testing + per component because each one reads its own flag. +* The worker dispatches on the **record's** state, never the event's. An event + field that could select `promote` would let a hand-crafted invocation cut a + knowledge base over without it ever verifying. +* A document deleted mid-migration is **not resurrected** — asserted by deleting it + between the snapshot and the ingest, which is the only window where the bug + exists. +* Catch-up **converges on quiet**, not after a fixed number of passes. +* Concurrent promotion yields **one winner**, which is a property of the + conditional write rather than of any locking here. + +Feature: managed-kb-migration +Requirements: 15.4, 15.5, 15.6, 15.7, 15.8, 15.10, 15.13, 15.14, 16.2, 16.3, +16.4, 16.5, 17.1, 17.4, 19.6, 24.5 +""" + +from decimal import Decimal +from typing import Any, Dict, List +from unittest.mock import MagicMock, patch + +import pytest + +from apis.app_api.kb_migration import dispatcher, worker +from apis.shared.kb_backend import records as r +from apis.shared.kb_backend.protocol import Chunk + +ASSISTANT_ID = "ast-migrate-001" +TABLE = "test-assistants" +BUCKET = "test-documents" + +BASE_ENV = { + "DYNAMODB_ASSISTANTS_TABLE_NAME": TABLE, + "S3_ASSISTANTS_DOCUMENTS_BUCKET_NAME": BUCKET, + "AWS_REGION": "us-west-2", +} + + +def _doc(document_id: str, status: str = "complete", size: int = 1024) -> Dict[str, Any]: + return { + "PK": f"AST#{ASSISTANT_ID}", + "SK": f"DOC#{document_id}", + "status": status, + "filename": f"{document_id}.pdf", + "s3Key": f"assistants/{ASSISTANT_ID}/documents/{document_id}/{document_id}.pdf", + "contentHash": f"hash-{document_id}", + "sizeBytes": Decimal(size), + } + + +def _kb_record(state: str = r.SHADOW, **overrides) -> Dict[str, Any]: + record = { + "PK": f"AST#{ASSISTANT_ID}", + "SK": f"KB#{ASSISTANT_ID}", + "appKbId": ASSISTANT_ID, + "ownerUserId": "user-migrate", + "migrationState": state, + "migrationGeneration": Decimal(1), + "totalBytes": Decimal(4096), + "awsKbId": "KB123", + "awsDataSourceId": "DS123", + } + record.update(overrides) + return record + + +async def _async_noop(*args, **kwargs): + """An awaitable that does nothing. + + Used as ``side_effect`` rather than assigning a coroutine to ``return_value``: + a coroutine object assigned that way is created once, so a mock called twice + raises and a mock called never emits "coroutine was never awaited" — noise that + makes a real leak invisible. + """ + return None + + +class StubBackend: + """Records what it was asked to ingest, delete and search.""" + + def __init__(self, chunks: List[Chunk] = None): + self.ingested: List[str] = [] + self.searched: List[str] = [] + self._chunks = chunks if chunks is not None else [ + Chunk(text="hit", relevance=1.0, document_id="d1", metadata={"document_id": "d1"}) + ] + + async def ingest_documents(self, kb_ref, sources, *, batch_size=10): + self.ingested.extend(source.document_id for source in sources) + + async def search(self, kb_ref, query, top_k=5): + self.searched.append(query) + return list(self._chunks) + + async def delete_documents(self, kb_ref, document_ids, *, batch_size=10): # pragma: no cover + raise NotImplementedError + + +# ── Dispatcher ─────────────────────────────────────────────────────────────── +class TestDispatcherFlag: + @pytest.mark.parametrize("value", [None, "", " ", "false", "0", "off", "no", "disabled"]) + def test_anything_but_a_truthy_spelling_is_off(self, value): + """An allow-list, not a truthiness test. The failure being designed around + is a value that is *present but empty*: ``bool("")`` is correct by luck, + ``bool("false")`` is not.""" + env = {} if value is None else {dispatcher.FLAG_MIGRATION_ENABLED: value} + with patch.dict("os.environ", env, clear=True): + assert dispatcher.migration_enabled() is False + + @pytest.mark.parametrize("value", ["1", "true", "TRUE", "yes", "on", "enabled"]) + def test_affirmative_spellings_are_on(self, value): + with patch.dict( + "os.environ", {dispatcher.FLAG_MIGRATION_ENABLED: value}, clear=True + ): + assert dispatcher.migration_enabled() is True + + @pytest.mark.asyncio + async def test_a_tick_with_the_flag_off_invokes_nothing(self): + with patch.dict("os.environ", {}, clear=True), patch.object( + dispatcher, "_invoke_worker" + ) as invoke, patch.object(dispatcher, "_due_records") as due: + counts = await dispatcher.dispatch_once() + + invoke.assert_not_called() + due.assert_not_called() + assert counts == {"Due": 0, "Dispatched": 0, "Failed": 0} + + +class TestDispatcherLimit: + def test_the_default_matches_the_sync_dispatcher(self): + with patch.dict("os.environ", {}, clear=True): + assert dispatcher.dispatch_limit() == 20 + + def test_an_override_is_honoured(self): + with patch.dict("os.environ", {"KB_MIGRATION_DISPATCH_LIMIT": "5"}, clear=True): + assert dispatcher.dispatch_limit() == 5 + + def test_an_override_above_the_ceiling_is_clamped(self): + """A larger sweep should require repeated observed ticks, not a variable + edit — and `StartIngestionJob` is 0.1 RPS account-wide and not + adjustable, so the only way to stay under it is to not ask.""" + with patch.dict("os.environ", {"KB_MIGRATION_DISPATCH_LIMIT": "5000"}, clear=True): + assert dispatcher.dispatch_limit() == dispatcher.DISPATCH_LIMIT_CEILING + + def test_a_nonsense_override_falls_back_to_the_default(self): + with patch.dict("os.environ", {"KB_MIGRATION_DISPATCH_LIMIT": "lots"}, clear=True): + assert dispatcher.dispatch_limit() == 20 + + @pytest.mark.asyncio + async def test_the_limit_bounds_the_tick_across_all_states_not_per_state(self): + """Three states each honouring the limit would quietly be a 3x limit. + + Asserted on what each **query asked for**, not on the tick's total: the + total is trimmed at the end, so a per-state sweep that read three times the + budget from DynamoDB would still *return* the right number while paying for + three times the reads. + + The first state deliberately returns fewer rows than the limit. That is the + only shape where the bug is observable — if the first query fills the + budget the loop exits either way, which is why the obvious version of this + test passes with the arithmetic removed. + """ + asked: List[int] = [] + rows = [_kb_record(r.SHADOW, appKbId=f"kb-{i}") for i in range(10)] + + def _query(state, now_iso, limit): + asked.append(limit) + # promote yields 2 of the 4 allowed; the rest could fill the tick. + available = 2 if state == r.PROMOTE else 10 + return rows[: min(limit, available)] + + with patch.dict( + "os.environ", + {**BASE_ENV, dispatcher.FLAG_MIGRATION_ENABLED: "true", "KB_MIGRATION_DISPATCH_LIMIT": "4"}, + clear=True, + ), patch("apis.shared.kb_backend.records.query_due_work", side_effect=_query), patch.object( + dispatcher, "_invoke_worker" + ) as invoke, patch.object(dispatcher, "_emit_metrics"): + counts = await dispatcher.dispatch_once() + + assert counts["Due"] == 4 + assert invoke.call_count == 4 + assert asked[0] == 4 + assert asked[1] == 2, ( + f"the second state was asked for {asked[1]} records when only " + f"{4 - 2} of the budget remained; each state is being given the whole " + f"limit ({asked})" + ) + + +class TestDispatcherSweep: + def test_every_work_eligible_state_is_swept(self): + """Derived from ``WORK_ELIGIBLE_STATES``, so a state added there cannot be + silently left unswept — it would stall forever with its work keys written + and nothing reading them.""" + assert set(dispatcher._work_states()) == set(r.WORK_ELIGIBLE_STATES) + + def test_a_state_added_to_the_records_module_is_still_swept(self): + """The assertion above passes today whether or not the derivation exists, + because the priority list happens to name every state. So add one the + dispatcher has never heard of and require it to be swept anyway — which is + the whole point of deriving rather than restating. + """ + extended = frozenset(set(r.WORK_ELIGIBLE_STATES) | {"reindex"}) + with patch.object(r, "WORK_ELIGIBLE_STATES", extended): + states = dispatcher._work_states() + + assert "reindex" in states, ( + "a new work-eligible state is not swept; its records would keep their " + "GSI7 work keys and never be handed to a worker" + ) + # Appended, not promoted ahead of the known order. + assert states[-1] == "reindex" + + def test_promote_is_swept_first(self): + """A record in ``promote`` is one conditional write from finished, so + draining beats starting new shadow work.""" + assert dispatcher._work_states()[0] == r.PROMOTE + + def test_no_terminal_state_is_swept(self): + assert not set(dispatcher._work_states()) & set(r.TERMINAL_STATES) + + @pytest.mark.asyncio + async def test_an_unaddressable_row_does_not_starve_the_sweep(self): + good = _kb_record(r.SHADOW, appKbId="kb-good") + bad = {"SK": "KB#kb-bad", "migrationState": r.SHADOW} # no PK + + calls = {"n": 0} + + def _query(state, now_iso, limit): + calls["n"] += 1 + return [bad, good] if calls["n"] == 1 else [] + + with patch.dict( + "os.environ", + {**BASE_ENV, dispatcher.FLAG_MIGRATION_ENABLED: "true"}, + clear=True, + ), patch("apis.shared.kb_backend.records.query_due_work", side_effect=_query), patch.object( + dispatcher, "_invoke_worker" + ) as invoke, patch.object(dispatcher, "_emit_metrics"): + counts = await dispatcher.dispatch_once() + + assert counts["Failed"] == 1 + assert counts["Dispatched"] == 1 + assert invoke.call_args.args[0]["appKbId"] == "kb-good" + + @pytest.mark.asyncio + async def test_a_failing_index_query_does_not_fail_the_tick(self): + with patch.dict( + "os.environ", + {**BASE_ENV, dispatcher.FLAG_MIGRATION_ENABLED: "true"}, + clear=True, + ), patch( + "apis.shared.kb_backend.records.query_due_work", + side_effect=RuntimeError("dynamodb down"), + ), patch.object(dispatcher, "_emit_metrics"): + counts = await dispatcher.dispatch_once() + + assert counts == {"Due": 0, "Dispatched": 0, "Failed": 0} + + def test_the_handler_reads_nothing_from_the_event(self): + """The reconciler's arming bypass came from forwarding an event field. + Nothing here may select a state, a limit or a knowledge base.""" + seen = {} + + async def _tick(): + seen["called"] = True + return {"Due": 0, "Dispatched": 0, "Failed": 0} + + with patch.object(dispatcher, "dispatch_once", side_effect=_tick) as tick: + dispatcher.lambda_handler({"migrationState": "promote", "armed": True}, None) + + assert seen.get("called") is True + tick.assert_called_once_with() + + +# ── Worker: state selection ────────────────────────────────────────────────── +class TestTheRecordDecidesTheStep: + @pytest.mark.asyncio + async def test_an_event_cannot_select_promote(self): + """A hand-crafted invocation must not be able to cut over a knowledge base + that never verified.""" + record = _kb_record(r.SHADOW) + + with patch.dict("os.environ", BASE_ENV, clear=True), patch( + "apis.shared.kb_backend.records.get_kb_record", return_value=record + ), patch.object(worker, "take_lease", return_value="later"), patch.object( + worker, "run_shadow" + ) as shadow, patch.object(worker, "run_promote") as promote: + shadow.return_value = worker.StepResult(ASSISTANT_ID, ASSISTANT_ID, r.SHADOW, r.VERIFY) + await worker.run_step(ASSISTANT_ID, ASSISTANT_ID) + + shadow.assert_called_once() + promote.assert_not_called() + + @pytest.mark.asyncio + async def test_a_terminal_record_is_a_no_op(self): + """The index is eventually consistent, so a record that finished a moment + ago can still be handed over once. That is not an error.""" + record = _kb_record(r.RETAIN) + + with patch.dict("os.environ", BASE_ENV, clear=True), patch( + "apis.shared.kb_backend.records.get_kb_record", return_value=record + ), patch.object(worker, "take_lease") as lease: + result = await worker.run_step(ASSISTANT_ID) + + lease.assert_not_called() + assert result.to_state == r.RETAIN + assert "not work-eligible" in result.detail + + @pytest.mark.asyncio + async def test_a_lost_lease_propagates_rather_than_failing_the_migration(self): + """Requirement 15.13. Two overlapping ticks is ordinary; marking the + migration `failed` because of it would strand a healthy knowledge base.""" + record = _kb_record(r.SHADOW) + + with patch.dict("os.environ", BASE_ENV, clear=True), patch( + "apis.shared.kb_backend.records.get_kb_record", return_value=record + ), patch( + "apis.shared.kb_backend.records.acquire_lease", + side_effect=RuntimeError("conditional check failed"), + ), patch( + "apis.shared.kb_backend.metrics.emit_count" + ), patch.object(worker, "_fail") as fail: + with pytest.raises(worker.LeaseLost): + await worker.run_step(ASSISTANT_ID) + + fail.assert_not_called() + + +# ── Worker: shadow and catch-up ────────────────────────────────────────────── +class TestShadowAndCatchUp: + @pytest.mark.asyncio + async def test_documents_are_ingested_from_their_existing_s3_keys(self): + """Requirement 15.4: a re-ingest, never a re-upload.""" + docs = [_doc("d1"), _doc("d2")] + backend = StubBackend() + captured = {} + + async def _capture(kb_ref, sources, *, batch_size=10): + captured["sources"] = list(sources) + backend.ingested.extend(s.document_id for s in sources) + + backend.ingest_documents = _capture + + with patch.dict("os.environ", BASE_ENV, clear=True), patch.object( + worker, "list_document_items", return_value=docs + ), patch.object( + worker, "get_document_item", side_effect=lambda a, d: _doc(d) + ), patch( + "apis.shared.kb_backend.byte_cap.reserve_snapshot" + ), patch( + "apis.shared.kb_backend.provisioning.provision_managed_kb" + ) as provision, patch( + "apis.shared.kb_backend.metrics.emit_count" + ), patch.object( + worker, "_record_progress" + ), patch( + "apis.shared.kb_backend.records.set_migration_state" + ): + provision.side_effect = _async_noop + result = await worker.run_shadow( + ASSISTANT_ID, ASSISTANT_ID, _kb_record(r.SHADOW), backend + ) + + assert result.to_state == r.VERIFY + assert sorted(backend.ingested) == ["d1", "d2"] + keys = {s.s3_key for s in captured["sources"]} + assert keys == { + f"assistants/{ASSISTANT_ID}/documents/d1/d1.pdf", + f"assistants/{ASSISTANT_ID}/documents/d2/d2.pdf", + } + + @pytest.mark.asyncio + async def test_only_complete_documents_are_migrated(self): + """Requirement 15.5. A non-complete document is not retrievable on legacy + either, so migrating it would create a difference where the point is + parity.""" + docs = [_doc("d1"), _doc("d2", status="failed"), _doc("d3", status="uploading")] + backend = StubBackend() + + with patch.dict("os.environ", BASE_ENV, clear=True), patch.object( + worker, "list_document_items", return_value=docs + ), patch.object( + worker, "get_document_item", side_effect=lambda a, d: next( + (x for x in docs if worker.document_id_of(x) == d), None + ) + ), patch( + "apis.shared.kb_backend.byte_cap.reserve_snapshot" + ), patch( + "apis.shared.kb_backend.provisioning.provision_managed_kb" + ) as provision, patch( + "apis.shared.kb_backend.metrics.emit_count" + ), patch.object( + worker, "_record_progress" + ), patch( + "apis.shared.kb_backend.records.set_migration_state" + ): + provision.side_effect = _async_noop + await worker.run_shadow(ASSISTANT_ID, ASSISTANT_ID, _kb_record(), backend) + + assert backend.ingested == ["d1"] + + @pytest.mark.asyncio + async def test_a_document_deleted_mid_migration_is_not_resurrected(self): + """Requirements 16.4, 16.5, and the reason the re-read is per document + rather than per batch: a PDF batch takes minutes, and the deletion this + guards against is most likely to land inside exactly that window. + + ``d2`` is in the snapshot but gone by the time its turn comes. + """ + docs = [_doc("d1"), _doc("d2")] + deleted = {"d2"} + backend = StubBackend() + + def _get(assistant_id, document_id): + return None if document_id in deleted else _doc(document_id) + + with patch.dict("os.environ", BASE_ENV, clear=True), patch.object( + worker, "list_document_items", return_value=docs + ), patch.object(worker, "get_document_item", side_effect=_get), patch( + "apis.shared.kb_backend.byte_cap.reserve_snapshot" + ), patch( + "apis.shared.kb_backend.provisioning.provision_managed_kb" + ) as provision, patch( + "apis.shared.kb_backend.metrics.emit_count" + ), patch.object( + worker, "_record_progress" + ), patch( + "apis.shared.kb_backend.records.set_migration_state" + ): + provision.side_effect = _async_noop + result = await worker.run_shadow(ASSISTANT_ID, ASSISTANT_ID, _kb_record(), backend) + + assert backend.ingested == ["d1"], "a deleted document was resurrected" + assert result.documents_skipped >= 1 + + @pytest.mark.asyncio + async def test_a_document_that_stopped_being_complete_is_skipped(self): + docs = [_doc("d1")] + backend = StubBackend() + + with patch.dict("os.environ", BASE_ENV, clear=True), patch.object( + worker, "list_document_items", return_value=docs + ), patch.object( + worker, "get_document_item", return_value=_doc("d1", status="deleting") + ), patch( + "apis.shared.kb_backend.byte_cap.reserve_snapshot" + ), patch( + "apis.shared.kb_backend.provisioning.provision_managed_kb" + ) as provision, patch( + "apis.shared.kb_backend.metrics.emit_count" + ), patch.object( + worker, "_record_progress" + ), patch( + "apis.shared.kb_backend.records.set_migration_state" + ): + provision.side_effect = _async_noop + await worker.run_shadow(ASSISTANT_ID, ASSISTANT_ID, _kb_record(), backend) + + assert backend.ingested == [] + + @pytest.mark.asyncio + async def test_the_whole_snapshot_is_reserved_before_anything_is_provisioned(self): + """Requirement 12.9. Reserving per document would let a migration run for + an hour and stop halfway, leaving a half-populated corpus and an owner over + their cap with no way back.""" + order: List[str] = [] + docs = [_doc("d1", size=2048), _doc("d2", size=4096)] + + def _reserve(assistant_id, app_kb_id, total, cap): + order.append(f"reserve:{total}") + + async def _provision(*args, **kwargs): + order.append("provision") + + with patch.dict("os.environ", BASE_ENV, clear=True), patch.object( + worker, "list_document_items", return_value=docs + ), patch.object( + worker, "get_document_item", side_effect=lambda a, d: _doc(d) + ), patch( + "apis.shared.kb_backend.byte_cap.reserve_snapshot", side_effect=_reserve + ), patch( + "apis.shared.kb_backend.provisioning.provision_managed_kb", side_effect=_provision + ), patch( + "apis.shared.kb_backend.metrics.emit_count" + ), patch.object( + worker, "_record_progress" + ), patch( + "apis.shared.kb_backend.records.set_migration_state" + ): + fresh = _kb_record() + fresh.pop("totalBytes") + await worker.run_shadow(ASSISTANT_ID, ASSISTANT_ID, fresh, StubBackend()) + + assert order == ["reserve:6144", "provision"] + + @pytest.mark.asyncio + async def test_an_over_cap_corpus_fails_before_provisioning(self): + from apis.shared.kb_backend.byte_cap import ByteCapExceeded + + async def _provision(*args, **kwargs): # pragma: no cover - must not run + raise AssertionError("provisioned despite the byte cap") + + with patch.dict("os.environ", BASE_ENV, clear=True), patch( + "apis.shared.kb_backend.records.get_kb_record", return_value=_kb_record() + ), patch.object(worker, "take_lease", return_value="later"), patch.object( + worker, "list_document_items", return_value=[_doc("d1")] + ), patch( + "apis.shared.kb_backend.byte_cap.reserve_snapshot", + side_effect=ByteCapExceeded(requested=1, cap=0), + ), patch( + "apis.shared.kb_backend.provisioning.provision_managed_kb", side_effect=_provision + ), patch( + "apis.shared.kb_backend.metrics.emit_count" + ), patch.object( + worker, "_fail" + ) as fail: + result = await worker.run_step(ASSISTANT_ID) + + assert result.to_state == r.MIGRATION_FAILED + fail.assert_called_once() + + @pytest.mark.asyncio + async def test_catch_up_converges_on_quiet_not_on_a_pass_count(self): + """Requirement 16.3. A new document appears during the first pass; the + second finds nothing and that is what ends it.""" + backend = StubBackend() + state = {"pass": 0} + + def _list(assistant_id): + state["pass"] += 1 + if state["pass"] == 1: + return [_doc("d1"), _doc("d2")] + return [_doc("d1"), _doc("d2")] + + with patch.dict("os.environ", BASE_ENV, clear=True), patch.object( + worker, "list_document_items", side_effect=_list + ), patch.object(worker, "get_document_item", side_effect=lambda a, d: _doc(d)): + passes, converged, counts = await worker.catch_up( + ASSISTANT_ID, ASSISTANT_ID, {"d1"}, backend + ) + + assert converged is True + assert passes == 2 + assert backend.ingested == ["d2"] + + @pytest.mark.asyncio + async def test_a_corpus_that_never_settles_does_not_converge(self): + """And staying in ``shadow`` is the correct outcome: the corpus keeps + serving from legacy while the owner keeps uploading.""" + backend = StubBackend() + counter = {"n": 0} + + def _list(assistant_id): + counter["n"] += 1 + return [_doc(f"d{i}") for i in range(counter["n"] + 1)] + + with patch.dict("os.environ", BASE_ENV, clear=True), patch.object( + worker, "list_document_items", side_effect=_list + ), patch.object(worker, "get_document_item", side_effect=lambda a, d: _doc(d)): + passes, converged, _ = await worker.catch_up( + ASSISTANT_ID, ASSISTANT_ID, set(), backend, max_passes=3 + ) + + assert converged is False + assert passes == 3 + + @pytest.mark.asyncio + async def test_an_unconverged_shadow_stays_in_shadow(self): + docs = [_doc("d1")] + transitions: List[str] = [] + + def _set_state(assistant_id, app_kb_id, new_state, generation, due=None, expected=None, error=None): + transitions.append(new_state) + + with patch.dict("os.environ", BASE_ENV, clear=True), patch.object( + worker, "list_document_items", return_value=docs + ), patch.object( + worker, "get_document_item", side_effect=lambda a, d: _doc(d) + ), patch( + "apis.shared.kb_backend.byte_cap.reserve_snapshot" + ), patch( + "apis.shared.kb_backend.provisioning.provision_managed_kb" + ) as provision, patch( + "apis.shared.kb_backend.metrics.emit_count" + ), patch.object( + worker, "_record_progress" + ), patch( + "apis.shared.kb_backend.records.set_migration_state", side_effect=_set_state + ), patch.object( + worker, "catch_up", return_value=(5, False, {"migrated": 0, "skipped": 0, "done": []}) + ): + provision.side_effect = _async_noop + result = await worker.run_shadow(ASSISTANT_ID, ASSISTANT_ID, _kb_record(), StubBackend()) + + assert transitions == [r.SHADOW] + assert result.to_state == r.SHADOW + assert result.converged is False + + +# ── Worker: verify ─────────────────────────────────────────────────────────── +class TestVerify: + def test_the_manifest_is_content_identity_not_a_count(self): + """Requirement 15.6. Count parity is satisfied by a corpus with the right + *number* of wrong documents — exactly what a migration that raced an upload + and a delete produces.""" + before = worker.source_manifest([_doc("d1"), _doc("d2")]) + changed = dict(_doc("d2")) + changed["contentHash"] = "hash-d2-edited" + after = worker.source_manifest([_doc("d1"), changed]) + + assert len(before) == len(after) + assert before != after, "the manifest is count-equivalent and cannot see an edit" + + def test_a_document_with_no_hash_still_contributes_a_changing_value(self): + item = {"SK": "DOC#d9", "status": "complete", "updatedAt": "2026-08-01T00:00:00Z"} + assert worker.manifest_entry(item) == "d9:2026-08-01T00:00:00Z" + + @pytest.mark.asyncio + async def test_verify_requires_a_canary_retrieval_to_return_something(self): + """Requirement 15.7. Bedrock reporting a document INDEXED precedes it being + retrievable by 0.75-1.03 s, and a knowledge base can hold documents while + returning nothing, so "we ingested everything" and "retrieval works" are + separate claims.""" + backend = StubBackend(chunks=[]) + + with patch.dict("os.environ", BASE_ENV, clear=True), patch.object( + worker, "list_document_items", return_value=[_doc("d1")] + ): + # Matched on "not queryable", not on "canary": both failure messages + # mention the canary, so the looser pattern passed even with the + # empty-result check removed — the *other* check raised and the test + # could not tell the difference. + with pytest.raises(worker.VerificationFailed, match="not queryable"): + await worker.run_verify(ASSISTANT_ID, ASSISTANT_ID, _kb_record(r.VERIFY), backend) + + @pytest.mark.asyncio + async def test_verify_rejects_a_canary_that_returns_foreign_documents(self): + backend = StubBackend( + chunks=[ + Chunk( + text="someone else's", + relevance=1.0, + document_id="not-ours", + metadata={"document_id": "not-ours"}, + ) + ] + ) + + with patch.dict("os.environ", BASE_ENV, clear=True), patch.object( + worker, "list_document_items", return_value=[_doc("d1")] + ): + with pytest.raises(worker.VerificationFailed): + await worker.run_verify(ASSISTANT_ID, ASSISTANT_ID, _kb_record(r.VERIFY), backend) + + @pytest.mark.asyncio + async def test_an_empty_corpus_cannot_be_verified(self): + with patch.dict("os.environ", BASE_ENV, clear=True), patch.object( + worker, "list_document_items", return_value=[_doc("d1", status="failed")] + ): + with pytest.raises(worker.VerificationFailed, match="nothing"): + await worker.run_verify( + ASSISTANT_ID, ASSISTANT_ID, _kb_record(r.VERIFY), StubBackend() + ) + + @pytest.mark.asyncio + async def test_a_successful_verify_moves_to_promote(self): + backend = StubBackend( + chunks=[ + Chunk(text="hit", relevance=1.0, document_id="d1", metadata={"document_id": "d1"}) + ] + ) + transitions: List[tuple] = [] + + def _set_state(assistant_id, app_kb_id, new_state, generation, due=None, expected=None, error=None): + transitions.append((new_state, expected)) + + with patch.dict("os.environ", BASE_ENV, clear=True), patch.object( + worker, "list_document_items", return_value=[_doc("d1")] + ), patch("apis.shared.kb_backend.records.set_migration_state", side_effect=_set_state): + result = await worker.run_verify( + ASSISTANT_ID, ASSISTANT_ID, _kb_record(r.VERIFY), backend + ) + + assert result.to_state == r.PROMOTE + assert transitions == [(r.PROMOTE, [r.VERIFY])] + + def test_the_canary_query_is_built_from_the_corpus(self): + """Not a fixed string: a constant like "test" can legitimately match + nothing in a real corpus, which would fail healthy knowledge bases and + train whoever is watching to ignore it.""" + query = worker._canary_query([{"filename": "student_handbook.pdf"}]) + assert "student" in query and "handbook" in query + assert ".pdf" not in query + + +# ── Worker: promote and rollback ───────────────────────────────────────────── +class TestPromote: + @pytest.mark.asyncio + async def test_promotion_is_refused_without_a_byte_cap_accumulator(self): + """Requirement 12.9: no traffic is promoted to an unmetered corpus.""" + record = _kb_record(r.PROMOTE) + record.pop("totalBytes") + + with patch.dict("os.environ", BASE_ENV, clear=True), patch( + "apis.shared.kb_backend.records.promote_engine" + ) as promote: + with pytest.raises(worker.MigrationError, match="totalBytes"): + await worker.run_promote(ASSISTANT_ID, ASSISTANT_ID, record) + + promote.assert_not_called() + + @pytest.mark.asyncio + async def test_promotion_writes_once_and_then_retains(self): + calls: List[str] = [] + + with patch.dict("os.environ", BASE_ENV, clear=True), patch( + "apis.shared.kb_backend.records.promote_engine", + side_effect=lambda *a: calls.append("promote"), + ), patch( + "apis.shared.kb_backend.metrics.emit_count" + ), patch.object( + worker, "_set_retain_until", side_effect=lambda *a: calls.append("retain_until") + ), patch( + "apis.shared.kb_backend.records.set_migration_state", + side_effect=lambda *a, **k: calls.append("state"), + ): + result = await worker.run_promote(ASSISTANT_ID, ASSISTANT_ID, _kb_record(r.PROMOTE)) + + assert calls == ["promote", "retain_until", "state"] + assert result.to_state == r.RETAIN + + @pytest.mark.asyncio + async def test_concurrent_promotion_yields_one_winner(self): + """Requirement 15.10. The property belongs to the conditional write, so the + test is that the loser's exception is not swallowed into a second success.""" + from botocore.exceptions import ClientError + + winners = {"n": 0} + + def _promote(assistant_id, app_kb_id, generation, now_iso): + winners["n"] += 1 + if winners["n"] > 1: + raise ClientError( + {"Error": {"Code": "ConditionalCheckFailedException"}}, "UpdateItem" + ) + + # The loser re-reads before deciding, because a refused write means either + # "somebody else promoted" (success) or "a guard genuinely failed" (not). + # Here the record is still unpromoted, so the refusal must propagate. + with patch.dict("os.environ", BASE_ENV, clear=True), patch( + "apis.shared.kb_backend.records.promote_engine", side_effect=_promote + ), patch( + "apis.shared.kb_backend.records.get_kb_record", return_value=_kb_record(r.PROMOTE) + ), patch("apis.shared.kb_backend.metrics.emit_count"), patch.object( + worker, "_set_retain_until" + ), patch("apis.shared.kb_backend.records.set_migration_state"): + first = await worker.run_promote(ASSISTANT_ID, ASSISTANT_ID, _kb_record(r.PROMOTE)) + with pytest.raises(ClientError): + await worker.run_promote(ASSISTANT_ID, ASSISTANT_ID, _kb_record(r.PROMOTE)) + + assert first.to_state == r.RETAIN + assert winners["n"] == 2 + + def test_the_retain_window_cannot_be_shortened_below_thirty_days(self): + """Requirement 15.11 says *at least* 30 days. Shortening the rollback + window is not a tuning knob.""" + with patch.dict("os.environ", {"KB_MIGRATION_RETAIN_DAYS": "3"}, clear=True): + assert worker._retain_days() == 30 + with patch.dict("os.environ", {"KB_MIGRATION_RETAIN_DAYS": "90"}, clear=True): + assert worker._retain_days() == 90 + + +class TestRollback: + @pytest.mark.asyncio + async def test_rollback_moves_no_data(self): + """Requirement 17.2. The legacy index was never mutated — that is what + building the managed corpus alongside it bought.""" + touched: List[str] = [] + + with patch.dict("os.environ", BASE_ENV, clear=True), patch( + "apis.shared.kb_backend.records.rollback_engine", + side_effect=lambda *a: touched.append("engine"), + ), patch("apis.shared.kb_backend.metrics.emit_count"): + result = await worker.rollback(ASSISTANT_ID, ASSISTANT_ID) + + assert touched == ["engine"] + assert "no data moved" in result.detail + + @pytest.mark.asyncio + async def test_rollback_does_not_delete_the_managed_knowledge_base(self): + """Deleting it here would turn a reversible decision into an irreversible + one at the moment somebody is least sure.""" + with patch.dict("os.environ", BASE_ENV, clear=True), patch( + "apis.shared.kb_backend.records.rollback_engine" + ), patch("apis.shared.kb_backend.metrics.emit_count"), patch( + "apis.shared.kb_backend.tombstones.delete_knowledge_base", create=True + ) as delete_kb: + await worker.rollback(ASSISTANT_ID, ASSISTANT_ID) + + delete_kb.assert_not_called() + + @pytest.mark.asyncio + async def test_a_pre_promotion_failure_leaves_the_record_on_legacy(self): + """Requirement 17.4. `failed` is terminal and removes the work keys; the + engine attribute was never written, so the knowledge base is still legacy + and still usable.""" + recorded: List[tuple] = [] + + def _set_state(assistant_id, app_kb_id, new_state, generation, due=None, expected=None, error=None): + recorded.append((new_state, error)) + + with patch.dict("os.environ", BASE_ENV, clear=True), patch( + "apis.shared.kb_backend.records.get_kb_record", return_value=_kb_record(r.VERIFY) + ), patch.object(worker, "take_lease", return_value="later"), patch.object( + worker, "run_verify", side_effect=worker.VerificationFailed("canary empty") + ), patch( + "apis.shared.kb_backend.metrics.emit_count" + ), patch( + "apis.shared.kb_backend.records.set_migration_state", side_effect=_set_state + ), patch( + "apis.shared.kb_backend.records.promote_engine" + ) as promote: + result = await worker.run_step(ASSISTANT_ID) + + assert result.to_state == r.MIGRATION_FAILED + assert recorded and recorded[0][0] == r.MIGRATION_FAILED + promote.assert_not_called() + + +class TestResumingWithoutRedoingWork: + """The two behaviours the convergence property test forced into existence.""" + + def test_the_completed_set_is_read_off_the_record(self): + assert worker.already_migrated({}) == set() + assert worker.already_migrated({"migratedDocIds": {"d1", "d2"}}) == {"d1", "d2"} + + def test_a_non_iterable_completed_set_degrades_to_empty(self): + """Re-ingesting is slow, not wrong — ``customDocumentIdentifier`` makes it a + replace — so a malformed attribute must not stop the migration.""" + assert worker.already_migrated({"migratedDocIds": 7}) == set() + + @pytest.mark.asyncio + async def test_a_resumed_shadow_skips_documents_it_already_ingested(self): + """Before this, a crash near the end of a PDF corpus re-parsed all of it — + 37-264 s per document, so an hour of work redone for nothing.""" + docs = [_doc("d1"), _doc("d2"), _doc("d3")] + backend = StubBackend() + record = _kb_record(r.SHADOW, migratedDocIds={"d1", "d2"}) + + with patch.dict("os.environ", BASE_ENV, clear=True), patch.object( + worker, "list_document_items", return_value=docs + ), patch.object( + worker, "get_document_item", side_effect=lambda a, d: _doc(d) + ), patch( + "apis.shared.kb_backend.byte_cap.reserve_snapshot" + ) as reserve, patch( + "apis.shared.kb_backend.provisioning.provision_managed_kb", side_effect=_async_noop + ), patch( + "apis.shared.kb_backend.metrics.emit_count" + ), patch.object( + worker, "_record_progress" + ), patch( + "apis.shared.kb_backend.records.set_migration_state" + ): + await worker.run_shadow(ASSISTANT_ID, ASSISTANT_ID, record, backend) + + assert backend.ingested == ["d3"] + # And the corpus is not reserved a second time: the accumulator is on the + # record, so re-reserving would double-count the owner's own corpus against + # their cap until the migration refused itself. + reserve.assert_not_called() + + @pytest.mark.asyncio + async def test_progress_persists_the_ids_not_just_a_count(self): + docs = [_doc("d1"), _doc("d2")] + captured = {} + + async def _progress(assistant_id, app_kb_id, *, migrated, total, skipped, newly_done=None): + captured["newly_done"] = list(newly_done or []) + captured["migrated"] = migrated + + with patch.dict("os.environ", BASE_ENV, clear=True), patch.object( + worker, "list_document_items", return_value=docs + ), patch.object( + worker, "get_document_item", side_effect=lambda a, d: _doc(d) + ), patch( + "apis.shared.kb_backend.byte_cap.reserve_snapshot" + ), patch( + "apis.shared.kb_backend.provisioning.provision_managed_kb", side_effect=_async_noop + ), patch( + "apis.shared.kb_backend.metrics.emit_count" + ), patch.object( + worker, "_record_progress", side_effect=_progress + ), patch( + "apis.shared.kb_backend.records.set_migration_state" + ): + await worker.run_shadow(ASSISTANT_ID, ASSISTANT_ID, _kb_record(), StubBackend()) + + assert sorted(captured["newly_done"]) == ["d1", "d2"], ( + "a count alone cannot tell a resume *which* documents to skip" + ) + + @pytest.mark.asyncio + async def test_a_record_already_promoted_finishes_instead_of_failing(self): + """The crash window between the promotion write and the state transition. + + The promotion write is guarded on ``attribute_not_exists(retrievalEngine)``, + so retrying it is refused — and treating that refusal as a failure would + mark a migration that actually succeeded as ``failed``, leaving a promoted + knowledge base with no retention window. + """ + record = _kb_record(r.PROMOTE, retrievalEngine="managed") + calls: List[str] = [] + + with patch.dict("os.environ", BASE_ENV, clear=True), patch( + "apis.shared.kb_backend.records.promote_engine", + side_effect=lambda *a: calls.append("promote"), + ), patch("apis.shared.kb_backend.metrics.emit_count"), patch.object( + worker, "_set_retain_until", side_effect=lambda *a: calls.append("retain_until") + ), patch( + "apis.shared.kb_backend.records.set_migration_state", + side_effect=lambda *a, **k: calls.append("state"), + ): + result = await worker.run_promote(ASSISTANT_ID, ASSISTANT_ID, record) + + assert "promote" not in calls, "promoted a second time" + assert calls == ["retain_until", "state"] + assert result.to_state == r.RETAIN + assert "already promoted" in result.detail + + @pytest.mark.asyncio + async def test_a_refused_promotion_on_an_unpromoted_record_still_raises(self): + """So "already promoted" cannot become a blanket swallow of the guard.""" + with patch.dict("os.environ", BASE_ENV, clear=True), patch( + "apis.shared.kb_backend.records.promote_engine", + side_effect=RuntimeError("conditional check failed"), + ), patch( + "apis.shared.kb_backend.records.get_kb_record", return_value=_kb_record(r.PROMOTE) + ), patch("apis.shared.kb_backend.metrics.emit_count"): + with pytest.raises(RuntimeError): + await worker.run_promote(ASSISTANT_ID, ASSISTANT_ID, _kb_record(r.PROMOTE)) + + +class TestRehydrationReappliesTheResourcePolicy: + """Task 13.7 / Requirement 24.12, asserted at the level a rehydration works at. + + A resource policy attaches to the AWS knowledge base ARN. Provisioning that + produces a *new* ``awsKbId`` — a rehydration, or a replacement after a failed + delete — therefore leaves the old policy on a resource nobody reads, and sharing + silently stops. The repair is a state comparison rather than an event, so it + cannot be bypassed by a code path that forgets to fire anything. + """ + + @pytest.mark.asyncio + async def test_a_new_aws_kb_id_makes_the_recorded_policy_stale(self): + from apis.shared.kb_backend.resource_policy import POLICY_KB_ID_ATTR, policy_is_stale + + rehydrated = _kb_record(r.RETAIN, awsKbId="KB-NEW", **{POLICY_KB_ID_ATTR: "KB123"}) + assert policy_is_stale(rehydrated) is True + + @pytest.mark.asyncio + async def test_the_policy_is_reapplied_to_the_new_arn(self): + from apis.shared.kb_backend.resource_policy import ( + POLICY_KB_ID_ATTR, + ensure_retrieve_policy, + ) + + client = MagicMock() + client.put_resource_policy.return_value = {"revisionId": "rev-after-rehydration"} + rehydrated = _kb_record(r.RETAIN, awsKbId="KB-NEW", **{POLICY_KB_ID_ATTR: "KB123"}) + + with patch.dict( + "os.environ", + { + **BASE_ENV, + "AWS_ACCOUNT_ID": "123456789012", + "MANAGED_KB_RETRIEVAL_PRINCIPAL_ARNS": "arn:aws:iam::123456789012:role/runtime", + }, + clear=True, + ), patch("apis.shared.kb_backend.records.set_resource_policy_state") as setter: + revision = await ensure_retrieve_policy( + ASSISTANT_ID, ASSISTANT_ID, shared=True, record=rehydrated, client=client + ) + + assert revision == "rev-after-rehydration" + assert client.put_resource_policy.call_args.kwargs["resourceArn"].endswith( + "knowledge-base/KB-NEW" + ) + setter.assert_called_once_with( + ASSISTANT_ID, ASSISTANT_ID, "KB-NEW", "rev-after-rehydration" + ) + + +# ── Mixed old/new deployment ───────────────────────────────────────────────── +class TestMixedDeployment: + def test_a_record_without_an_engine_resolves_to_legacy(self): + """Requirements 1.6, 24.8. Old and new code serving simultaneously agree, + because "absence means legacy" is a property of the data rather than of the + code version reading it.""" + for item in ({}, None, _kb_record(), {"appKbId": "x", "migrationState": r.SHADOW}): + assert r.resolve_engine(item) == r.ENGINE_LEGACY + + def test_only_an_explicit_managed_value_resolves_to_managed(self): + assert r.resolve_engine({"retrievalEngine": "managed"}) == r.ENGINE_MANAGED + for wrong in ("MANAGED", "Managed", "s3vectors", "", None, True): + assert r.resolve_engine({"retrievalEngine": wrong}) == r.ENGINE_LEGACY + + def test_a_mid_migration_record_still_serves_legacy(self): + """Requirements 15.3, 16.1. `shadow` and `verify` never touch + `retrievalEngine`, so a knowledge base being migrated is indistinguishable + from one that is not, to anything doing retrieval.""" + for state in (r.SHADOW, r.VERIFY, r.PROMOTE): + assert r.resolve_engine(_kb_record(state)) == r.ENGINE_LEGACY diff --git a/backend/tests/lambdas/test_kb_reconciler.py b/backend/tests/lambdas/test_kb_reconciler.py new file mode 100644 index 000000000..52c6dd8e4 --- /dev/null +++ b/backend/tests/lambdas/test_kb_reconciler.py @@ -0,0 +1,1011 @@ +"""Daily reconciler — the join, the age gate, and the disarmed default. + +Feature: managed-kb-migration, task 10.3. +Requirements: 24.4, 14.1-14.8, 19.7, 19.8. + +Three assertions here are the reason the file exists, and each guards a mistake +that a passing test suite would otherwise hide: + +**The age gate reads AWS's ``createdAt``, never discovery time.** Asserted from +both ends. An orphan that AWS says is eight days old is deletable on the *very +first* run that ever sees it — an implementation that started a 24-hour clock at +discovery would leave it, and would then leave it again after any reconciler +outage. And a knowledge base AWS says is 30 seconds old is left alone even though +it is equally newly discovered, because that one is an in-flight create. + +**Record-only marks and never deletes.** A KB_Record whose AWS knowledge base has +gone means the *vectors* are gone. The uploaded bytes are still in S3 and the +``DOC#`` rows still describe them, so the corpus rebuilds on the next ingest and +the owner re-uploads nothing. The record is the only pointer to that corpus, so +deleting it is the single action in this module that would lose user data. + +**Report-only really is a no-op.** The shipped mode logs intended deletions and +issues none, and the arming flag treats an empty string as off — an unset GitHub +Actions variable expands to ``""``. + +No test contacts AWS. DynamoDB is moto; ``bedrock-agent`` is a stub +(Requirement 24.11). +""" + +from datetime import datetime, timedelta, timezone + +import boto3 +import pytest +from moto import mock_aws + +from apis.app_api.kb_migration import reconciler as rec +from apis.shared.kb_backend import tombstones as tomb +from apis.shared.kb_backend import tags as kb_tags +from tests.shared.test_kb_tombstones import FakeBedrockAgent + +REGION = "us-east-1" +TABLE = "test-kb-reconciler" +PREFIX = "testprefix" +ENV = "testenv" +NOW = datetime(2026, 6, 1, 12, 0, 0, tzinfo=timezone.utc) + + +@pytest.fixture() +def table(monkeypatch): + monkeypatch.setenv("AWS_DEFAULT_REGION", REGION) + monkeypatch.setenv("AWS_ACCESS_KEY_ID", "testing") + monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "testing") + monkeypatch.setenv("AWS_SESSION_TOKEN", "testing") + monkeypatch.setenv("DYNAMODB_ASSISTANTS_TABLE_NAME", TABLE) + monkeypatch.setenv(kb_tags.ENV_TAG_VALUE_PREFIX, PREFIX) + monkeypatch.setenv(kb_tags.ENV_TAG_VALUE_ENVIRONMENT, ENV) + # Never inherited from the developer's shell: the whole point of the flag is + # that the reconciler is disarmed unless something says otherwise. + monkeypatch.delenv(rec.FLAG_RECONCILER_ARMED, raising=False) + + with mock_aws(): + boto3.client("dynamodb", region_name=REGION).create_table( + TableName=TABLE, + KeySchema=[ + {"AttributeName": "PK", "KeyType": "HASH"}, + {"AttributeName": "SK", "KeyType": "RANGE"}, + ], + AttributeDefinitions=[ + {"AttributeName": "PK", "AttributeType": "S"}, + {"AttributeName": "SK", "AttributeType": "S"}, + ], + BillingMode="PAY_PER_REQUEST", + ) + yield boto3.resource("dynamodb", region_name=REGION).Table(TABLE) + + +@pytest.fixture(autouse=True) +def no_metrics(monkeypatch): + monkeypatch.setattr(rec, "emit_count", lambda *a, **k: None) + monkeypatch.setattr(tomb, "emit_count", lambda *a, **k: None) + + +def _iso(moment): + """The exact timestamp shape this feature writes everywhere. + + Spelled out rather than ``isoformat()`` because every comparison in the + idleness path is lexicographic on this format; an offset-style string would + sort differently and the test would be measuring the wrong thing. + """ + return moment.strftime("%Y-%m-%dT%H:%M:%SZ") + + +def _arn(kb_id): + return f"arn:aws:bedrock:{REGION}:123456789012:knowledge-base/{kb_id}" + + +def _aws_kb(kb_id, created_at, status="ACTIVE", app_kb_id=None): + """One knowledge base as AWS reports it, with AWS's own ``createdAt``.""" + return { + "knowledgeBaseId": kb_id, + "name": f"{PREFIX}-kb-{app_kb_id or kb_id}", + "status": status, + "knowledgeBaseArn": _arn(kb_id), + "roleArn": "arn:aws:iam::123456789012:role/kb", + "createdAt": created_at, + } + + +def _ours(kb_id, app_kb_id): + return { + # Built through the canonical helper, not spelled out: a fixture that + # hardcodes tag keys is a fixture that keeps passing after the keys change + # under it, which is how the three-way drift stayed invisible. + _arn(kb_id): kb_tags.build_tags(app_kb_id, "u-1", PREFIX, ENV) + } + + +def _seed_record(table, assistant_id, aws_kb_id=None, **extra): + item = { + "PK": f"AST#{assistant_id}", + "SK": f"KB#{assistant_id}", + "appKbId": assistant_id, + "retrievalEngine": "managed", + } + if aws_kb_id: + item["awsKbId"] = aws_kb_id + item["awsDataSourceId"] = f"DS{aws_kb_id}" + item.update(extra) + table.put_item(Item=item) + return item + + +def _record(table, assistant_id): + return table.get_item( + Key={"PK": f"AST#{assistant_id}", "SK": f"KB#{assistant_id}"} + ).get("Item") + + +def _run(client, table, **kwargs): + kwargs.setdefault("now", NOW) + kwargs.setdefault("stored_bytes_resolver", lambda _assistant_id: None) + return rec.reconcile(client=client, **kwargs) + + +# ── Requirement 19.7, 19.8: the arming flag ────────────────────────────────── +class TestArmingFlag: + @pytest.mark.parametrize( + "value", + ["", " ", "0", "false", "False", "off", "no", "disabled", "maybe"], + ) + def test_falsy_and_empty_values_are_off(self, monkeypatch, value): + """An **empty string must read as off** (Requirement 19.8). + + An unset GitHub Actions variable expands to ``""``, so a truthiness test + on the raw value is the exact bug this guards. ``"false"`` matters too: + ``bool("false")`` is ``True``. + """ + monkeypatch.setenv(rec.FLAG_RECONCILER_ARMED, value) + assert rec.reconciler_armed() is False + + def test_unset_is_off(self, monkeypatch): + monkeypatch.delenv(rec.FLAG_RECONCILER_ARMED, raising=False) + assert rec.reconciler_armed() is False + + @pytest.mark.parametrize("value", ["1", "true", "TRUE", "yes", "on", "enabled", " true "]) + def test_affirmative_values_arm(self, monkeypatch, value): + monkeypatch.setenv(rec.FLAG_RECONCILER_ARMED, value) + assert rec.reconciler_armed() is True + + def test_reconcile_defaults_to_the_flag(self, table, monkeypatch): + monkeypatch.setenv(rec.FLAG_RECONCILER_ARMED, "") + client = FakeBedrockAgent( + knowledge_bases=[_aws_kb("KBORPH1", NOW - timedelta(days=8))], + tags=_ours("KBORPH1", "ast-orph1"), + ) + + report = rec.reconcile( + client=client, now=NOW, stored_bytes_resolver=lambda _a: None + ) + + assert report.armed is False + assert report.to_dict()["mode"] == "report-only" + + +# ── Requirement 14.7: report-only deletes nothing ──────────────────────────── +class TestReportOnlyDeletesNothing: + def test_an_eligible_orphan_is_reported_and_not_deleted(self, table): + """The shipped mode. It must plan the deletion and perform none of it.""" + client = FakeBedrockAgent( + knowledge_bases=[_aws_kb("KBORPH1", NOW - timedelta(days=8))], + tags=_ours("KBORPH1", "ast-orph1"), + ) + + report = _run(client, table, armed=False) + + assert report.orphans == 1 + assert [p.kb_id for p in report.planned_deletions] == ["KBORPH1"] + assert report.deletions_performed == 0 + assert client.delete_calls == [], ( + "report-only mode issued a DeleteKnowledgeBase call" + ) + + def test_report_only_makes_no_mutating_call_at_all(self, table): + """Nothing happens: no AWS delete, and no DynamoDB side effect. + + Asserted on the AWS call log rather than on the end state of the table, + because the saga cleans up after itself — a run that wrote a tombstone, + deleted the knowledge base and then cleared the tombstone leaves the table + looking exactly as untouched as a run that did nothing. + """ + client = FakeBedrockAgent( + knowledge_bases=[_aws_kb("KBORPH1", NOW - timedelta(days=8))], + tags=_ours("KBORPH1", "ast-orph1"), + ) + + _run(client, table, armed=False) + + performed = [op for op, _probe in client.observations if op.startswith("delete_")] + assert performed == [], f"report-only mode issued mutating calls: {performed}" + assert tomb.iter_tombstones("ast-orph1") == [] + assert table.scan()["Items"] == [] + + def test_armed_actually_deletes_through_the_saga(self, table): + """The contrast case, so the report-only assertion means something.""" + client = FakeBedrockAgent( + knowledge_bases=[_aws_kb("KBORPH1", NOW - timedelta(days=8), app_kb_id="ast-orph1")], + tags=_ours("KBORPH1", "ast-orph1"), + ) + + report = _run(client, table, armed=True) + + assert client.delete_calls == ["KBORPH1"] + assert report.deletions_performed == 1 + assert report.planned_deletions[0].performed is True + # The saga cleared its own tombstone once AWS confirmed absence. + assert tomb.iter_tombstones("ast-orph1") == [] + + def test_armed_delete_writes_the_tombstone_before_calling_aws(self, table): + """The orphan path must go through the saga, not a bare delete call.""" + client = FakeBedrockAgent( + knowledge_bases=[_aws_kb("KBORPH1", NOW - timedelta(days=8), app_kb_id="ast-orph1")], + tags=_ours("KBORPH1", "ast-orph1"), + probe=lambda: table.get_item( + Key={"PK": "AST#ast-orph1", "SK": "KBTOMB#ast-orph1"} + ).get("Item") + is not None, + ) + + _run(client, table, armed=True) + + assert client.probes_for("delete_knowledge_base") == [True], ( + "the orphan was deleted without a tombstone in place first" + ) + + def test_an_orphan_tombstone_declares_its_partition_synthetic(self, table): + """An orphan has no assistant id, so its ``PK`` is not a real partition. + + The tombstone still has to exist — a delete that fails mid-flight must + leave a work item either way — but it lands under the ``appKbId`` tag + rather than an assistant, so ``iter_tombstones()`` will never + surface it. Unmarked, that item reads as a tombstone for an assistant that + does not exist, which sends whoever is triaging it looking for a record + that was never there. Asserted while the tombstone is still in place, + i.e. from inside the delete call, because a successful saga clears it. + """ + seen = {} + + def probe(): + item = table.get_item( + Key={"PK": "AST#ast-orph1", "SK": "KBTOMB#ast-orph1"} + ).get("Item") + if item: + seen.update(item) + return item is not None + + client = FakeBedrockAgent( + knowledge_bases=[_aws_kb("KBORPH1", NOW - timedelta(days=8), app_kb_id="ast-orph1")], + tags=_ours("KBORPH1", "ast-orph1"), + probe=probe, + ) + + _run(client, table, armed=True) + + assert seen, "no tombstone was ever written for the orphan" + assert seen.get(tomb.SYNTHETIC_PARTITION) is True, ( + f"the orphan tombstone did not declare its partition synthetic: {dict(seen)}" + ) + # And it says which identifier the partition was derived from, which is the + # first thing an operator needs in order to go find the resource. + assert seen.get("anchorSource") == f"tag:{kb_tags.TAG_KEY_APP_KB_ID}" + assert seen.get("awsKbId") == "KBORPH1" + + def test_a_tombstone_for_a_real_record_is_not_marked_synthetic(self, table): + """The contrast case: the marker must distinguish, not decorate everything. + + A record-backed delete anchors on a genuine assistant partition, so the + flag must be absent there — otherwise it carries no information. + """ + client = FakeBedrockAgent(knowledge_bases=[], tags={}) + probe = {} + + def spy(): + probe.update( + table.get_item(Key={"PK": "AST#ast-real", "SK": "KBTOMB#ast-real"}).get("Item") + or {} + ) + return True + + client.probe = spy + tomb.write_kb_tombstone("ast-real", "ast-real", "KBREAL") + spy() + + assert probe, "the control tombstone was not written" + assert tomb.SYNTHETIC_PARTITION not in probe, ( + f"a record-backed tombstone was flagged synthetic: {dict(probe)}" + ) + + +# ── Requirement 14.3, 14.4: the age gate ───────────────────────────────────── +class TestAgeGateUsesAwsCreatedAt: + def test_an_orphan_aws_calls_old_is_deletable_on_its_first_discovery(self, table): + """TRAP: age-gating on discovery time would skip this. + + The reconciler has never seen this knowledge base before — this is its + first ever run. AWS says the resource is eight days old, so it is + immediately eligible. An implementation that stamped a ``firstSeenAt`` and + waited 24 hours from there would report zero planned deletions here, and + would do so again after every reconciler outage. + """ + client = FakeBedrockAgent( + knowledge_bases=[_aws_kb("KBOLD", NOW - timedelta(days=8))], + tags=_ours("KBOLD", "ast-old"), + ) + + report = _run(client, table, armed=False) + + assert [p.kb_id for p in report.planned_deletions] == ["KBOLD"], ( + "an 8-day-old orphan was not eligible on first discovery, which is " + "what age-gating on discovery time looks like" + ) + assert report.skipped_too_young == [] + + def test_a_freshly_created_knowledge_base_is_left_alone(self, table): + """The other half of the trap: newly discovered is not newly created. + + 30 seconds old by AWS's clock — an in-flight create whose record has not + been attached yet. Deleting this is the failure mode that loses a user's + upload mid-provisioning. + """ + client = FakeBedrockAgent( + knowledge_bases=[_aws_kb("KBNEW", NOW - timedelta(seconds=30))], + tags=_ours("KBNEW", "ast-new"), + ) + + report = _run(client, table, armed=True) + + assert report.planned_deletions == [] + assert report.skipped_too_young == ["KBNEW"] + assert client.delete_calls == [], "an in-flight create was deleted" + + def test_the_boundary_is_twenty_four_hours(self, table): + """23 h 59 m survives; 24 h 01 m does not.""" + client = FakeBedrockAgent( + knowledge_bases=[ + _aws_kb("KBJUSTUNDER", NOW - timedelta(hours=23, minutes=59)), + _aws_kb("KBJUSTOVER", NOW - timedelta(hours=24, minutes=1)), + ], + tags={**_ours("KBJUSTUNDER", "a1"), **_ours("KBJUSTOVER", "a2")}, + ) + + report = _run(client, table, armed=False) + + assert [p.kb_id for p in report.planned_deletions] == ["KBJUSTOVER"] + assert report.skipped_too_young == ["KBJUSTUNDER"] + + def test_the_gate_is_a_pure_function_of_the_aws_timestamp(self): + eight_days = NOW - timedelta(days=8) + thirty_seconds = NOW - timedelta(seconds=30) + + assert rec.orphan_is_deletable(eight_days, now=NOW) is True + assert rec.orphan_is_deletable(thirty_seconds, now=NOW) is False + # Identical answer regardless of when it is asked, which is the property a + # discovery-time clock does not have. + assert rec.orphan_is_deletable(eight_days, now=NOW + timedelta(days=30)) is True + + def test_a_missing_created_at_fails_closed(self): + """No timestamp from AWS means no deletion. Never a guess.""" + assert rec.orphan_is_deletable(None, now=NOW) is False + assert rec.orphan_is_deletable("not-a-date", now=NOW) is False + + def test_an_orphan_without_a_created_at_is_not_deleted(self, table): + kb = _aws_kb("KBNODATE", None) + kb.pop("createdAt") + client = FakeBedrockAgent(knowledge_bases=[kb], tags=_ours("KBNODATE", "a3")) + + report = _run(client, table, armed=True) + + assert report.planned_deletions == [] + assert report.skipped_too_young == ["KBNODATE"] + assert client.delete_calls == [] + + @pytest.mark.parametrize( + "created", + [ + datetime(2026, 5, 1, tzinfo=timezone.utc), + "2026-05-01T00:00:00Z", + datetime(2026, 5, 1).timestamp(), + ], + ) + def test_aws_timestamp_shapes_all_parse(self, created): + """boto3 gives a datetime; a stub or a JSON round-trip gives the others.""" + assert rec.parse_aws_timestamp(created) is not None + + def test_min_age_is_read_at_call_time(self, monkeypatch): + """The threshold must be patchable, not frozen into a default argument.""" + created = NOW - timedelta(hours=2) + assert rec.orphan_is_deletable(created, now=NOW) is False + + monkeypatch.setattr(rec, "ORPHAN_MIN_AGE_HOURS", 1.0) + assert rec.orphan_is_deletable(created, now=NOW) is True + + +# ── Requirement 14.5: record-only never deletes the record ─────────────────── +class TestRecordOnlyMarksMissing: + def test_a_stale_pointer_is_marked_not_removed(self, table): + """TRAP: the record is the only pointer to a recoverable corpus. + + The vectors are gone; the documents are not. Deleting the record would + destroy the mapping the rebuild depends on, and the owner would have to + re-upload. + """ + _seed_record(table, "ast-stale", aws_kb_id="KBGONE") + client = FakeBedrockAgent(knowledge_bases=[], tags={}) + + report = _run(client, table, armed=True) + + assert report.marked_missing == ["ast-stale"] + record = _record(table, "ast-stale") + assert record is not None, ( + "the KB_Record was deleted; its documents are still valid and the " + "knowledge base rebuilds from them on the next ingest" + ) + assert record["vectorState"] == rec.VECTOR_STATE_MISSING + assert record["vectorStateObservedAt"] + + def test_the_documents_and_identifiers_are_left_intact(self, table): + """Nothing else about the record is touched, including its ``DOC#`` rows.""" + _seed_record(table, "ast-stale", aws_kb_id="KBGONE") + table.put_item( + Item={"PK": "AST#ast-stale", "SK": "DOC#doc-1", "status": "complete"} + ) + client = FakeBedrockAgent(knowledge_bases=[], tags={}) + + _run(client, table, armed=True) + + record = _record(table, "ast-stale") + assert record["awsKbId"] == "KBGONE" + assert record["retrievalEngine"] == "managed" + doc = table.get_item(Key={"PK": "AST#ast-stale", "SK": "DOC#doc-1"})["Item"] + assert doc["status"] == "complete" + + def test_marking_missing_is_not_a_deletion_even_when_armed(self, table): + """Being armed licenses deleting *orphans*, never records.""" + _seed_record(table, "ast-stale", aws_kb_id="KBGONE") + client = FakeBedrockAgent(knowledge_bases=[], tags={}) + + report = _run(client, table, armed=True) + + assert report.deletions_performed == 0 + assert client.delete_calls == [] + assert _record(table, "ast-stale") is not None + + def test_an_unprovisioned_record_is_not_marked_missing(self, table): + """No ``awsKbId`` means provisioning has not finished, not that AWS lost it.""" + _seed_record(table, "ast-provisioning", aws_kb_id=None) + client = FakeBedrockAgent(knowledge_bases=[], tags={}) + + report = _run(client, table, armed=True) + + assert report.marked_missing == [] + assert _record(table, "ast-provisioning").get("vectorState") is None + + def test_a_tombstone_row_is_not_mistaken_for_a_record(self, table): + """``KBTOMB#`` must not be swept up by the ``KB#`` prefix scan.""" + tomb.write_kb_tombstone("ast-t", "ast-t", "KBX", "DSX") + client = FakeBedrockAgent(knowledge_bases=[], tags={}) + + report = _run(client, table, armed=False) + + assert report.records == 0 + assert report.marked_missing == [] + + +# ── Requirement 14.6: both sides agree ─────────────────────────────────────── +class TestBothSidesRefreshStoredBytes: + def test_stored_bytes_is_re_anchored_from_the_resolver(self, table): + _seed_record(table, "ast-both", aws_kb_id="KBBOTH", storedBytes=10) + client = FakeBedrockAgent( + knowledge_bases=[_aws_kb("KBBOTH", NOW - timedelta(days=8))], + tags=_ours("KBBOTH", "ast-both"), + ) + + report = _run(client, table, armed=False, stored_bytes_resolver=lambda _a: 4096) + + assert report.matched == 1 + assert report.orphans == 0 + assert report.refreshed_bytes == ["ast-both"] + assert int(_record(table, "ast-both")["storedBytes"]) == 4096 + + def test_an_unchanged_total_writes_nothing(self, table): + """A daily no-op write per knowledge base would be pure cost.""" + _seed_record(table, "ast-both", aws_kb_id="KBBOTH", storedBytes=4096) + client = FakeBedrockAgent( + knowledge_bases=[_aws_kb("KBBOTH", NOW - timedelta(days=8))], + tags=_ours("KBBOTH", "ast-both"), + ) + + report = _run(client, table, armed=False, stored_bytes_resolver=lambda _a: 4096) + + assert report.refreshed_bytes == [] + + def test_a_failed_size_lookup_leaves_stored_bytes_alone(self, table): + """Writing a zero on a failed listing hands the owner their quota back.""" + _seed_record(table, "ast-both", aws_kb_id="KBBOTH", storedBytes=4096) + client = FakeBedrockAgent( + knowledge_bases=[_aws_kb("KBBOTH", NOW - timedelta(days=8))], + tags=_ours("KBBOTH", "ast-both"), + ) + + _run(client, table, armed=False, stored_bytes_resolver=lambda _a: None) + + assert int(_record(table, "ast-both")["storedBytes"]) == 4096 + + def test_a_recovered_knowledge_base_clears_a_stale_missing_marker(self, table): + """Otherwise the UI keeps reporting a knowledge base broken after the fix.""" + _seed_record( + table, + "ast-both", + aws_kb_id="KBBOTH", + storedBytes=4096, + vectorState=rec.VECTOR_STATE_MISSING, + ) + client = FakeBedrockAgent( + knowledge_bases=[_aws_kb("KBBOTH", NOW - timedelta(days=8))], + tags=_ours("KBBOTH", "ast-both"), + ) + + _run(client, table, armed=False, stored_bytes_resolver=lambda _a: 4096) + + assert _record(table, "ast-both").get("vectorState") is None + + def test_stored_bytes_from_s3_totals_the_prefix(self, table): + class FakeS3: + def list_objects_v2(self, **kwargs): + assert kwargs["Prefix"] == "assistants/ast-s3/documents/" + return {"Contents": [{"Size": 100}, {"Size": 23}], "IsTruncated": False} + + assert rec.stored_bytes_from_s3("ast-s3", bucket="b", s3_client=FakeS3()) == 123 + + def test_stored_bytes_from_s3_returns_none_on_failure(self, table): + class Boom: + def list_objects_v2(self, **kwargs): + raise RuntimeError("access denied") + + assert rec.stored_bytes_from_s3("ast-s3", bucket="b", s3_client=Boom()) is None + + +# ── Requirement 14.8: bounded per-run action limit ─────────────────────────── +class TestPerRunActionLimit: + def _five_orphans(self): + kbs = [_aws_kb(f"KBORPH{i}", NOW - timedelta(days=8)) for i in range(5)] + tags = {} + for i in range(5): + tags.update(_ours(f"KBORPH{i}", f"ast-orph{i}")) + return FakeBedrockAgent(knowledge_bases=kbs, tags=tags) + + def test_the_limit_caps_planned_deletions_in_report_only_mode(self, table, monkeypatch): + """The report must describe what an armed run would really do. + + A report listing five intended deletions from a run that would only ever + perform two is a misleading artifact, and the report-only period exists + precisely so the artifact can be trusted. + """ + monkeypatch.setattr(rec, "MAX_DELETIONS_PER_RUN", 2) + client = self._five_orphans() + + report = _run(client, table, armed=False) + + assert report.orphans == 5 + assert len(report.planned_deletions) == 2 + assert report.limit_reached is True + + def test_the_limit_caps_actual_deletions_when_armed(self, table, monkeypatch): + monkeypatch.setattr(rec, "MAX_DELETIONS_PER_RUN", 2) + client = self._five_orphans() + + report = _run(client, table, armed=True) + + assert len(client.delete_calls) == 2, ( + f"the per-run limit did not bound the deletions: {client.delete_calls}" + ) + assert report.deletions_performed == 2 + assert report.limit_reached is True + + def test_without_the_limit_being_hit_nothing_is_flagged(self, table, monkeypatch): + monkeypatch.setattr(rec, "MAX_DELETIONS_PER_RUN", 25) + client = self._five_orphans() + + report = _run(client, table, armed=False) + + assert len(report.planned_deletions) == 5 + assert report.limit_reached is False + + def test_the_limit_is_read_at_call_time(self, monkeypatch): + assert rec.max_deletions_per_run() == rec.MAX_DELETIONS_PER_RUN + monkeypatch.setattr(rec, "MAX_DELETIONS_PER_RUN", 3) + assert rec.max_deletions_per_run() == 3 + monkeypatch.setenv("MANAGED_KB_RECONCILER_MAX_DELETIONS", "7") + assert rec.max_deletions_per_run() == 7 + + def test_the_environment_can_lower_the_limit_but_not_lift_it(self, monkeypatch): + """A bound an env var can raise without limit is not a bound. + + This is the only limit whose failure mode is irreversible, so the ceiling + has to hold against the variable rather than merely default below it. + """ + monkeypatch.setenv("MANAGED_KB_RECONCILER_MAX_DELETIONS", "3") + assert rec.max_deletions_per_run() == 3, "the env var could not lower the limit" + + monkeypatch.setenv("MANAGED_KB_RECONCILER_MAX_DELETIONS", "1000000") + assert rec.max_deletions_per_run() == rec.MAX_DELETIONS_CEILING, ( + "the environment lifted the per-run deletion bound past its ceiling" + ) + + def test_a_negative_limit_does_not_become_unbounded(self, monkeypatch): + """A negative slice bound would silently mean 'all of them' downstream.""" + monkeypatch.setenv("MANAGED_KB_RECONCILER_MAX_DELETIONS", "-5") + assert rec.max_deletions_per_run() == 0 + + +# ── Requirement 14.1: paginated and tag-filtered ───────────────────────────── +class TestJoinIsPaginatedAndTagFiltered: + def test_orphans_on_later_pages_are_still_found(self, table): + """Reading only page one would make account size decide correctness.""" + kbs = [_aws_kb(f"KBP{i}", NOW - timedelta(days=8)) for i in range(5)] + tags = {} + for i in range(5): + tags.update(_ours(f"KBP{i}", f"ast-p{i}")) + client = FakeBedrockAgent(knowledge_bases=kbs, tags=tags, page_size=2) + + report = _run(client, table, armed=False) + + assert report.aws_knowledge_bases == 5 + assert len(report.planned_deletions) == 5 + + def test_another_projects_knowledge_base_is_invisible(self, table): + client = FakeBedrockAgent( + knowledge_bases=[ + _aws_kb("KBMINE", NOW - timedelta(days=8)), + _aws_kb("KBTHEIRS", NOW - timedelta(days=8)), + ], + tags={ + **_ours("KBMINE", "ast-mine"), + _arn("KBTHEIRS"): kb_tags.build_tags("ast-theirs", "u-2", "other-project", "prod"), + }, + ) + + report = _run(client, table, armed=True) + + assert report.aws_knowledge_bases == 1 + assert client.delete_calls == ["KBMINE"], ( + "the reconciler acted outside its tag scope" + ) + + def test_an_untagged_knowledge_base_is_never_deleted(self, table): + client = FakeBedrockAgent( + knowledge_bases=[_aws_kb("KBBARE", NOW - timedelta(days=8))], tags={} + ) + + report = _run(client, table, armed=True) + + assert report.aws_knowledge_bases == 0 + assert client.delete_calls == [] + + def test_a_truncated_aws_walk_suppresses_missing_vector_marks(self, table, monkeypatch): + """An unmatched record on a partial walk may be one we never reached.""" + monkeypatch.setattr(rec, "MAX_KNOWLEDGE_BASES_PER_RUN", 1) + _seed_record(table, "ast-a", aws_kb_id="KBA") + _seed_record(table, "ast-b", aws_kb_id="KBB") + client = FakeBedrockAgent( + knowledge_bases=[ + _aws_kb("KBA", NOW - timedelta(days=8)), + _aws_kb("KBB", NOW - timedelta(days=8)), + ], + tags={**_ours("KBA", "ast-a"), **_ours("KBB", "ast-b")}, + ) + + report = _run(client, table, armed=True) + + assert report.limit_reached is True + assert report.marked_missing == [] + assert _record(table, "ast-b").get("vectorState") is None + + +# ── Requirement 13.7 seen from the reconciler ──────────────────────────────── +class TestDeleteUnsuccessfulOrphan: + def test_it_is_surfaced_and_not_retried(self, table): + """Retrying does not help and the resource keeps billing.""" + client = FakeBedrockAgent( + knowledge_bases=[ + _aws_kb("KBSTUCK", NOW - timedelta(days=200), status="DELETE_UNSUCCESSFUL") + ], + tags=_ours("KBSTUCK", "ast-stuck"), + ) + + report = _run(client, table, armed=True) + + assert len(report.planned_deletions) == 1 + planned = report.planned_deletions[0] + assert planned.error == tomb.KB_STATUS_DELETE_UNSUCCESSFUL + assert planned.performed is False + assert client.delete_calls == [] + + def test_it_appears_in_the_serialized_report(self, table): + client = FakeBedrockAgent( + knowledge_bases=[ + _aws_kb("KBSTUCK", NOW - timedelta(days=200), status="DELETE_UNSUCCESSFUL") + ], + tags=_ours("KBSTUCK", "ast-stuck"), + ) + + payload = _run(client, table, armed=False).to_dict() + + assert payload["plannedDeletions"][0]["status"] == "DELETE_UNSUCCESSFUL" + assert payload["deletionsPerformed"] == 0 + + +# ── Mixed and degenerate cases ─────────────────────────────────────────────── +class TestMixedRun: + def test_all_three_outcomes_in_one_pass(self, table): + _seed_record(table, "ast-both", aws_kb_id="KBBOTH", storedBytes=1) + _seed_record(table, "ast-stale", aws_kb_id="KBVANISHED") + client = FakeBedrockAgent( + knowledge_bases=[ + _aws_kb("KBBOTH", NOW - timedelta(days=8)), + _aws_kb("KBORPH", NOW - timedelta(days=8)), + ], + tags={**_ours("KBBOTH", "ast-both"), **_ours("KBORPH", "ast-orph")}, + ) + + report = _run(client, table, armed=False, stored_bytes_resolver=lambda _a: 99) + + assert report.records == 2 + assert report.matched == 1 + assert report.orphans == 1 + assert report.marked_missing == ["ast-stale"] + assert report.refreshed_bytes == ["ast-both"] + assert [p.kb_id for p in report.planned_deletions] == ["KBORPH"] + assert _record(table, "ast-stale") is not None + + def test_an_empty_account_and_empty_table_is_a_clean_no_op(self, table): + client = FakeBedrockAgent(knowledge_bases=[], tags={}) + + report = _run(client, table, armed=True) + + assert report.to_dict() == { + "armed": True, + "mode": "armed", + "awsKnowledgeBases": 0, + "records": 0, + "matched": 0, + "orphans": 0, + "plannedDeletions": [], + "deletionsPerformed": 0, + "skippedTooYoung": [], + "markedMissing": [], + "refreshedBytes": [], + "limitReached": False, + # Fleet gauges. Zero here, and asserted as an exact dict on purpose: the + # report is a stored artifact an operator reads, so a field appearing or + # vanishing should be a deliberate change to this list. + "storedBytes": 0, + "idleBytes": 0, + "unmeasuredIdleness": 0, + } + + def test_a_failing_delete_does_not_end_the_run(self, table, monkeypatch): + """One stuck orphan must not stop the reconciler reaching the others.""" + client = FakeBedrockAgent( + knowledge_bases=[ + _aws_kb("KBA", NOW - timedelta(days=8)), + _aws_kb("KBB", NOW - timedelta(days=8)), + ], + tags={**_ours("KBA", "ast-a"), **_ours("KBB", "ast-b")}, + polls_before_gone=10_000, + ) + monkeypatch.setattr(tomb, "KB_DELETE_POLL_TIMEOUT_SECONDS", 0.0) + monkeypatch.setattr(tomb, "KB_DELETE_POLL_INTERVAL_SECONDS", 0.0) + + report = _run(client, table, armed=True) + + assert len(report.planned_deletions) == 2 + assert report.deletions_performed == 0 + assert all(p.error for p in report.planned_deletions) + # And the tombstones survive as work items for the next run. + assert tomb.iter_tombstones("ast-a") + assert tomb.iter_tombstones("ast-b") + + +class TestLambdaHandler: + """The scheduled entry point, and the one input nobody reviews. + + ``lambda_handler`` takes an *event*. An event is not reviewable configuration: + an EventBridge target can carry a constant payload, and any principal with + ``lambda:InvokeFunction`` can supply one. So the flag has to be the only way + to arm (Requirement 19.7) — otherwise deletion of billed user resources is + reachable while every reviewable setting still reads report-only, and the only + trace left is an ``Invoke`` in CloudTrail. + """ + + @pytest.fixture() + def stub_client(self, monkeypatch): + """Make the un-injected client path safe: no AWS, and a delete log to read. + + ``lambda_handler`` deliberately passes no client, so this patches the + factory ``reconcile`` reaches for. Without it the test would try to build + a real ``bedrock-agent`` client (Requirement 24.11). + """ + from apis.shared.kb_backend import managed_backend + + # 2020: comfortably older than the 24h gate against real wall-clock time, + # since lambda_handler passes no ``now``. + client = FakeBedrockAgent( + knowledge_bases=[ + _aws_kb("KBORPH1", datetime(2020, 1, 1, tzinfo=timezone.utc), app_kb_id="ast-orph1") + ], + tags=_ours("KBORPH1", "ast-orph1"), + ) + monkeypatch.setattr(managed_backend, "bedrock_agent_client", lambda: client) + return client + + def test_it_returns_the_serialized_report(self, table, monkeypatch): + monkeypatch.setattr(rec, "reconcile", lambda **kwargs: rec.ReconcileReport(armed=False)) + + result = rec.lambda_handler({}, None) + + assert result["statusCode"] == 200 + assert result["report"]["mode"] == "report-only" + + @pytest.mark.parametrize("payload", [True, "true", 1, "1", "yes"]) + def test_the_event_cannot_arm_the_reconciler(self, table, stub_client, payload): + """A flag-off invocation carrying ``armed`` deletes nothing. + + Parametrised over a real boolean and the string/int spellings alike, + because the boolean is the one that would previously have worked: an + ``isinstance(x, bool)`` override honours ``True`` exactly, so a test that + only passed ``"true"`` proved nothing about the path that actually armed. + """ + result = rec.lambda_handler({"armed": payload}, None) + + assert result["report"]["mode"] == "report-only", ( + f"the event payload armed={payload!r} put the reconciler in armed mode" + ) + assert result["report"]["deletionsPerformed"] == 0 + assert stub_client.delete_calls == [], ( + f"the event payload armed={payload!r} caused a real DeleteKnowledgeBase" + ) + # And the orphan it declined to delete is still reported, so suppressing + # the delete has not also suppressed the finding. + assert result["report"]["orphans"] == 1 + + def test_the_flag_is_what_arms_it(self, table, stub_client, monkeypatch): + """The contrast case: same event, same orphan, flag on — now it deletes. + + Without this, the assertions above would also pass on a reconciler that + could never delete at all. + """ + monkeypatch.setenv(rec.FLAG_RECONCILER_ARMED, "true") + + result = rec.lambda_handler({"armed": False}, None) + + assert result["report"]["mode"] == "armed" + assert stub_client.delete_calls == ["KBORPH1"] + assert result["report"]["deletionsPerformed"] == 1 + + def test_an_ignored_arming_request_is_logged(self, table, stub_client, caplog): + """Silently dropping the field would hide a misconfigured schedule.""" + import logging + + with caplog.at_level(logging.WARNING): + rec.lambda_handler({"armed": True}, None) + + assert any( + "ignoring armed" in r.message and rec.FLAG_RECONCILER_ARMED in r.message + for r in caplog.records + ), f"no warning named the ignored override: {[r.message for r in caplog.records]}" + + +# ── Requirements 22.1, 22.5: the fleet gauges ──────────────────────────────── +class TestFleetGaugeAccumulation: + """The reconciler is where the gauges are computed, because it is already the + one pass that walks every knowledge base.""" + + def test_stored_bytes_sum_across_records(self, table): + _seed_record(table, "ast-a", aws_kb_id="KBA", storedBytes=3_000_000_000) + _seed_record(table, "ast-b", aws_kb_id="KBB", storedBytes=1_000_000_000) + client = FakeBedrockAgent(knowledge_bases=[], tags={}) + + report = _run(client, table) + + assert report.records == 2 + assert report.stored_bytes == 4_000_000_000 + + def test_an_agent_used_today_is_not_idle_however_stale_its_retrievals(self, table): + """Requirement 22.5, end to end through the reconciler. + + The knowledge base was last retrieved from 200 days ago but its agent was + used today — an agent answering questions its documents do not cover. Judged + by retrieval alone its bytes would count as idle and the follow-up spec's + eviction pass would delete a live corpus. + """ + _seed_record( + table, + "ast-busy", + aws_kb_id="KBA", + storedBytes=5_000_000_000, + lastRetrievedAt=_iso(NOW - timedelta(days=200)), + ) + table.put_item( + Item={ + "PK": "AST#ast-busy", + "SK": "METADATA", + "lastUsedAt": _iso(NOW - timedelta(hours=2)), + } + ) + client = FakeBedrockAgent(knowledge_bases=[], tags={}) + + report = _run(client, table) + + assert report.stored_bytes == 5_000_000_000 + assert report.idle_bytes == 0, ( + "a busy agent's corpus was counted as idle; idleness was derived from " + "retrieval alone" + ) + + def test_a_genuinely_dormant_knowledge_base_counts_as_idle(self, table): + """So the test above cannot pass by never counting anything.""" + _seed_record( + table, + "ast-cold", + aws_kb_id="KBA", + storedBytes=2_000_000_000, + lastRetrievedAt=_iso(NOW - timedelta(days=200)), + ) + table.put_item( + Item={ + "PK": "AST#ast-cold", + "SK": "METADATA", + "lastUsedAt": _iso(NOW - timedelta(days=180)), + } + ) + client = FakeBedrockAgent(knowledge_bases=[], tags={}) + + report = _run(client, table) + + assert report.idle_bytes == 2_000_000_000 + + def test_a_knowledge_base_with_no_activity_signal_is_unmeasured_not_idle(self, table): + """What a corpus provisioned an hour ago looks like. Counting it as idle + would report every new knowledge base as abandoned.""" + _seed_record(table, "ast-new", aws_kb_id="KBA", storedBytes=9_000_000_000) + client = FakeBedrockAgent(knowledge_bases=[], tags={}) + + report = _run(client, table) + + assert report.unmeasured_idleness == 1 + assert report.idle_bytes == 0 + + def test_the_gauges_are_emitted_once_per_pass(self, table, monkeypatch): + emitted = [] + monkeypatch.setattr(rec, "emit_fleet_gauges", lambda **kw: emitted.append(kw)) + _seed_record(table, "ast-a", aws_kb_id="KBA", storedBytes=1_000_000_000) + client = FakeBedrockAgent(knowledge_bases=[], tags={}) + + _run(client, table) + + assert len(emitted) == 1 + assert emitted[0]["kb_count"] == 1 + assert emitted[0]["stored_bytes"] == 1_000_000_000 + + def test_an_idleness_failure_does_not_end_the_pass(self, table, monkeypatch): + """A gauge is never worth a reconciliation.""" + _seed_record(table, "ast-a", aws_kb_id="KBA", storedBytes=1_000_000_000) + monkeypatch.setattr( + "apis.shared.kb_backend.idleness.idle_days", + lambda *a, **k: (_ for _ in ()).throw(RuntimeError("boom")), + ) + client = FakeBedrockAgent(knowledge_bases=[], tags={}) + + report = _run(client, table) + + assert report.records == 1 + assert report.stored_bytes == 1_000_000_000 + + def test_the_idle_threshold_is_resolved_at_call_time(self, monkeypatch): + from apis.shared.kb_backend.metrics import IDLE_THRESHOLD_DAYS + + monkeypatch.delenv("KB_IDLE_THRESHOLD_DAYS", raising=False) + assert rec.idle_threshold_days() == IDLE_THRESHOLD_DAYS + monkeypatch.setenv("KB_IDLE_THRESHOLD_DAYS", "7") + assert rec.idle_threshold_days() == 7 diff --git a/backend/tests/property/test_pbt_kb_byte_cap.py b/backend/tests/property/test_pbt_kb_byte_cap.py new file mode 100644 index 000000000..2a9538b8d --- /dev/null +++ b/backend/tests/property/test_pbt_kb_byte_cap.py @@ -0,0 +1,328 @@ +"""Property-based tests for byte cap accounting. + +Feature: managed-kb-migration + +**Property 5: the cap is never exceeded, under any interleaving.** + +Managed storage costs $5.00/GB-month, so the cap is the only thing standing between +the measured ~$169/month fleet cost and the ~$15,000/month that unbounded uploads +would permit. "Usually holds" is not a cap. + +The property is asserted against real DynamoDB semantics (via moto) rather than +against a Python model of them, because the entire correctness argument rests on +one specific database behaviour: that a conditional ``ADD`` is atomic. A test that +simulated the arithmetic in Python would pass just as happily against a +read-then-write implementation, which is precisely the broken version. + +Why the accumulator matters +--------------------------- +DynamoDB cannot do arithmetic inside a condition expression — verified, it fails to +parse. So the guard compares a single ``totalBytes`` accumulator against a literal +computed before the call (``cap - n``). The invariant +``totalBytes == storedBytes + reservedBytes`` is what makes that sound, and several +tests below assert it directly rather than only checking the total. + +Validates: Requirements 12.4, 12.5, 12.6, 24.7. +""" + +import boto3 +import pytest +from hypothesis import HealthCheck, given, settings, strategies as st +from moto import mock_aws + +from apis.shared.kb_backend import byte_cap as bc +from apis.shared.kb_backend.records import kb_pk, kb_sk + +REGION = "us-east-1" +TABLE = "test-byte-cap" +ASSISTANT_ID = "ast-cap01" +APP_KB_ID = ASSISTANT_ID +CAP = 1000 + +# --------------------------------------------------------------------------- +# Strategies +# --------------------------------------------------------------------------- + +#: Reservation sizes, including 0 (a no-op) and sizes larger than the whole cap. +st_size = st.integers(min_value=0, max_value=CAP + 500) + +#: An arbitrary sequence of reservations. Length and sizes both vary so the +#: sequence sometimes fits entirely, sometimes overruns partway, and sometimes +#: overruns on the very first item. +st_sequence = st.lists(st_size, min_size=1, max_size=15) + + +@pytest.fixture() +def table(monkeypatch): + monkeypatch.setenv("AWS_DEFAULT_REGION", REGION) + monkeypatch.setenv("AWS_ACCESS_KEY_ID", "testing") + monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "testing") + monkeypatch.setenv("AWS_SESSION_TOKEN", "testing") + monkeypatch.setenv("DYNAMODB_ASSISTANTS_TABLE_NAME", TABLE) + + with mock_aws(): + ddb = boto3.client("dynamodb", region_name=REGION) + ddb.create_table( + TableName=TABLE, + KeySchema=[ + {"AttributeName": "PK", "KeyType": "HASH"}, + {"AttributeName": "SK", "KeyType": "RANGE"}, + ], + AttributeDefinitions=[ + {"AttributeName": "PK", "AttributeType": "S"}, + {"AttributeName": "SK", "AttributeType": "S"}, + ], + BillingMode="PAY_PER_REQUEST", + ) + t = boto3.resource("dynamodb", region_name=REGION).Table(TABLE) + t.put_item(Item={"PK": kb_pk(ASSISTANT_ID), "SK": kb_sk(APP_KB_ID)}) + yield t + + +def _counters(table): + item = table.get_item(Key={"PK": kb_pk(ASSISTANT_ID), "SK": kb_sk(APP_KB_ID)})["Item"] + return ( + int(item.get("totalBytes", 0)), + int(item.get("reservedBytes", 0)), + int(item.get("storedBytes", 0)), + ) + + +def _reset(table): + table.put_item(Item={"PK": kb_pk(ASSISTANT_ID), "SK": kb_sk(APP_KB_ID)}) + + +# --------------------------------------------------------------------------- +# The cap holds +# --------------------------------------------------------------------------- +@given(sizes=st_sequence) +@settings(max_examples=60, deadline=None, suppress_health_check=[HealthCheck.function_scoped_fixture]) +def test_the_cap_is_never_exceeded(table, sizes): + """However the sequence interleaves, the accumulator never passes the cap.""" + _reset(table) + accepted = [] + for n in sizes: + try: + bc.reserve(ASSISTANT_ID, APP_KB_ID, n, CAP) + accepted.append(n) + except bc.ByteCapExceeded: + pass + + total, _, _ = _counters(table) + assert total <= CAP, f"cap breached at {total} > {CAP}" + + total, reserved, _ = _counters(table) + assert total == sum(accepted) + assert reserved == sum(accepted) + + +@given(sizes=st_sequence) +@settings(max_examples=60, deadline=None, suppress_health_check=[HealthCheck.function_scoped_fixture]) +def test_the_accumulator_invariant_holds(table, sizes): + """totalBytes == storedBytes + reservedBytes, always. + + This is what makes comparing a single attribute a valid cap check. If the two + ever diverge the guard is measuring something that is not the owner's usage. + """ + _reset(table) + for n in sizes: + try: + bc.reserve(ASSISTANT_ID, APP_KB_ID, n, CAP) + # Commit half the time so both counters move. + if n % 2 == 0: + bc.commit(ASSISTANT_ID, APP_KB_ID, n) + except bc.ByteCapExceeded: + pass + + total, reserved, stored = _counters(table) + assert total == reserved + stored, f"{total} != {reserved} + {stored}" + + +@given(sizes=st.lists(st.integers(min_value=1, max_value=200), min_size=1, max_size=10)) +@settings(max_examples=60, deadline=None, suppress_health_check=[HealthCheck.function_scoped_fixture]) +def test_released_reservations_are_fully_returned(table, sizes): + """Release restores the allowance exactly. + + A release that returned less than it reserved would shrink the owner's cap on + every failed upload, presenting weeks later as "uploads stopped working" with + no failing request to point at. + """ + _reset(table) + for n in sizes: + bc.reserve(ASSISTANT_ID, APP_KB_ID, n, CAP) + bc.release(ASSISTANT_ID, APP_KB_ID, n) + + total, reserved, stored = _counters(table) + assert (total, reserved, stored) == (0, 0, 0) + + +@given(sizes=st.lists(st.integers(min_value=1, max_value=100), min_size=1, max_size=8)) +@settings(max_examples=60, deadline=None, suppress_health_check=[HealthCheck.function_scoped_fixture]) +def test_commit_does_not_double_count(table, sizes): + """Commit moves bytes; it must not add them again. + + Double-counting on commit would halve every owner's effective allowance, and it + would do so only for *successful* uploads — so the symptom would be that the + cap tightens the more correctly the system works. + """ + _reset(table) + for n in sizes: + bc.reserve(ASSISTANT_ID, APP_KB_ID, n, CAP) + bc.commit(ASSISTANT_ID, APP_KB_ID, n) + + total, reserved, stored = _counters(table) + assert total == sum(sizes) + assert stored == sum(sizes) + assert reserved == 0 + + +# --------------------------------------------------------------------------- +# Boundary and rejection behaviour +# --------------------------------------------------------------------------- +def test_a_reservation_exactly_filling_the_cap_is_allowed(table): + """The cap is inclusive: exactly at the limit is within it.""" + _reset(table) + bc.reserve(ASSISTANT_ID, APP_KB_ID, CAP, CAP) + assert _counters(table)[0] == CAP + + +def test_one_byte_over_is_rejected(table): + _reset(table) + with pytest.raises(bc.ByteCapExceeded): + bc.reserve(ASSISTANT_ID, APP_KB_ID, CAP + 1, CAP) + assert _counters(table)[0] == 0, "a rejected reservation must leave no trace" + + +def test_a_rejected_reservation_does_not_consume_allowance(table): + """The failed attempt must not partially apply. + + An ADD that landed before the condition was evaluated would leak allowance on + every rejection, so a user who hit the cap once could never upload again. + """ + _reset(table) + bc.reserve(ASSISTANT_ID, APP_KB_ID, 900, CAP) + with pytest.raises(bc.ByteCapExceeded): + bc.reserve(ASSISTANT_ID, APP_KB_ID, 200, CAP) + + total, reserved, _ = _counters(table) + assert (total, reserved) == (900, 900) + # And the remaining allowance is still usable. + bc.reserve(ASSISTANT_ID, APP_KB_ID, 100, CAP) + assert _counters(table)[0] == CAP + + +def test_zero_is_a_no_op(table): + _reset(table) + bc.reserve(ASSISTANT_ID, APP_KB_ID, 0, CAP) + assert _counters(table) == (0, 0, 0) + + +def test_a_negative_reservation_is_rejected(table): + """Otherwise 'reserving' a negative size would be a way to mint allowance.""" + _reset(table) + with pytest.raises(ValueError): + bc.reserve(ASSISTANT_ID, APP_KB_ID, -100, CAP) + + +def test_the_exception_carries_the_numbers_for_the_user(table): + """Requirement 12.12 wants a plain-language reason and an upgrade path, which + needs the figures, not just a failure.""" + _reset(table) + with pytest.raises(bc.ByteCapExceeded) as excinfo: + bc.reserve(ASSISTANT_ID, APP_KB_ID, CAP + 1, CAP) + assert excinfo.value.requested == CAP + 1 + assert excinfo.value.cap == CAP + + +# --------------------------------------------------------------------------- +# Migration snapshot (Requirement 12.11/12.12) +# --------------------------------------------------------------------------- +def test_a_snapshot_that_cannot_fit_is_rejected_up_front(table): + """The whole corpus is reserved before migration starts. + + Reserving per-document instead would let a migration run for an hour and stop + halfway, leaving a half-populated managed knowledge base behind. + """ + _reset(table) + with pytest.raises(bc.ByteCapExceeded): + bc.reserve_snapshot(ASSISTANT_ID, APP_KB_ID, CAP * 2, CAP) + assert _counters(table)[0] == 0, "a rejected migration must reserve nothing" + + +def test_a_snapshot_that_fits_reserves_the_whole_corpus(table): + _reset(table) + bc.reserve_snapshot(ASSISTANT_ID, APP_KB_ID, 800, CAP) + total, reserved, _ = _counters(table) + assert (total, reserved) == (800, 800) + + +def test_a_snapshot_is_rejected_when_existing_usage_leaves_no_room(table): + """The interesting case: the corpus fits an empty cap but not this owner's.""" + _reset(table) + bc.reserve(ASSISTANT_ID, APP_KB_ID, 700, CAP) + bc.commit(ASSISTANT_ID, APP_KB_ID, 700) + + with pytest.raises(bc.ByteCapExceeded): + bc.reserve_snapshot(ASSISTANT_ID, APP_KB_ID, 400, CAP) + + assert _counters(table)[0] == 700 + + +# --------------------------------------------------------------------------- +# Cap resolution +# --------------------------------------------------------------------------- +def test_the_default_cap_is_below_the_user_files_precedent(monkeypatch): + """100 MB, deliberately under the existing 1 GB user-files limit. + + At $5.00/GB-month that precedent would permit roughly $150,000/month across the + fleet — a number large enough that it is not really a limit. + """ + monkeypatch.delenv("MANAGED_KB_PER_OWNER_DEFAULT_BYTES", raising=False) + assert bc.per_owner_cap() == 100 * 1024 * 1024 + assert bc.per_owner_cap() < 1024 * 1024 * 1024 + + +def test_the_elevated_tier_is_larger_than_the_default(monkeypatch): + monkeypatch.delenv("MANAGED_KB_PER_OWNER_DEFAULT_BYTES", raising=False) + monkeypatch.delenv("MANAGED_KB_PER_OWNER_ELEVATED_BYTES", raising=False) + assert bc.per_owner_cap(elevated=True) > bc.per_owner_cap() + + +def test_caps_are_overridable_from_the_environment(monkeypatch): + monkeypatch.setenv("MANAGED_KB_PER_OWNER_DEFAULT_BYTES", "12345") + assert bc.per_owner_cap() == 12345 + + +def test_a_malformed_override_falls_back_rather_than_crashing(monkeypatch): + """A typo in an operator-set variable must not take retrieval down.""" + monkeypatch.setenv("MANAGED_KB_PER_OWNER_DEFAULT_BYTES", "not-a-number") + assert bc.per_owner_cap() == 100 * 1024 * 1024 + + +def test_the_per_kb_ceiling_is_below_the_elevated_owner_cap(monkeypatch): + """A single knowledge base must not be able to eat an entire elevated + allowance and starve the owner's others.""" + for var in ( + "MANAGED_KB_PER_KB_CEILING_BYTES", + "MANAGED_KB_PER_OWNER_ELEVATED_BYTES", + ): + monkeypatch.delenv(var, raising=False) + assert bc.per_kb_ceiling() < bc.per_owner_cap(elevated=True) + + +# --------------------------------------------------------------------------- +# Sizing authority +# --------------------------------------------------------------------------- +def test_size_comes_from_s3_not_from_the_caller(monkeypatch): + """A client-reported size is an input, and an input that can lower its own + cost is not a measurement.""" + from unittest.mock import MagicMock, patch + + monkeypatch.setenv("AWS_DEFAULT_REGION", REGION) + with patch("boto3.client") as client: + s3 = MagicMock() + s3.head_object.return_value = {"ContentLength": 4242} + client.return_value = s3 + + assert bc.object_size_bytes("bucket", "key") == 4242 + s3.head_object.assert_called_once_with(Bucket="bucket", Key="key") diff --git a/backend/tests/property/test_pbt_kb_engine_resolution.py b/backend/tests/property/test_pbt_kb_engine_resolution.py new file mode 100644 index 000000000..8f1ccf2be --- /dev/null +++ b/backend/tests/property/test_pbt_kb_engine_resolution.py @@ -0,0 +1,205 @@ +"""Property-based tests for engine resolution by absence. + +Feature: managed-kb-migration + +**Property 1: absence means legacy.** + +This is the invariant the whole migration rests on. Every knowledge base that +existed before this feature carries no ``retrievalEngine`` attribute, and must +resolve to the legacy backend on that basis alone. Two consequences follow, and +both are why this file exists: + +* **No backfill.** 1,692 ``DOC#`` records and their knowledge bases are already + correct without being touched. A migration that had to stamp a value on each + one would be a data migration in its own right, with its own failure modes. +* **Rollback is a pointer flip.** Rolling back ``REMOVE``s the attribute, + restoring the original shape exactly. A rolled-back record is + indistinguishable from one that never migrated. + +Both consequences evaporate the moment any code path writes the literal +``"s3vectors"`` onto a record that did not already carry it. That write would +look harmless, pass a naive test, and convert every future rollback into a +rewrite. The second half of this file exists to make that specific mistake fail +loudly. + +Validates: Requirements 1.6, 1.7, 6.6. +""" + +import json +from typing import Any, Dict, List + +import pytest +from hypothesis import given, settings, strategies as st + +from apis.shared.kb_backend import records as r + +# --------------------------------------------------------------------------- +# Shared Hypothesis strategies +# --------------------------------------------------------------------------- + +st_attribute_name = st.text( + alphabet="abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789_", + min_size=1, + max_size=24, +) + +st_attribute_value = st.one_of( + st.text(max_size=40), + st.integers(min_value=-1000, max_value=10**9), + st.booleans(), + st.none(), + st.lists(st.text(max_size=10), max_size=4), + st.dictionaries(st.text(min_size=1, max_size=8), st.integers(), max_size=3), +) + +#: An arbitrary stored item that carries no opinion about its engine. Extra keys +#: are deliberately unconstrained: real records accumulate attributes over time +#: and resolution must not depend on which ones happen to be present. +st_item_without_engine = st.dictionaries( + st_attribute_name, st_attribute_value, max_size=12 +).map(lambda d: {k: v for k, v in d.items() if k != "retrievalEngine"}) + +#: Anything that is not the one value we accept. Includes the legacy literal +#: itself: even if some historical record somehow carried "s3vectors", it must +#: resolve to legacy, which it does — but it must never be *written*. +st_non_managed_engine = st.one_of( + st.just(r.ENGINE_LEGACY), + st.just(""), + st.just("Managed"), + st.just("MANAGED"), + st.just("managed "), + st.text(max_size=20).filter(lambda s: s != r.ENGINE_MANAGED), +) + + +# --------------------------------------------------------------------------- +# Property 1: absence means legacy +# --------------------------------------------------------------------------- +@given(item=st_item_without_engine) +@settings(max_examples=200) +def test_any_record_without_the_attribute_resolves_to_legacy(item): + """No matter what else the record contains, a missing engine means legacy.""" + assert "retrievalEngine" not in item + assert r.resolve_engine(item) == r.ENGINE_LEGACY + + +@given(item=st_item_without_engine, engine=st_non_managed_engine) +@settings(max_examples=200) +def test_only_the_exact_managed_literal_selects_the_managed_backend(item, engine): + """Resolution is exact-match, so a typo or casing slip fails safe. + + Failing safe matters asymmetrically here: resolving to legacy when it should + be managed serves slightly worse answers, while resolving to managed when the + record is not really migrated queries a knowledge base that may not exist. + """ + item["retrievalEngine"] = engine + assert r.resolve_engine(item) == r.ENGINE_LEGACY + + +@given(item=st_item_without_engine) +@settings(max_examples=100) +def test_the_managed_literal_selects_managed(item): + """The positive case, so the tests above cannot pass by always returning legacy.""" + item["retrievalEngine"] = r.ENGINE_MANAGED + assert r.resolve_engine(item) == r.ENGINE_MANAGED + + +@pytest.mark.parametrize("empty", [None, {}]) +def test_a_missing_record_resolves_to_legacy(empty): + """Absence of the whole record is an answer too, not an error. + + A knowledge base with no KB_Record is every knowledge base today. + """ + assert r.resolve_engine(empty) == r.ENGINE_LEGACY + + +@given(item=st_item_without_engine) +@settings(max_examples=100) +def test_resolution_does_not_mutate_the_item(item): + """Resolution is a read. A resolver that defaulted the attribute *in place* + would silently create the backfill this design exists to avoid.""" + before = json.dumps(item, sort_keys=True, default=str) + r.resolve_engine(item) + assert json.dumps(item, sort_keys=True, default=str) == before + + +# --------------------------------------------------------------------------- +# Property 1, second half: the legacy literal is never written +# --------------------------------------------------------------------------- +class _RecordingTable: + """Captures write payloads instead of performing them. + + Used rather than moto because the assertion here is about what the module + *sends*, not about what DynamoDB does with it — and because it lets a single + test observe every transition without needing each one's preconditions to + hold. + """ + + def __init__(self) -> None: + self.calls: List[Dict[str, Any]] = [] + + def put_item(self, **kwargs): + self.calls.append(kwargs) + return {} + + def update_item(self, **kwargs): + self.calls.append(kwargs) + return {} + + def serialized(self) -> str: + return json.dumps(self.calls, sort_keys=True, default=str) + + +@pytest.fixture() +def recorder(monkeypatch): + table = _RecordingTable() + monkeypatch.setattr(r, "_table", lambda: table) + return table + + +def _drive_every_write(table_unused) -> None: + """Invoke every write path in the module once.""" + r.create_provisioning( + "ast-1", r.KbRecord(app_kb_id="ast-1", owner_user_id="opaque-owner") + ) + r.attach_aws_ids("ast-1", "ast-1", "kb-1", "ds-1", "2026-08-24T12:00:00Z") + r.promote_engine("ast-1", "ast-1", 0, "2026-08-24T12:00:00Z") + r.rollback_engine("ast-1", "ast-1", "2026-08-24T12:00:00Z") + r.acquire_lease("ast-1", "ast-1", "2026-08-24T13:00:00Z", "2026-08-24T12:00:00Z") + for state in (r.SHADOW, r.VERIFY, r.PROMOTE): + r.set_migration_state("ast-1", "ast-1", state, 0, due_at="2026-08-24T12:00:00Z") + for state in (r.RETAIN, r.MIGRATION_FAILED): + r.set_migration_state("ast-1", "ast-1", state, 0, error="a reason") + + +def test_no_write_path_ever_persists_the_legacy_literal(recorder): + """The load-bearing negative. Every write in the module, inspected. + + If this fails, someone has made legacy an explicitly stored value. The + feature would still appear to work, and the next rollback would stop being a + pointer flip. + """ + _drive_every_write(recorder) + assert recorder.calls, "no writes captured; the fixture is not wired" + + payload = recorder.serialized() + assert r.ENGINE_LEGACY not in payload, ( + f"a write path persists the legacy literal {r.ENGINE_LEGACY!r}; " + "absence must remain the only representation of legacy" + ) + + +def test_rollback_removes_the_attribute_rather_than_setting_it(recorder): + """Rollback must restore the original shape, not write a value.""" + r.rollback_engine("ast-1", "ast-1", "2026-08-24T12:00:00Z") + expression = recorder.calls[0]["UpdateExpression"] + assert "REMOVE retrievalEngine" in expression + assert "retrievalEngine = " not in expression + + +def test_the_only_engine_value_ever_written_is_managed(recorder): + """Complements the negative test: promotion writes exactly one engine value.""" + r.promote_engine("ast-1", "ast-1", 0, "2026-08-24T12:00:00Z") + values = recorder.calls[0]["ExpressionAttributeValues"] + engine_values = [v for v in values.values() if v in (r.ENGINE_MANAGED, r.ENGINE_LEGACY)] + assert engine_values == [r.ENGINE_MANAGED] diff --git a/backend/tests/property/test_pbt_kb_migration_convergence.py b/backend/tests/property/test_pbt_kb_migration_convergence.py new file mode 100644 index 000000000..f8b3adbd4 --- /dev/null +++ b/backend/tests/property/test_pbt_kb_migration_convergence.py @@ -0,0 +1,489 @@ +""" +Property-based tests for migration convergence. + +**Property 6: an interrupted migration converges without duplication** + +For any interruption point in the state machine, a resumed run reaches the same +terminal state, creates exactly **one** knowledge base, promotes exactly **once**, +and leaves each document in the corpus exactly once. + +What "without duplication" can and cannot mean +---------------------------------------------- +A worker can die between a successful ``IngestKnowledgeBaseDocuments`` and the +DynamoDB write that records it, and no transaction spans Bedrock and DynamoDB. So +"each document is ingested at most once" is not achievable, and asserting it would +be asserting something false. Two things *are* achievable, and both are asserted: + +* **Each document appears in the corpus exactly once**, because + ``customDocumentIdentifier`` is the platform document id and a re-ingest + therefore replaces. A migration that derived its own identifier would fail here. +* **Redundant re-ingests are bounded by one batch** — the size of the crash window. + That bound is what proves progress is persisted *as the migration proceeds* + rather than only at the end. It is not a theoretical distinction: this test + initially failed because the completed-document set lived inside the + ``migrationProgress`` map, which a later write replaced wholesale, so a crash + near the end of a 25-document corpus re-ingested all 25. + +Why this needs to be a property rather than a set of cases +---------------------------------------------------------- +The interruption points are not a short list. A migration can be cut off between +any two of: reserving bytes, creating the knowledge base, creating the data source, +writing the AWS identifiers back, ingesting each individual batch, recording +progress, promoting, and stamping the retention window. Enumerating them by hand +produces the cases somebody thought of, and the ones that matter are the ones +nobody did — this feature has already been bitten by a crash window between an AWS +create and the database write that records it. + +So the interruption index is a hypothesis input over the sequence of effects, and +the invariants are asserted after replaying from the start, which is what a retry +actually does. + +The model is deliberately in-memory +----------------------------------- +A fake DynamoDB and a fake Bedrock, both of which enforce the properties that make +convergence possible rather than assuming them: + +* ``create_knowledge_base`` is deduplicated by ``clientToken`` — which is how AWS + behaves, and the reason the worker persists the token before calling AWS. +* ``ingest_knowledge_base_documents`` records every ``customDocumentIdentifier`` + it is handed, so "at most once" is measured over the whole replay rather than + per attempt. + +A test against real clients could not interrupt at a chosen point, and a test with +no model at all would assert only that the code does not raise. + +Feature: managed-kb-migration +**Validates: Requirements 15.9, 15.10, 15.13, 7.4** +""" + +from typing import Any, Dict, List, Optional, Set + +import pytest +from hypothesis import HealthCheck, given, settings, strategies as st + +# --------------------------------------------------------------------------- +# The model +# --------------------------------------------------------------------------- + + +class Interrupted(Exception): + """The simulated crash. Raised at the chosen effect index.""" + + +class Clock: + """Counts effects and raises at the interruption point. + + Every externally-visible side effect passes through :meth:`tick`, so the + interruption index addresses effects rather than lines of code — the unit a + crash actually lands between. + """ + + def __init__(self, interrupt_at: Optional[int] = None): + self.count = 0 + self.interrupt_at = interrupt_at + self.log: List[str] = [] + + def tick(self, what: str) -> None: + self.count += 1 + self.log.append(what) + if self.interrupt_at is not None and self.count == self.interrupt_at: + raise Interrupted(f"crashed at effect {self.count}: {what}") + + +class FakeAws: + """Bedrock's idempotency, modelled rather than assumed.""" + + def __init__(self, clock: Clock): + self.clock = clock + self.kbs_by_token: Dict[str, str] = {} + self.data_sources: Dict[str, str] = {} + #: Every ingest ever accepted, across every attempt. The duplication + #: invariant is measured here. + self.ingest_log: List[str] = [] + #: document_id -> times written. Distinct keys are the corpus; the counts + #: are the redundant work a resume did. + self.corpus: Dict[str, int] = {} + self.next_id = 0 + + def create_knowledge_base(self, client_token: str) -> str: + # Deduplicated by token: this is what makes a retried create safe, and the + # reason the record persists the token *before* the AWS call. + if client_token in self.kbs_by_token: + return self.kbs_by_token[client_token] + self.clock.tick("CreateKnowledgeBase") + self.next_id += 1 + kb_id = f"KB{self.next_id:04d}" + self.kbs_by_token[client_token] = kb_id + return kb_id + + def create_data_source(self, kb_id: str, client_token: str) -> str: + if client_token in self.data_sources: + return self.data_sources[client_token] + self.clock.tick("CreateDataSource") + ds_id = f"DS-{kb_id}" + self.data_sources[client_token] = ds_id + return ds_id + + def ingest(self, document_ids: List[str]) -> None: + self.clock.tick(f"Ingest({','.join(document_ids)})") + self.ingest_log.extend(document_ids) + for document_id in document_ids: + # ``customDocumentIdentifier`` is the platform document id, so a + # re-ingest *replaces* rather than appends. Modelled because it is what + # makes the unavoidable crash window survivable: a worker can die + # between a successful Ingest and the bookkeeping write, and no + # transaction spans Bedrock and DynamoDB. + self.corpus[document_id] = self.corpus.get(document_id, 0) + 1 + + @property + def corpus_document_count(self) -> int: + """Distinct documents in the knowledge base.""" + return len(self.corpus) + + @property + def knowledge_base_count(self) -> int: + return len(set(self.kbs_by_token.values())) + + +class FakeRecord: + """The KB_Record, with the conditional writes that matter.""" + + def __init__(self, clock: Clock, document_ids: List[str]): + self.clock = clock + self.item: Dict[str, Any] = {} + self.documents: Dict[str, str] = {d: "complete" for d in document_ids} + self.promotions = 0 + + # -- reads --------------------------------------------------------------- + def get(self) -> Dict[str, Any]: + return dict(self.item) + + def list_complete(self) -> List[str]: + return sorted(d for d, status in self.documents.items() if status == "complete") + + def status_of(self, document_id: str) -> Optional[str]: + return self.documents.get(document_id) + + # -- writes -------------------------------------------------------------- + def create_provisioning(self, client_token: str) -> None: + if self.item: + return # attribute_not_exists guard: the retry anchor already exists + self.clock.tick("CreateProvisioning") + self.item = { + "clientToken": client_token, + "provisioningState": "provisioning", + "migrationState": "shadow", + "migrationGeneration": 1, + "totalBytes": 0, + } + + def attach_ids(self, kb_id: str, ds_id: str) -> None: + if self.item.get("awsKbId"): + return + self.clock.tick("AttachAwsIds") + self.item["awsKbId"] = kb_id + self.item["awsDataSourceId"] = ds_id + self.item["provisioningState"] = "active" + + def reserve(self, total: int) -> None: + self.clock.tick("ReserveSnapshot") + self.item["totalBytes"] = total + + def set_progress(self, migrated: int, total: int, newly_done: List[str] = None) -> None: + self.clock.tick("SetProgress") + self.item["migrationProgress"] = {"migrated": migrated, "total": total} + if newly_done: + # ADD on a string set: additive, and a *separate attribute* from the + # progress map this write replaces. Modelled that way because the + # first version of this test kept the completed set inside the map, + # the map got overwritten, and the resumed run re-ingested a corpus + # it had already finished. The worker had the same bug. + existing = set(self.item.get("migratedDocIds") or ()) + self.item["migratedDocIds"] = existing | set(newly_done) + + def add_done(self, document_ids: List[str]) -> None: + """The per-batch ADD, which is what survives a crash between batches.""" + if not document_ids: + return + self.clock.tick(f"AddDone({','.join(document_ids)})") + existing = set(self.item.get("migratedDocIds") or ()) + self.item["migratedDocIds"] = existing | set(document_ids) + + def set_state(self, new_state: str, expected: Optional[Set[str]] = None) -> bool: + if expected is not None and self.item.get("migrationState") not in expected: + return False + self.clock.tick(f"SetState({new_state})") + self.item["migrationState"] = new_state + return True + + def promote(self) -> bool: + progress = self.item.get("migrationProgress") or {} + if self.item.get("retrievalEngine"): + # attribute_not_exists(retrievalEngine): already promoted. Every other + # guard stays true after a successful promotion, so without this one a + # crash between the promotion and the state transition promotes twice — + # and two concurrent workers both succeed. + return False + if self.item.get("migrationState") != "promote": + return False + if progress.get("migrated") != progress.get("total"): + # Requirement 15.9 in the model: convergence is part of the condition, + # not a separate check somebody could forget to call. + return False + self.clock.tick("Promote") + self.promotions += 1 + self.item["retrievalEngine"] = "managed" + return True + + +BATCH = 10 + + +def run_migration(record: FakeRecord, aws: FakeAws, clock: Clock) -> str: + """Replay the whole machine from the start. Idempotent by construction. + + This mirrors the real worker's ordering exactly, and the ordering is the thing + under test: the record is written *before* AWS is called, the persisted token is + reused on resume, and every document's status is re-read immediately before it + is ingested. + """ + token = "kb-token-fixed-length-padding-000000" + + # shadow + record.create_provisioning(token) + if not record.item.get("totalBytes"): + record.reserve(len(record.list_complete()) * 1024) + + kb_id = record.item.get("awsKbId") or aws.create_knowledge_base( + record.item.get("clientToken") or token + ) + ds_id = record.item.get("awsDataSourceId") or aws.create_data_source(kb_id, token) + record.attach_ids(kb_id, ds_id) + + already = set(record.item.get("migratedDocIds") or ()) + snapshot = record.list_complete() + pending = [d for d in snapshot if d not in already] + + migrated = set(already) + for start in range(0, len(pending), BATCH): + batch = [ + d + for d in pending[start : start + BATCH] + # Requirement 16.4: re-read immediately before ingesting. + if record.status_of(d) == "complete" + ] + if not batch: + continue + aws.ingest(batch) + migrated.update(batch) + # Persisted per batch, so a crash between batches loses only the batch in + # flight rather than the whole run's progress. + record.add_done(batch) + + # catch-up until quiet + passes = 0 + while passes < 5: + passes += 1 + new = [d for d in record.list_complete() if d not in migrated] + if not new: + break + aws.ingest(new) + migrated.update(new) + record.add_done(new) + + record.set_progress(len(migrated), len(record.list_complete())) + record.set_state("verify", {"shadow"}) + + # verify + record.set_state("promote", {"verify"}) + + # promote. An already-promoted record still finishes: the promotion write is + # guarded on the engine attribute being absent, so a resume after a crash + # between the promotion and the state transition must continue to `retain` + # rather than treat the refusal as a failure. + if record.promote() or record.item.get("retrievalEngine") == "managed": + record.set_state("retain", {"promote"}) + + return record.item.get("migrationState", "") + + +# --------------------------------------------------------------------------- +# Strategies +# --------------------------------------------------------------------------- + +st_document_ids = st.lists( + st.text(alphabet="abcdefghijklmnopqrstuvwxyz0123456789", min_size=1, max_size=6), + min_size=1, + max_size=25, + unique=True, +) + +#: Effects, not lines. A migration of 25 documents produces roughly a dozen; the +#: upper bound is generous so an index past the end simply means "not interrupted", +#: which is a case worth generating too. +st_interrupt_at = st.integers(min_value=1, max_value=30) + + +# --------------------------------------------------------------------------- +# The property +# --------------------------------------------------------------------------- + + +@settings(max_examples=200, deadline=None, suppress_health_check=[HealthCheck.too_slow]) +@given(document_ids=st_document_ids, interrupt_at=st_interrupt_at) +def test_an_interrupted_migration_converges_without_duplication(document_ids, interrupt_at): + """The whole property, in one test. + + Run once with a crash injected at ``interrupt_at``; then run again from the + start, as a retry does. Assert the terminal state, exactly one knowledge base, + and each document ingested at most once across **both** runs. + """ + clock = Clock(interrupt_at=interrupt_at) + aws = FakeAws(clock) + record = FakeRecord(clock, document_ids) + + try: + run_migration(record, aws, clock) + except Interrupted: + pass + + # The retry. No interruption this time. + clock.interrupt_at = None + final_state = run_migration(record, aws, clock) + + assert final_state == "retain", ( + f"a resumed migration did not converge: state={final_state!r}, " + f"effects={clock.log}" + ) + + assert aws.knowledge_base_count == 1, ( + f"{aws.knowledge_base_count} knowledge bases were created; the persisted " + f"clientToken is not deduplicating the retried create" + ) + + counts: Dict[str, int] = {} + for document_id in aws.ingest_log: + counts[document_id] = counts.get(document_id, 0) + 1 + redundant = sum(n - 1 for n in counts.values()) + + # Each document appears in the corpus exactly once. This is the invariant that + # actually matters, and it is real rather than tautological: it holds because + # `customDocumentIdentifier` is the platform document id, so a re-ingest + # replaces. A migration that derived its own identifier would fail here. + assert aws.corpus_document_count == len(document_ids) + assert all(document_id in aws.corpus for document_id in document_ids) + + # Redundant re-ingests are bounded by one batch: the crash window between a + # successful Ingest and the write that records it. No transaction spans Bedrock + # and DynamoDB, so that window cannot be closed — but it can be *bounded*, and + # the bound is what proves progress is persisted per batch. Before the + # completed-document set was persisted, a crash near the end of a 25-document + # corpus re-ingested all 25; this assertion is what caught that. + assert redundant <= BATCH, ( + f"{redundant} redundant ingests after one interruption, which is more than " + f"the single batch that can be in flight; progress is not being persisted " + f"as the migration proceeds. effects={clock.log}" + ) + + assert set(aws.ingest_log) == set(document_ids), ( + "the resumed migration did not end up with every document" + ) + + assert record.promotions == 1, ( + f"promotion happened {record.promotions} times; the conditional write is " + f"not the single cutover" + ) + + +@settings(max_examples=100, deadline=None) +@given(document_ids=st_document_ids, delete_index=st.integers(min_value=0, max_value=24)) +def test_a_document_deleted_mid_migration_is_never_ingested(document_ids, delete_index): + """Requirements 16.4, 16.5, as a property over which document is deleted. + + The deletion lands after the snapshot is taken and before the document's turn + comes, which is the only window in which resurrection is possible. + """ + clock = Clock() + aws = FakeAws(clock) + record = FakeRecord(clock, document_ids) + + victim = document_ids[delete_index % len(document_ids)] + + original_status_of = record.status_of + + def _status_with_deletion(document_id: str): + if document_id == victim: + return None + return original_status_of(document_id) + + record.status_of = _status_with_deletion + record.documents.pop(victim) + + run_migration(record, aws, clock) + + assert victim not in aws.ingest_log, ( + f"document {victim!r} was deleted mid-migration and still reached the " + f"managed corpus" + ) + + +@settings(max_examples=100, deadline=None) +@given(document_ids=st_document_ids) +def test_promotion_is_refused_until_catch_up_converges(document_ids): + """Requirement 15.9, asserted through the promotion condition itself. + + Progress is deliberately left short of the total, as an unconverged catch-up + leaves it. Promotion must be refused — and refused by the condition, so no + caller can reach past it. + """ + clock = Clock() + record = FakeRecord(clock, document_ids) + + record.item = { + "migrationState": "promote", + "migrationProgress": {"migrated": max(len(document_ids) - 1, 0), "total": len(document_ids)}, + } + + assert record.promote() is False + assert record.promotions == 0 + assert "retrievalEngine" not in record.item + + +@settings(max_examples=50, deadline=None) +@given(document_ids=st_document_ids) +def test_only_one_of_two_concurrent_promotions_wins(document_ids): + """Requirement 15.10. The second attempt sees a record no longer in ``promote`` + and is refused, which is what the real conditional write does.""" + clock = Clock() + record = FakeRecord(clock, document_ids) + + total = len(document_ids) + record.item = { + "migrationState": "promote", + "migrationProgress": {"migrated": total, "total": total}, + } + + first = record.promote() + record.set_state("retain", {"promote"}) + second = record.promote() + + assert first is True + assert second is False + assert record.promotions == 1 + + +def test_the_model_can_actually_be_interrupted(): + """Guards the guard. + + If ``Clock.tick`` stopped raising, every property above would pass while + testing nothing but the happy path. So assert that some interruption index + genuinely prevents convergence on the first run. + """ + clock = Clock(interrupt_at=1) + aws = FakeAws(clock) + record = FakeRecord(clock, ["d1", "d2"]) + + with pytest.raises(Interrupted): + run_migration(record, aws, clock) + + assert record.item.get("migrationState") != "retain" diff --git a/backend/tests/property/test_pbt_kb_query_clamp.py b/backend/tests/property/test_pbt_kb_query_clamp.py new file mode 100644 index 000000000..a61ca0a4d --- /dev/null +++ b/backend/tests/property/test_pbt_kb_query_clamp.py @@ -0,0 +1,245 @@ +"""Property-based tests for the retrieval query clamp. + +Feature: managed-kb-migration + +**Property 3: the clamp is total and non-throwing.** + +Managed Knowledge Base rejects a ``Retrieve`` query over 10,000 characters +outright, and the quota is not adjustable. So the clamp sits on a request path +where the only acceptable behaviours are "shortened" or "unchanged" — never +"raised". A clamp that threw would convert a fixable input into a failed chat +turn, which is strictly worse than answering a slightly truncated question. + +"Total" is the load-bearing word: *every* input must map to an output, including +the awkward ones. The strategies below deliberately include empty strings, strings +made entirely of astral-plane characters, and lengths sitting exactly on the +boundary, because those are where a length check written against the wrong unit or +with an off-by-one starts returning 10,001 characters to an API that rejects +10,001 characters. + +Validates: Requirements 4.1, 4.3, 4.4. +""" + +from unittest.mock import patch + +import pytest +from hypothesis import given, settings, strategies as st + +from apis.shared.assistants.kb_access import granted +from apis.shared.kb_backend.query_guard import MAX_QUERY_CHARS, clamp_query + +# --------------------------------------------------------------------------- +# Strategies +# --------------------------------------------------------------------------- + +#: Any text at all, including empty and including characters that are one code +#: point but more than one byte — the clamp counts characters, and a byte-based +#: implementation would pass a naive ASCII-only test. +st_any_text = st.text(max_size=200) + +st_long_text = st.text(min_size=1, max_size=50).map( + lambda s: s * (MAX_QUERY_CHARS // max(len(s), 1) + 2) +) + +st_multibyte_text = st.text( + alphabet=st.characters(min_codepoint=0x1F300, max_codepoint=0x1F5FF), + min_size=1, + max_size=40, +).map(lambda s: s * (MAX_QUERY_CHARS // max(len(s), 1) + 2)) + +#: Lengths straddling the cap, where off-by-one errors live. +st_boundary_length = st.integers( + min_value=MAX_QUERY_CHARS - 2, max_value=MAX_QUERY_CHARS + 2 +) + + +# --------------------------------------------------------------------------- +# Totality and the cap +# --------------------------------------------------------------------------- +@given(query=st_any_text) +@settings(max_examples=200) +def test_short_queries_pass_through_unchanged(query): + """Below the cap the clamp must be the identity, not a normalizer. + + Anything else would silently change what users are asking. + """ + with patch("apis.shared.kb_backend.query_guard.emit_count"): + result, truncated = clamp_query(query) + assert result == query + assert truncated is False + + +@given(query=st.one_of(st_long_text, st_multibyte_text)) +@settings(max_examples=100) +def test_output_never_exceeds_the_cap(query): + """The whole point: the value handed to the backend always fits.""" + with patch("apis.shared.kb_backend.query_guard.emit_count"): + result, truncated = clamp_query(query) + assert len(result) <= MAX_QUERY_CHARS + assert truncated is True + + +@given(length=st_boundary_length) +@settings(max_examples=50) +def test_the_boundary_is_inclusive(length): + """Exactly MAX_QUERY_CHARS is allowed; one more is not. + + Managed KB accepts 10,000 and rejects 10,001, so an off-by-one here is a + request error rather than a shorter answer. + """ + with patch("apis.shared.kb_backend.query_guard.emit_count"): + result, truncated = clamp_query("x" * length) + + assert len(result) == min(length, MAX_QUERY_CHARS) + assert truncated == (length > MAX_QUERY_CHARS) + + +@given(query=st.one_of(st_any_text, st_long_text, st_multibyte_text)) +@settings(max_examples=200) +def test_the_clamp_never_raises(query): + """Totality. A raise here would turn a long question into a failed chat turn.""" + with patch("apis.shared.kb_backend.query_guard.emit_count"): + try: + clamp_query(query) + except Exception as exc: # pragma: no cover - the assertion is the point + pytest.fail(f"clamp_query raised {type(exc).__name__}: {exc}") + + +@given(query=st_long_text) +@settings(max_examples=50) +def test_truncation_keeps_the_head(query): + """Keep the beginning: for a natural-language query that is where the intent + is. Head-truncating would change the question rather than shorten it.""" + with patch("apis.shared.kb_backend.query_guard.emit_count"): + result, _ = clamp_query(query) + assert query.startswith(result) + + +@given(query=st_long_text) +@settings(max_examples=50) +def test_the_clamp_is_idempotent(query): + """Clamping twice equals clamping once, and the second pass reports no + truncation — so a retry does not double-count the metric.""" + with patch("apis.shared.kb_backend.query_guard.emit_count"): + once, first = clamp_query(query) + twice, second = clamp_query(once) + assert twice == once + assert first is True + assert second is False + + +# --------------------------------------------------------------------------- +# The truncation signal +# --------------------------------------------------------------------------- +@given(length=st_boundary_length) +@settings(max_examples=50) +def test_the_metric_is_emitted_exactly_when_truncation_happened(length): + """The signal must track reality in both directions. + + A metric that over-reports trains operators to ignore it; one that + under-reports hides the fact that users are already sending queries the + managed backend would reject. + """ + with patch("apis.shared.kb_backend.query_guard.emit_count") as emit: + _, truncated = clamp_query("x" * length) + + assert truncated == (length > MAX_QUERY_CHARS) + assert emit.called == truncated + + +def test_a_metric_failure_does_not_break_the_clamp(): + """Observability is never control flow: if CloudWatch is down the query still + gets clamped and the search still runs.""" + with patch( + "apis.shared.kb_backend.query_guard.emit_count", + side_effect=RuntimeError("cloudwatch unavailable"), + ): + with pytest.raises(RuntimeError): + # Confirms the patch is actually wired, so the next assertion is not + # vacuous. + clamp_query("x" * (MAX_QUERY_CHARS + 1)) + + # emit_count's real implementation swallows its own failures, which is what + # makes the above impossible in production. Assert that contract directly. + from apis.shared.kb_backend.metrics import emit_count + + with patch("boto3.client", side_effect=RuntimeError("no credentials")): + emit_count("KbQueryClamped") # must not raise + + +@pytest.mark.parametrize("falsy", ["", None]) +def test_empty_input_is_handled_without_a_metric(falsy): + """An empty query is not a truncation.""" + with patch("apis.shared.kb_backend.query_guard.emit_count") as emit: + result, truncated = clamp_query(falsy) + assert result == "" + assert truncated is False + emit.assert_not_called() + + +# --------------------------------------------------------------------------- +# Guards the properties above cannot provide +# +# Every test above refers to MAX_QUERY_CHARS symbolically, so all of them follow +# the constant wherever it goes — raise it to 32,000 and they all still pass while +# the managed backend starts rejecting requests. These three assertions were added +# after mutation testing showed exactly that: three separate mutations survived a +# suite that looked thorough. +# --------------------------------------------------------------------------- +def test_the_cap_is_the_literal_managed_kb_limit(): + """Pinned to 10,000 as a LITERAL, not to the constant. + + This is the one assertion in the file that cannot be satisfied by moving the + constant. 10,000 is Managed KB's `Retrieve` input quota and it is not + adjustable, so this number is a property of AWS, not a tuning knob. Raising it + does not buy longer queries; it buys rejected requests. + """ + assert MAX_QUERY_CHARS == 10_000 + + +@pytest.mark.asyncio +async def test_the_facade_actually_clamps_before_dispatch(): + """The clamp must be WIRED, not merely correct. + + Nothing else in this file would notice if the facade stopped calling + clamp_query: the unit-level properties would all still pass while every long + query went to the backend intact. Asserted by inspecting what the backend + actually received. + """ + from apis.shared.assistants import rag_service + + seen = {} + + class _RecordingBackend: + async def search(self, kb_ref, query, top_k=5): + seen["query"] = query + return [] + + with patch.object(rag_service, "resolve_backend", return_value=_RecordingBackend()), patch.object( + rag_service, "emit_count" + ), patch("apis.shared.kb_backend.query_guard.emit_count"): + await rag_service.search_assistant_knowledgebase_with_formatting( + "ast-1", + "x" * (MAX_QUERY_CHARS + 500), + access=granted("ast-1", "user-clamp", "owner"), + ) + + assert seen["query"] is not None + assert len(seen["query"]) == MAX_QUERY_CHARS, ( + "the facade dispatched an unclamped query; the clamp is dead code" + ) + + +def test_the_metric_namespace_is_not_a_reserved_aws_one(): + """CloudWatch rejects PutMetricData into any namespace beginning with "AWS". + + A reserved namespace would make every publish silently denied — the grant looks + correct, the code looks correct, and no metric ever arrives. The CDK grant + conditions on this same namespace, so the two must agree; this is the backend + half of that assertion. + """ + from apis.shared.kb_backend.metrics import metric_namespace + + ns = metric_namespace() + assert not ns.startswith("AWS"), f"{ns!r} is a reserved namespace; writes are rejected" + assert ns.endswith("/ManagedKb") diff --git a/backend/tests/property/test_pbt_kb_score_direction.py b/backend/tests/property/test_pbt_kb_score_direction.py new file mode 100644 index 000000000..62758ab4e --- /dev/null +++ b/backend/tests/property/test_pbt_kb_score_direction.py @@ -0,0 +1,299 @@ +""" +Property-based tests for score direction across knowledge base backends. + +**Property 2: ranking is backend-independent** + +For any list of chunks with distinct scores, both backends return the known-best +chunk first after adapter conversion, and the ``relevance`` values they attach +agree with the order they return. + +This is the only test in the suite that can catch a silent ranking inversion. +S3 Vectors reports cosine *distance* (lower is better); Managed KB reports +*relevance* (higher is better). If the legacy adapter forwards distance as +relevance, nothing raises: every request still succeeds, still returns five +chunks, and still logs "Found 5 relevant chunks". The only symptom is that the +worst passages are ranked best and answers quietly degrade. There is no error +path, so there is nothing else to assert on. + +Feature: managed-kb-migration +**Validates: Requirements 2.1, 2.2, 2.3, 2.4, 24.1** +""" + +from typing import Any, Dict, List +from unittest.mock import patch + +from hypothesis import given, settings, strategies as st + +from apis.shared.kb_backend.protocol import ( + DEFAULT_TOP_K, + Chunk, + KnowledgeBaseBackend, + distance_from_relevance, + relevance_from_distance, +) +from apis.shared.kb_backend.s3vectors_backend import S3VectorsBackend + +# --------------------------------------------------------------------------- +# Strategies +# --------------------------------------------------------------------------- + +# Cosine distance lives in [0, 2]. Distinct values only: the property is about +# strict ranking, and ties would make "the known-best chunk" ambiguous rather +# than wrong. +st_distances = st.lists( + st.floats(min_value=0.0, max_value=2.0, allow_nan=False, allow_infinity=False), + min_size=2, + max_size=8, + unique=True, +) + +# Bounded so that ``score + 1.0`` is genuinely a larger float. Near the top of +# the double range adding 1.0 is a no-op, which would make the pairwise +# comparison below vacuous rather than false. +st_any_score = st.floats( + min_value=-1e6, max_value=1e6, allow_nan=False, allow_infinity=False +) + + +# --------------------------------------------------------------------------- +# A managed backend stand-in +# --------------------------------------------------------------------------- + + +class FakeManagedBackend: + """A protocol-conforming backend that reports relevance natively. + + Stands in for ``managed_backend.ManagedKbBackend``, which task 8.3 builds. + The property under test is about score *direction* — a per-adapter concern + that is fully determined by whether the adapter converts or passes through — + so a stand-in that passes relevance through unchanged, exactly as Managed KB + requires (Requirement 2.3), exercises the property faithfully. Nothing here + depends on Bedrock's wire format. + """ + + def __init__(self, results: List[Dict[str, Any]]): + # results: [{"document_id", "relevance", "text", "key"}], best first, + # which is the order Bedrock's Retrieve returns. + self._results = results + + async def search(self, kb_ref: str, query: str, top_k: int = DEFAULT_TOP_K) -> List[Chunk]: + return [ + Chunk( + text=result["text"], + # Pass-through. Managed already counts in the canonical direction. + relevance=result["relevance"], + document_id=result["document_id"], + metadata={"document_id": result["document_id"], "text": result["text"]}, + key=result["key"], + ) + for result in self._results + ] + + async def ingest(self, kb_ref: str, source) -> None: # pragma: no cover - unused here + raise NotImplementedError + + async def delete_document(self, kb_ref: str, document_id: str) -> None: # pragma: no cover + raise NotImplementedError + + +# --------------------------------------------------------------------------- +# Helpers +# --------------------------------------------------------------------------- + + +def _s3_vectors_response(distances: List[float]) -> Dict[str, Any]: + """Build an S3 Vectors query response, nearest-first as the API returns it.""" + return { + "vectors": [ + { + "key": f"doc-{index}#0", + "distance": distance, + "metadata": {"document_id": f"doc-{index}", "text": f"passage {index}"}, + } + for index, distance in enumerate(sorted(distances)) + ] + } + + +async def _legacy_search(distances: List[float]) -> List[Chunk]: + response = _s3_vectors_response(distances) + with patch( + "apis.shared.embeddings.bedrock_embeddings.search_assistant_knowledgebase", + return_value=response, + ): + return await S3VectorsBackend().search("ast-1", "a query") + + +def _is_non_increasing(values: List[float]) -> bool: + return all(earlier >= later for earlier, later in zip(values, values[1:])) + + +# --------------------------------------------------------------------------- +# Property 2 +# --------------------------------------------------------------------------- + + +@given(distances=st_distances) +@settings(max_examples=200, deadline=None) +def test_legacy_backend_ranks_known_best_chunk_first(distances): + """ + **Validates: Requirements 2.1, 2.2, 2.4** + + The chunk with the *lowest* S3 Vectors distance is the known-best chunk. After + conversion it must be first in the returned list and must carry the *highest* + relevance. + + The relevance-ordering assertion is the one that catches an inversion. The + positional one does not on its own: the adapter preserves the index's order, + so a chunk stays first whatever score is stapled to it. Only the claim that + scores descend can detect that the numbers now disagree with the order. + """ + import asyncio + + chunks = asyncio.run(_legacy_search(distances)) + + best_distance = min(distances) + best_key = f"doc-{sorted(distances).index(best_distance)}#0" + + assert chunks[0].key == best_key, "known-best chunk is not first" + + relevances = [chunk.relevance for chunk in chunks] + assert _is_non_increasing(relevances), ( + f"relevance must descend with rank, got {relevances}. " + f"A rising sequence means distance was forwarded as relevance: the " + f"ranking is inverted and the worst chunks are being served as the best." + ) + + argmax = max(chunks, key=lambda chunk: chunk.relevance) + assert argmax.key == best_key, ( + f"highest relevance is {argmax.key}, expected the nearest chunk {best_key}" + ) + + +@given(distances=st_distances) +@settings(max_examples=200, deadline=None) +def test_managed_backend_ranks_known_best_chunk_first(distances): + """ + **Validates: Requirements 2.1, 2.3, 2.4** + + The managed backend passes relevance through, so the known-best chunk is the + one with the highest relevance and it must come back first. + """ + import asyncio + + # The same logical corpus, expressed in the managed backend's own units. + scored = sorted( + ( + { + "document_id": f"doc-{index}", + "text": f"passage {index}", + "key": f"doc-{index}#0", + "relevance": relevance_from_distance(distance), + } + for index, distance in enumerate(sorted(distances)) + ), + key=lambda result: result["relevance"], + reverse=True, + ) + + backend = FakeManagedBackend(scored) + chunks = asyncio.run(backend.search("ast-1", "a query")) + + best_key = scored[0]["key"] + + assert chunks[0].key == best_key, "known-best chunk is not first" + + relevances = [chunk.relevance for chunk in chunks] + assert _is_non_increasing(relevances), ( + f"relevance must descend with rank, got {relevances}" + ) + + argmax = max(chunks, key=lambda chunk: chunk.relevance) + assert argmax.key == best_key + + +@given(distances=st_distances) +@settings(max_examples=200, deadline=None) +def test_both_backends_agree_on_ranking(distances): + """ + **Validates: Requirement 2.4** + + Given the same corpus and the same relative scores, both backends must return + the same documents in the same order. This is the parity claim a migration + rests on: a knowledge base that moves engines must not reorder its answers. + """ + import asyncio + + legacy_chunks = asyncio.run(_legacy_search(distances)) + + managed_results = [ + { + "document_id": chunk.document_id, + "text": chunk.text, + "key": chunk.key, + "relevance": chunk.relevance, + } + for chunk in sorted(legacy_chunks, key=lambda chunk: chunk.relevance, reverse=True) + ] + managed_chunks = asyncio.run(FakeManagedBackend(managed_results).search("ast-1", "q")) + + assert [chunk.document_id for chunk in legacy_chunks] == [ + chunk.document_id for chunk in managed_chunks + ], "the two backends ranked the same corpus differently" + + assert [chunk.relevance for chunk in legacy_chunks] == [ + chunk.relevance for chunk in managed_chunks + ], "the two backends scored the same corpus differently" + + +# --------------------------------------------------------------------------- +# The derived distance key must be the same value, not a nearby one +# --------------------------------------------------------------------------- + + +@given(distance=st.floats(min_value=0.0, max_value=2.0, allow_nan=False)) +@settings(max_examples=200, deadline=None) +def test_distance_relevance_round_trip_is_exact(distance): + """ + **Validates: Requirement 2.2** + + The facade derives the ``distance`` it emits from ``relevance``, and that + value reaches an HTTP response body. The conversion must therefore be exactly + reversible, not merely close: a ``1.0 - x`` formulation would turn ``0.1`` + into ``0.09999999999999998`` and change a value clients already read. + """ + assert distance_from_relevance(relevance_from_distance(distance)) == distance + + +@given(score=st_any_score) +@settings(max_examples=200, deadline=None) +def test_conversion_inverts_direction_for_every_score(score): + """ + **Validates: Requirements 2.1, 2.2** + + Direction inversion is the whole contract: for any two distinct distances, + the smaller one must produce the larger relevance. Asserted pointwise against + a second score so no clamping, absolute value, or identity mapping can pass. + """ + other = score + 1.0 # strictly greater distance + assert relevance_from_distance(score) > relevance_from_distance(other), ( + "a nearer chunk (smaller distance) must receive a higher relevance" + ) + + +def test_none_score_is_preserved_not_fabricated(): + """ + **Validates: Requirement 2.2** + + A response without a distance yields ``None``, which the facade emits + verbatim as it always has. Defaulting to ``0.0`` would make an unscored + chunk the best-ranked chunk in the list. + """ + assert relevance_from_distance(None) is None + assert distance_from_relevance(None) is None + + +def test_backends_satisfy_the_protocol(): + """Both implementations structurally conform to KnowledgeBaseBackend.""" + assert isinstance(S3VectorsBackend(), KnowledgeBaseBackend) + assert isinstance(FakeManagedBackend([]), KnowledgeBaseBackend) diff --git a/backend/tests/property/test_pbt_kb_status_fail_closed.py b/backend/tests/property/test_pbt_kb_status_fail_closed.py new file mode 100644 index 000000000..eff74cfd7 --- /dev/null +++ b/backend/tests/property/test_pbt_kb_status_fail_closed.py @@ -0,0 +1,196 @@ +"""Property-based tests for fail-closed document status filtering. + +Feature: managed-kb-migration + +**Property 4: unconfirmable status never leaks.** + +The filter's job is to keep chunks belonging to deleted or half-deleted documents +out of retrieval results. Its old fallback returned everything unfiltered whenever +it could not reach DynamoDB, which meant the guard vanished at exactly the moment +it was most likely to matter — and vanished *silently*, since the response looks +identical either way. + +This inverts that (Requirement 5, superseding `reliable-document-deletion` +Requirement 3.4). The property asserted here is deliberately absolute: no matter +how many chunks, how many distinct documents, or what shape of table-level failure +is injected, the result is empty. There is no "mostly" — a single leaked chunk from +a deleted document is the entire failure mode. + +The per-document lookup failure is a different case and is *not* covered by this +property: that one already skipped only its own document, which is correct, and is +left unchanged. + +Validates: Requirements 5.1, 5.2, 24.6. +""" + +from unittest.mock import MagicMock, patch + +import pytest +from hypothesis import HealthCheck, given, settings, strategies as st + +# Imported at module scope, deliberately, and NOT inside the patched context of a +# test. Importing it lazily made the first-ever run differ from every later one: +# the import itself happened while `boto3.resource` was mocked, so module-level +# import work was performed against a mock exactly once and was then cached in +# sys.modules for the rest of the session. That produced a test that failed on a +# cold run and passed on every warm one — the worst failure mode a guard can have, +# because CI is cold and local re-runs are warm. +from apis.shared.assistants.rag_service import _filter_vectors_by_document_status + +ASSISTANT_ID = "ast-failclosed" + +# --------------------------------------------------------------------------- +# Strategies +# --------------------------------------------------------------------------- + +st_document_id = st.text( + alphabet="abcdefghijklmnopqrstuvwxyz0123456789-", min_size=1, max_size=16 +).map(lambda s: f"doc-{s}") + +#: A non-empty set of vectors spread over an arbitrary number of documents. Both +#: axes matter: the filter dedupes document ids before lookup, so "many chunks, +#: one document" and "one chunk each, many documents" exercise different paths. +st_vectors = st.lists(st_document_id, min_size=1, max_size=12).map( + lambda ids: [ + { + "key": f"vec-{i}", + "distance": 0.1, + "metadata": {"document_id": d, "text": f"chunk {i}", "assistant_id": ASSISTANT_ID}, + } + for i, d in enumerate(ids) + ] +) + +#: Table-level failures. Any exception type, raised from the resource or the +#: table handle — the guard must not depend on recognising a specific error. +st_failure = st.sampled_from( + [ + Exception("DynamoDB unavailable"), + RuntimeError("connection reset"), + ValueError("malformed region"), + KeyError("credentials"), + TimeoutError("timed out"), + ] +) + + +# --------------------------------------------------------------------------- +# The property +# --------------------------------------------------------------------------- +@given(vectors=st_vectors, failure=st_failure) +@settings(max_examples=150, suppress_health_check=[HealthCheck.function_scoped_fixture]) +def test_a_table_level_failure_never_leaks_a_chunk(vectors, failure): + """Any table-level failure, any corpus shape → zero chunks.""" + with patch("apis.shared.kb_backend.metrics.emit_count"), patch( + "apis.shared.assistants.rag_service.emit_count" + ), patch("boto3.resource") as resource, patch.dict( + "os.environ", {"DYNAMODB_ASSISTANTS_TABLE_NAME": "t", "AWS_REGION": "us-west-2"} + ): + dynamo = MagicMock() + dynamo.Table.side_effect = failure + resource.return_value = dynamo + + assert _filter_vectors_by_document_status(vectors, ASSISTANT_ID) == [] + + +@given(vectors=st_vectors) +@settings(max_examples=100, suppress_health_check=[HealthCheck.function_scoped_fixture]) +def test_a_missing_table_name_never_leaks_a_chunk(vectors): + """The other former fail-open path: no table configured → zero chunks.""" + with patch("apis.shared.kb_backend.metrics.emit_count"), patch( + "apis.shared.assistants.rag_service.emit_count" + ), patch("boto3.resource") as resource, patch.dict("os.environ", {}, clear=True): + assert _filter_vectors_by_document_status(vectors, ASSISTANT_ID) == [] + # Never contacted, so this is a guard rather than a failed call. + resource.assert_not_called() + + +@given(vectors=st_vectors, failure=st_failure) +@settings(max_examples=100, suppress_health_check=[HealthCheck.function_scoped_fixture]) +def test_the_degradation_is_always_reported(vectors, failure): + """An empty result from this path must be distinguishable from an empty corpus. + + Without the signal, a total retrieval outage looks exactly like "nobody's + documents matched", which is the kind of failure that survives for weeks. + """ + with patch("apis.shared.assistants.rag_service.emit_count") as emit, patch( + "boto3.resource" + ) as resource, patch.dict( + "os.environ", {"DYNAMODB_ASSISTANTS_TABLE_NAME": "t", "AWS_REGION": "us-west-2"} + ): + dynamo = MagicMock() + dynamo.Table.side_effect = failure + resource.return_value = dynamo + + _filter_vectors_by_document_status(vectors, ASSISTANT_ID) + emit.assert_called_once() + + +# --------------------------------------------------------------------------- +# What must NOT change +# --------------------------------------------------------------------------- +@given(vectors=st_vectors) +@settings(max_examples=50, suppress_health_check=[HealthCheck.function_scoped_fixture]) +def test_a_per_document_failure_still_only_drops_that_document(vectors): + """The inner handler was already correct and is deliberately untouched. + + Inverting the table-level fallback must not be over-applied: one unreadable + document should cost that document, not the whole result. Here every lookup + fails individually, so everything drops — but via the per-document path, which + must NOT report a table-level degradation. + """ + with patch("apis.shared.assistants.rag_service.emit_count") as emit, patch( + "boto3.resource" + ) as resource, patch.dict( + "os.environ", {"DYNAMODB_ASSISTANTS_TABLE_NAME": "t", "AWS_REGION": "us-west-2"} + ): + table = MagicMock() + table.get_item.side_effect = Exception("per-item failure") + dynamo = MagicMock() + dynamo.Table.return_value = table + resource.return_value = dynamo + + assert _filter_vectors_by_document_status(vectors, ASSISTANT_ID) == [] + emit.assert_not_called() + + +@given(vectors=st_vectors) +@settings(max_examples=50, suppress_health_check=[HealthCheck.function_scoped_fixture]) +def test_complete_documents_are_still_returned(vectors): + """The happy path, so the properties above cannot pass by always returning [].""" + with patch("apis.shared.assistants.rag_service.emit_count"), patch( + "boto3.resource" + ) as resource, patch.dict( + "os.environ", {"DYNAMODB_ASSISTANTS_TABLE_NAME": "t", "AWS_REGION": "us-west-2"} + ): + table = MagicMock() + table.get_item.return_value = {"Item": {"status": "complete"}} + dynamo = MagicMock() + dynamo.Table.return_value = table + resource.return_value = dynamo + + assert len(_filter_vectors_by_document_status(vectors, ASSISTANT_ID)) == len(vectors) + + +@pytest.mark.parametrize("status", ["deleting", "failed", "uploading", "chunking"]) +def test_a_non_complete_status_is_excluded(status): + """Unchanged behaviour, pinned: only `complete` is served. + + Production carried 200 of 1,692 document records in a non-complete state + (101 deleting, 95 failed, 4 uploading), so this is the common case, not an edge. + """ + with patch("apis.shared.assistants.rag_service.emit_count"), patch( + "boto3.resource" + ) as resource, patch.dict( + "os.environ", {"DYNAMODB_ASSISTANTS_TABLE_NAME": "t", "AWS_REGION": "us-west-2"} + ): + table = MagicMock() + table.get_item.return_value = {"Item": {"status": status}} + dynamo = MagicMock() + dynamo.Table.return_value = table + resource.return_value = dynamo + + vectors = [ + {"key": "v1", "distance": 0.1, "metadata": {"document_id": "doc-a", "text": "t"}} + ] + assert _filter_vectors_by_document_status(vectors, ASSISTANT_ID) == [] diff --git a/backend/tests/routes/test_kb_upgrade.py b/backend/tests/routes/test_kb_upgrade.py new file mode 100644 index 000000000..19f24c844 --- /dev/null +++ b/backend/tests/routes/test_kb_upgrade.py @@ -0,0 +1,686 @@ +"""The owner-facing upgrade surface: enrolment, status, and who may see it. + +Requirements 21 and 23. The assertions here concentrate on the failures that +would ship looking correct: + +* **The offer is gated on the migration flag.** Offering an upgrade the worker + cannot perform parks a record in ``shadow`` behind a spinner that never moves. + "Off" includes present-but-empty, which is the shape of the reconciler-arming + defect and is therefore re-tested per component. +* **Enrolment writes the work keys.** A record created with + ``migrationState="shadow"`` but no ``GSI7_PK``/``GSI7_SK`` is invisible to the + dispatcher's sparse-index sweep *forever*, while every surface reports an + upgrade in progress. ``KbRecord.to_item`` does not write those keys, so this is + a live trap rather than a hypothetical one. +* **Enrolment never writes ``retrievalEngine``.** Promotion belongs to the worker + and only after verification. An HTTP request that could set it would cut a + knowledge base over to an empty corpus. +* **A viewer cannot enrol, and is not offered it.** Requirement 23.7. +* **A retry bumps the generation**, which is what fences a straggler worker from + the abandoned attempt. +* **``deleting`` documents are surfaced, not hidden.** They are 101 of the 200 + affected production records, and the ordinary document list filters them out. +* **No user-facing string says "vector"** (Requirement 23.6), asserted over every + string the module can emit rather than spot-checked. + +Feature: managed-kb-migration +Requirements: 21.1, 21.2, 21.3, 21.4, 23.1, 23.2, 23.3, 23.4, 23.5, 23.6, 23.7, +23.8 +""" + +from decimal import Decimal +from typing import Any, Dict, List, Optional +from unittest.mock import MagicMock, patch + +import pytest +from fastapi import FastAPI + +from apis.app_api.kb_upgrade import models as m +from apis.app_api.kb_upgrade import service as s +from apis.shared.kb_backend import records as r + +ASSISTANT_ID = "ast-upgrade-001" +OWNER_ID = "user-owner" +TABLE = "test-assistants" + +ENV_ON = { + "DYNAMODB_ASSISTANTS_TABLE_NAME": TABLE, + "MANAGED_KB_MIGRATION_ENABLED": "true", + "AWS_REGION": "us-west-2", +} + + +def _doc( + document_id: str, + status: str = "complete", + filename: Optional[str] = None, +) -> Dict[str, Any]: + return { + "PK": f"AST#{ASSISTANT_ID}", + "SK": f"DOC#{document_id}", + "documentId": document_id, + "status": status, + "filename": filename or f"{document_id}.pdf", + } + + +def _kb_record(**overrides) -> Dict[str, Any]: + record: Dict[str, Any] = { + "PK": f"AST#{ASSISTANT_ID}", + "SK": f"KB#{ASSISTANT_ID}", + "appKbId": ASSISTANT_ID, + "ownerUserId": OWNER_ID, + "migrationGeneration": Decimal(0), + } + record.update(overrides) + return record + + +class _FakeTable: + """Enough DynamoDB for these paths, recording every write for assertion.""" + + def __init__(self, item: Optional[Dict[str, Any]] = None, docs=None): + self.item = item + self.docs = list(docs or []) + self.updates: List[Dict[str, Any]] = [] + self.puts: List[Dict[str, Any]] = [] + + def get_item(self, Key, **kwargs): + return {"Item": self.item} if self.item else {} + + def query(self, **kwargs): + return {"Items": self.docs} + + def update_item(self, **kwargs): + self.updates.append(kwargs) + return {} + + def put_item(self, **kwargs): + self.puts.append(kwargs) + return {} + + +@pytest.fixture +def table(monkeypatch): + """Point both ``records`` and the service's document query at one fake.""" + fake = _FakeTable() + + def _resource(*args, **kwargs): + resource = MagicMock() + resource.Table.return_value = fake + return resource + + monkeypatch.setattr("boto3.resource", _resource) + for key, value in ENV_ON.items(): + monkeypatch.setenv(key, value) + return fake + + +# ── The flag gate (Requirement 23.1) ───────────────────────────────────────── +class TestFlagGate: + @pytest.mark.parametrize( + "raw", ["", " ", "false", "no", "off", "0", "disabled", "True-ish", None] + ) + def test_absent_empty_or_negative_reads_as_off(self, monkeypatch, raw): + """Present-but-empty must read as off, not as truthy-by-accident.""" + if raw is None: + monkeypatch.delenv(s.FLAG_MIGRATION_ENABLED, raising=False) + else: + monkeypatch.setenv(s.FLAG_MIGRATION_ENABLED, raw) + assert s.migration_enabled() is False + + @pytest.mark.parametrize("raw", ["1", "true", "TRUE", " yes ", "on", "enabled"]) + def test_affirmative_spellings_read_as_on(self, monkeypatch, raw): + monkeypatch.setenv(s.FLAG_MIGRATION_ENABLED, raw) + assert s.migration_enabled() is True + + def test_the_flag_is_read_at_call_time(self, monkeypatch): + """Bound as a default argument the flag would be unpatchable. + + Asserted by flipping it *between* two calls on the same import. + """ + monkeypatch.setenv(s.FLAG_MIGRATION_ENABLED, "true") + assert s.migration_enabled() is True + monkeypatch.setenv(s.FLAG_MIGRATION_ENABLED, "false") + assert s.migration_enabled() is False + + @pytest.mark.asyncio + async def test_no_offer_while_the_flag_is_off(self, table, monkeypatch): + monkeypatch.setenv(s.FLAG_MIGRATION_ENABLED, "false") + table.docs = [_doc("d1")] + status = await s.get_upgrade_status(ASSISTANT_ID, can_edit=True) + assert status.phase == "none" + assert status.can_upgrade is False + + @pytest.mark.asyncio + async def test_enrolment_refuses_while_the_flag_is_off(self, table, monkeypatch): + monkeypatch.setenv(s.FLAG_MIGRATION_ENABLED, "false") + with pytest.raises(s.UpgradeUnavailable): + await s.enroll(ASSISTANT_ID, owner_user_id=OWNER_ID) + assert table.updates == [], "a refused enrolment must not write" + assert table.puts == [], "a refused enrolment must not create a record" + + +# ── Status derivation (Requirement 23.1–23.5) ──────────────────────────────── +class TestStatus: + @pytest.mark.asyncio + async def test_empty_knowledge_base_shows_nothing(self, table): + """Requirement 23.1: no action required means no badge, banner or prompt.""" + table.docs = [] + status = await s.get_upgrade_status(ASSISTANT_ID, can_edit=True) + assert status.phase == "none" + assert status.documents_not_carried == [] + + @pytest.mark.asyncio + async def test_legacy_with_documents_is_available(self, table): + table.docs = [_doc("d1"), _doc("d2")] + status = await s.get_upgrade_status(ASSISTANT_ID, can_edit=True) + assert status.phase == "available" + assert status.can_upgrade is True + assert status.progress is not None + assert status.progress.total == 2 + + @pytest.mark.asyncio + async def test_a_viewer_is_never_offered_the_control(self, table): + """Requirement 23.7 — and the server, not the client, decides.""" + table.docs = [_doc("d1")] + status = await s.get_upgrade_status(ASSISTANT_ID, can_edit=False) + assert status.can_upgrade is False + assert status.phase == "none" + + @pytest.mark.asyncio + @pytest.mark.parametrize("state", [r.SHADOW, r.VERIFY, r.PROMOTE]) + async def test_working_states_collapse_to_in_progress(self, table, state): + """The client is not told which internal step is running.""" + table.item = _kb_record( + migrationState=state, + migrationProgress={ + "migrated": Decimal(12), + "total": Decimal(40), + "skipped": Decimal(0), + }, + ) + status = await s.get_upgrade_status(ASSISTANT_ID, can_edit=True) + assert status.phase == "in_progress" + assert status.can_upgrade is False, "no second upgrade while one runs" + assert status.progress.completed == 12 + assert status.progress.total == 40 + + @pytest.mark.asyncio + async def test_failed_stays_retryable_and_explains_itself(self, table): + """Requirement 23.5: plain-language reason, retry offered, never a dead end.""" + table.item = _kb_record( + migrationState=r.MIGRATION_FAILED, + migrationError="ByteCapExceeded: 900000000 over cap", + ) + status = await s.get_upgrade_status(ASSISTANT_ID, can_edit=True) + assert status.phase == "failed" + assert status.can_upgrade is True + assert status.reason and "size limit" in status.reason + assert "ByteCapExceeded" not in status.reason, ( + "the operator's error string must not reach the user verbatim" + ) + + @pytest.mark.asyncio + async def test_an_unrecognised_failure_does_not_leak_the_operator_string( + self, table + ): + """The fallback must be copy, not a pass-through. + + Found by mutation: mapping a *recognised* error proved nothing about the + unrecognised path, and ``return stored or _FAILURE_FALLBACK`` survived — + a stack-trace fragment rendered into the card. + """ + leak = "ClientError: An error occurred (AccessDeniedException) calling ..." + table.item = _kb_record(migrationState=r.MIGRATION_FAILED, migrationError=leak) + status = await s.get_upgrade_status(ASSISTANT_ID, can_edit=True) + assert status.phase == "failed" + assert status.reason == s._FAILURE_FALLBACK + assert "AccessDeniedException" not in status.reason + assert "ClientError" not in status.reason + + @pytest.mark.asyncio + async def test_a_viewer_cannot_retry_a_failure(self, table): + table.item = _kb_record(migrationState=r.MIGRATION_FAILED) + status = await s.get_upgrade_status(ASSISTANT_ID, can_edit=False) + assert status.phase == "failed" + assert status.can_upgrade is False + + @pytest.mark.asyncio + async def test_promoted_owes_a_one_time_notice(self, table): + """Requirement 23.4: dismissible notice, never a permanent badge.""" + table.item = _kb_record( + retrievalEngine=r.ENGINE_MANAGED, migrationState=r.RETAIN + ) + status = await s.get_upgrade_status(ASSISTANT_ID, can_edit=True) + assert status.phase == "succeeded" + assert status.notice_pending is True + + @pytest.mark.asyncio + async def test_a_dismissed_notice_never_returns(self, table): + table.item = _kb_record( + retrievalEngine=r.ENGINE_MANAGED, + migrationState=r.RETAIN, + upgradeNoticeDismissedAt="2026-08-01T00:00:00Z", + ) + status = await s.get_upgrade_status(ASSISTANT_ID, can_edit=True) + assert status.phase == "succeeded" + assert status.notice_pending is False + + @pytest.mark.asyncio + async def test_an_unreadable_document_list_fails_loudly_here( + self, table, monkeypatch + ): + """The service raises; the *route* is what degrades. + + Split deliberately: a misconfigured table name must not make an + upgradeable knowledge base look clean to anything that calls the service + directly. Only the HTTP layer is allowed to soften it, and only to + "nothing to show". + """ + monkeypatch.setenv(s.FLAG_MIGRATION_ENABLED, "true") + monkeypatch.delenv("DYNAMODB_ASSISTANTS_TABLE_NAME", raising=False) + # KeyError from the record read, which happens first; RuntimeError from + # the document query if it ever gets there. Either is loud, and neither + # is "phase: none". + with pytest.raises((RuntimeError, KeyError)): + await s.get_upgrade_status(ASSISTANT_ID, can_edit=True) + + +# ── Stranded documents (Requirement 21) ────────────────────────────────────── +class TestStrandedDocuments: + def test_complete_documents_are_not_flagged(self): + assert s.classify_document(_doc("d1", "complete")) is None + + def test_an_unsupported_format_is_distinguished_from_a_failure(self): + """Requirement 21.4 — the two demand different actions from the user.""" + unsupported = s.classify_document( + _doc("d1", "failed", filename="keynote-deck.pages") + ) + broken = s.classify_document(_doc("d2", "failed", filename="report.pdf")) + assert unsupported.kind == "unsupported_format" + assert unsupported.retryable is False + assert broken.kind == "processing_failure" + assert broken.retryable is True + assert unsupported.message != broken.message + + def test_the_unsupported_set_comes_from_the_ingestion_pipeline(self): + """A copied extension list is the tag-contract defect's exact shape.""" + from apis.app_api.documents.ingestion.processors.docling_processor import ( + DOCLING_SUPPORTED_EXTENSIONS, + ) + + assert s._supported_extensions() == frozenset(DOCLING_SUPPORTED_EXTENSIONS) + # A format the pipeline genuinely supports must never be called + # unsupported, however it failed. + assert ".pdf" in DOCLING_SUPPORTED_EXTENSIONS + issue = s.classify_document(_doc("d1", "failed", filename="x.pdf")) + assert issue.kind == "processing_failure" + + def test_stuck_deleting_documents_are_surfaced(self): + """101 of the 200 affected production records are in this status. + + The ordinary document list filters ``deleting`` out as soft-deleted, + which is right there and wrong here: a user never shown them cannot tell + that they are stuck. + """ + issue = s.classify_document(_doc("d1", "deleting")) + assert issue is not None + assert issue.kind == "being_removed" + + @pytest.mark.parametrize("status", ["uploading", "chunking", "embedding"]) + def test_in_flight_documents_are_surfaced_as_still_processing(self, status): + issue = s.classify_document(_doc("d1", status)) + assert issue.kind == "still_processing" + assert issue.retryable is True + + @pytest.mark.asyncio + async def test_stranded_documents_ride_along_with_the_offer(self, table): + """Requirement 21.1/21.3: surfaced *before* the user commits, not after.""" + table.docs = [ + _doc("d1", "complete"), + _doc("d2", "failed"), + _doc("d3", "deleting"), + ] + status = await s.get_upgrade_status(ASSISTANT_ID, can_edit=True) + assert status.phase == "available" + assert {d.document_id for d in status.documents_not_carried} == {"d2", "d3"} + assert status.progress.total == 1, "only the complete document is carried" + assert status.progress.skipped == 2 + + @pytest.mark.asyncio + async def test_an_all_stranded_corpus_is_not_offered_but_is_reported(self, table): + table.docs = [_doc("d1", "failed"), _doc("d2", "failed")] + status = await s.get_upgrade_status(ASSISTANT_ID, can_edit=True) + assert status.phase == "none", "an upgrade carrying nothing is not an upgrade" + assert len(status.documents_not_carried) == 2, ( + "but the owner still needs to see them (Requirement 21.3)" + ) + + def test_a_document_id_survives_a_missing_attribute(self): + """Falls back to the sort key rather than emitting a blank retry target.""" + item = _doc("d9", "failed") + del item["documentId"] + assert s.classify_document(item).document_id == "d9" + + +# ── Copy (Requirement 23.6) ────────────────────────────────────────────────── +class TestCopy: + def test_no_user_facing_string_says_vector(self): + """Swept over every string the module can emit, not spot-checked.""" + emitted = list(s._FAILURE_COPY.values()) + [s._FAILURE_FALLBACK] + for status in ["failed", "deleting", "uploading", "chunking"]: + for filename in ["a.pdf", "b.pages"]: + issue = s.classify_document(_doc("d1", status, filename=filename)) + if issue: + emitted.append(issue.message) + for text in emitted: + assert "vector" not in text.lower(), f"user-facing copy says vector: {text}" + + @pytest.mark.asyncio + async def test_enrolment_copy_promises_continued_service(self, table): + """Requirement 23.2's honest claim, and the one users act on.""" + result = await s.enroll(ASSISTANT_ID, owner_user_id=OWNER_ID) + assert "keeps working" in result.message + assert "vector" not in result.message.lower() + + +# ── Enrolment (Requirements 23.2, 23.8) ────────────────────────────────────── +def _update_expressions(fake: _FakeTable) -> str: + return " | ".join(u.get("UpdateExpression", "") for u in fake.updates) + + +class TestEnrolment: + @pytest.mark.asyncio + async def test_enrolment_creates_the_record_then_enters_shadow(self, table): + result = await s.enroll(ASSISTANT_ID, owner_user_id=OWNER_ID) + assert result.started is True + assert result.phase == "in_progress" + assert len(table.puts) == 1, "the record is created exactly once" + assert len(table.updates) == 1, "then transitioned exactly once" + + @pytest.mark.asyncio + async def test_enrolment_writes_the_dispatcher_work_keys(self, table): + """Without these the record is invisible to the sweep, forever. + + ``KbRecord.to_item`` does not write them, so a one-put enrolment would + look correct and strand the knowledge base behind a permanent spinner. + """ + await s.enroll(ASSISTANT_ID, owner_user_id=OWNER_ID) + expression = _update_expressions(table) + assert "GSI7_PK" in expression + assert "GSI7_SK" in expression + values = table.updates[0]["ExpressionAttributeValues"] + assert values[":wpk"] == r.work_pk(r.SHADOW) + assert values[":wsk"], "a work-eligible state requires a due time" + + @pytest.mark.asyncio + async def test_the_created_record_does_not_claim_to_be_migrating(self, table): + """The put must not carry ``migrationState``. + + If it did, a crash between the two writes would leave a record that says + it is migrating with no work keys to make it so. + """ + await s.enroll(ASSISTANT_ID, owner_user_id=OWNER_ID) + assert "migrationState" not in table.puts[0]["Item"] + + @pytest.mark.asyncio + async def test_enrolment_never_writes_the_retrieval_engine(self, table): + """Promotion is the worker's, after verification. Requirement 23.8. + + An HTTP path that could set this would cut a knowledge base over to a + corpus nothing had carried across yet. + """ + await s.enroll(ASSISTANT_ID, owner_user_id=OWNER_ID) + assert "retrievalEngine" not in table.puts[0]["Item"] + assert "retrievalEngine" not in _update_expressions(table) + + @pytest.mark.asyncio + async def test_enrolling_twice_starts_one_migration(self, table): + """A double-click is not an error, and not two provisioning sagas.""" + await s.enroll(ASSISTANT_ID, owner_user_id=OWNER_ID) + table.item = _kb_record(migrationState=r.SHADOW) + again = await s.enroll(ASSISTANT_ID, owner_user_id=OWNER_ID) + assert again.started is False + assert again.phase == "in_progress" + assert len(table.puts) == 1, "no second record" + + @pytest.mark.asyncio + async def test_enrolling_an_already_upgraded_kb_is_a_no_op(self, table): + table.item = _kb_record(retrievalEngine=r.ENGINE_MANAGED) + result = await s.enroll(ASSISTANT_ID, owner_user_id=OWNER_ID) + assert result.started is False + assert result.phase == "succeeded" + assert table.updates == [] + + @pytest.mark.asyncio + async def test_a_lost_transition_reports_the_running_upgrade(self, table): + """Losing the race is normal and must not surface as an error.""" + with patch.object( + r, "set_migration_state", side_effect=r.TransitionLost("raced") + ): + result = await s.enroll(ASSISTANT_ID, owner_user_id=OWNER_ID) + assert result.started is False + assert result.phase == "in_progress" + + @pytest.mark.asyncio + async def test_a_concurrently_created_record_does_not_fail_enrolment(self, table): + """``create_provisioning`` losing means someone else made it, not an error.""" + with patch.object( + r, "create_provisioning", side_effect=r.TransitionLost("raced") + ): + result = await s.enroll(ASSISTANT_ID, owner_user_id=OWNER_ID) + assert result.phase == "in_progress" + + +# ── Retry (Requirement 23.5) ───────────────────────────────────────────────── +class TestRetry: + @pytest.mark.asyncio + async def test_retry_bumps_the_generation_and_re_enters_shadow(self, table): + """The bump is what fences a straggler from the abandoned attempt.""" + table.item = _kb_record( + migrationState=r.MIGRATION_FAILED, migrationGeneration=Decimal(3) + ) + result = await s.retry(ASSISTANT_ID, owner_user_id=OWNER_ID) + assert result.started is True + values = table.updates[0]["ExpressionAttributeValues"] + assert values[":gen"] == Decimal(3), "guarded on the generation it read" + assert values[":next"] == Decimal(4) + assert values[":shadow"] == r.SHADOW + + @pytest.mark.asyncio + async def test_retry_is_one_atomic_write(self, table): + """Two writes leave a crash window with a new generation and no work keys.""" + table.item = _kb_record(migrationState=r.MIGRATION_FAILED) + await s.retry(ASSISTANT_ID, owner_user_id=OWNER_ID) + assert len(table.updates) == 1 + expression = table.updates[0]["UpdateExpression"] + assert "migrationGeneration" in expression + assert "migrationState" in expression + assert "GSI7_PK" in expression + + @pytest.mark.asyncio + async def test_retry_is_guarded_on_still_being_failed(self, table): + table.item = _kb_record(migrationState=r.MIGRATION_FAILED) + await s.retry(ASSISTANT_ID, owner_user_id=OWNER_ID) + condition = table.updates[0]["ConditionExpression"] + assert "migrationGeneration = :gen" in condition + assert "migrationState = :failed" in condition + + @pytest.mark.asyncio + async def test_retry_clears_the_previous_error(self, table): + """A stale reason would be read as the next failure's.""" + table.item = _kb_record( + migrationState=r.MIGRATION_FAILED, migrationError="ByteCapExceeded" + ) + await s.retry(ASSISTANT_ID, owner_user_id=OWNER_ID) + assert "REMOVE migrationError" in table.updates[0]["UpdateExpression"] + + @pytest.mark.asyncio + async def test_concurrent_retries_yield_one_attempt(self, table): + table.item = _kb_record(migrationState=r.MIGRATION_FAILED) + with patch.object( + r, "retry_from_failed", side_effect=r.TransitionLost("raced") + ): + result = await s.retry(ASSISTANT_ID, owner_user_id=OWNER_ID) + assert result.started is False + assert result.phase == "in_progress" + + @pytest.mark.asyncio + async def test_retry_on_a_running_upgrade_does_not_restart_it(self, table): + table.item = _kb_record(migrationState=r.VERIFY) + result = await s.retry(ASSISTANT_ID, owner_user_id=OWNER_ID) + assert result.started is False + assert table.updates == [] + + @pytest.mark.asyncio + async def test_retry_with_no_record_enrols_instead(self, table): + table.item = None + result = await s.retry(ASSISTANT_ID, owner_user_id=OWNER_ID) + assert result.started is True + assert len(table.puts) == 1 + + +# ── Notice dismissal (Requirement 23.4) ────────────────────────────────────── +class TestNotice: + @pytest.mark.asyncio + async def test_dismissal_is_guarded_on_the_record_existing(self, table): + table.item = _kb_record(retrievalEngine=r.ENGINE_MANAGED) + await s.dismiss_notice(ASSISTANT_ID) + assert "upgradeNoticeDismissedAt" in table.updates[0]["UpdateExpression"] + assert "attribute_exists(PK)" in table.updates[0]["ConditionExpression"] + + @pytest.mark.asyncio + async def test_dismissing_a_missing_record_is_not_an_error(self, table): + with patch.object( + r, "dismiss_upgrade_notice", side_effect=r.TransitionLost("absent") + ): + await s.dismiss_notice(ASSISTANT_ID) # must not raise + + +# ── Routes (Requirement 23.7) ──────────────────────────────────────────────── +@pytest.fixture +def app(): + from apis.app_api.kb_upgrade.routes import router + + application = FastAPI() + application.include_router(router) + return application + + +class _FakeAssistant: + owner_id = OWNER_ID + visibility = "PRIVATE" + + +def _permission(permission: Optional[str], exists: bool = True): + """Patch the shared permission resolver the routes gate on.""" + from unittest.mock import AsyncMock + + assistant = _FakeAssistant() if exists else None + return patch( + "apis.app_api.kb_upgrade.routes.resolve_assistant_permission", + new=AsyncMock(return_value=(assistant, permission)), + ) + + +class TestRoutePermissions: + @pytest.mark.parametrize( + "method,path", + [ + ("post", ""), + ("post", "/retry"), + ("post", "/notice"), + ], + ) + def test_a_viewer_cannot_write( + self, app, authenticated_client, make_user, method, path + ): + """Requirement 23.7 enforced server-side, not by hiding a button.""" + client = authenticated_client(app, make_user(user_id="someone-else")) + with _permission("viewer"): + response = getattr(client, method)( + f"/assistants/{ASSISTANT_ID}/knowledge-base/upgrade{path}" + ) + assert response.status_code == 403 + + def test_no_permission_reads_as_not_found( + self, app, authenticated_client, make_user + ): + """404 rather than 403, so the endpoint does not confirm existence.""" + client = authenticated_client(app, make_user(user_id="stranger")) + with _permission(None): + response = client.get( + f"/assistants/{ASSISTANT_ID}/knowledge-base/upgrade" + ) + assert response.status_code == 404 + + def test_a_viewer_may_read_but_is_told_it_cannot_upgrade( + self, app, authenticated_client, make_user, table + ): + table.docs = [_doc("d1")] + client = authenticated_client(app, make_user(user_id="viewer-1")) + with _permission("viewer"): + response = client.get( + f"/assistants/{ASSISTANT_ID}/knowledge-base/upgrade" + ) + assert response.status_code == 200 + assert response.json()["canUpgrade"] is False + + def test_an_owner_can_start_an_upgrade( + self, app, authenticated_client, make_user, table + ): + client = authenticated_client(app, make_user(user_id=OWNER_ID)) + with _permission("owner"): + response = client.post( + f"/assistants/{ASSISTANT_ID}/knowledge-base/upgrade" + ) + assert response.status_code == 202 + assert response.json()["started"] is True + + def test_the_status_read_degrades_rather_than_500s( + self, app, authenticated_client, make_user, monkeypatch + ): + """A card that cannot be described must not take the page down.""" + monkeypatch.delenv("DYNAMODB_ASSISTANTS_TABLE_NAME", raising=False) + client = authenticated_client(app, make_user(user_id=OWNER_ID)) + with _permission("owner"): + response = client.get( + f"/assistants/{ASSISTANT_ID}/knowledge-base/upgrade" + ) + assert response.status_code == 200 + assert response.json()["phase"] == "none" + + def test_enrolment_conflicts_while_the_flag_is_off( + self, app, authenticated_client, make_user, table, monkeypatch + ): + monkeypatch.setenv(s.FLAG_MIGRATION_ENABLED, "false") + client = authenticated_client(app, make_user(user_id=OWNER_ID)) + with _permission("owner"): + response = client.post( + f"/assistants/{ASSISTANT_ID}/knowledge-base/upgrade" + ) + assert response.status_code == 409 + + +class TestWireContract: + """The client reads camelCase; a rename here silently empties the card.""" + + def test_status_serialises_camel_case(self): + payload = m.UpgradeStatusResponse( + phase="available", canUpgrade=True + ).model_dump(by_alias=True) + assert "canUpgrade" in payload + assert "noticePending" in payload + assert "documentsNotCarried" in payload + + def test_stranded_document_serialises_camel_case(self): + payload = m.DocumentNotCarried( + documentId="d1", + filename="a.pdf", + status="failed", + kind="processing_failure", + message="x", + retryable=True, + ).model_dump(by_alias=True) + assert payload["documentId"] == "d1" diff --git a/backend/tests/shared/test_kb_authorization.py b/backend/tests/shared/test_kb_authorization.py new file mode 100644 index 000000000..4a5dd4629 --- /dev/null +++ b/backend/tests/shared/test_kb_authorization.py @@ -0,0 +1,625 @@ +""" +Authorization, isolation and publication: the app is the authority. + +Requirement 25. Managed KB ships two features whose names overstate what they +provide — metadata filters are "filter-level (logical) isolation, *not* +IAM-enforced", and ACL-aware retrieval "is not authorization" by AWS's own +statement. This feature therefore keeps authorization in the application, and +these tests are what stop that from eroding. + +Four things are asserted, in the order they can fail silently: + +1. **The gate runs before the backend does.** Not "an unauthorized caller gets + an empty list" — that is also what an authorized caller with an empty corpus + gets. What matters is that no backend was contacted at all, so a stand-in + backend records its calls and the assertion is on that record. +2. **The gate fails closed**, including when the permission lookup raises. +3. **Filters are not the tenant boundary**, and ACL-aware retrieval is not + adopted anywhere in the seam. +4. **Publication semantics**: an engine swap is not a corpus change, and an agent + on the store shelf is exempt from reclaim — asked as ``is_on_shelf`` rather + than ``is_listed``, because a live listing sitting in ``changes_requested`` + reads as unlisted by state name alone. + +Feature: managed-kb-migration +Requirements: 25.1, 25.2, 25.3, 25.4, 25.5, 25.6, 25.7, 25.8, 25.9, 25.10, +25.11, 24.6, 24.12, 24.14, 11.5 +""" + +import json +from pathlib import Path +from typing import Any, Dict, List +from unittest.mock import MagicMock, patch + +import pytest + +from apis.shared.assistants.kb_access import ( + KB_READ_PERMISSIONS, + KB_WRITE_PERMISSIONS, + granted, + is_shared_beyond_owner, + resolve_kb_access, +) +from apis.shared.assistants.kb_publication import ( + is_reclaim_exempt, + migration_requires_review, + reclaim_exemption_reason, +) +from apis.shared.assistants.rag_service import search_assistant_knowledgebase_with_formatting +from apis.shared.kb_backend.protocol import DEFAULT_TOP_K, Chunk +from apis.shared.kb_backend.resource_policy import ( + POLICY_KB_ID_ATTR, + POLICY_REVISION_ATTR, + RETRIEVE_ACTIONS, + ResourcePolicyError, + ensure_retrieve_policy, + knowledge_base_arn, + policy_is_stale, + retrieval_principals, + retrieve_policy_document, +) + +ASSISTANT_ID = "ast-authz-001" +USER_ID = "user-authz-001" +TABLE_NAME = "test-table" + + +class RecordingBackend: + """A backend that remembers whether it was asked anything. + + The whole point of Requirement 25.1 is *ordering*: a check that runs after + the corpus has been read is an audit trail, not an access control. An empty + return value cannot distinguish the two, so this records calls instead. + """ + + def __init__(self, chunks: List[Chunk] = None): + self._chunks = chunks or [] + self.calls: List[Dict[str, Any]] = [] + + async def search(self, kb_ref: str, query: str, top_k: int = DEFAULT_TOP_K) -> List[Chunk]: + self.calls.append({"kb_ref": kb_ref, "query": query, "top_k": top_k}) + return list(self._chunks) + + async def ingest(self, kb_ref: str, source) -> None: # pragma: no cover - unused + raise NotImplementedError + + async def delete_document(self, kb_ref: str, document_id: str) -> None: # pragma: no cover + raise NotImplementedError + + +def _chunk(document_id: str = "doc-1") -> Chunk: + return Chunk( + text=f"passage from {document_id}", + relevance=1.0, + document_id=document_id, + metadata={"document_id": document_id}, + key=f"{document_id}#0", + ) + + +def _complete_document_table() -> MagicMock: + """A DynamoDB stand-in whose documents are all ``complete``. + + So that a test about authorization cannot pass because the *status* filter + dropped everything — a false green that would survive removing the gate. + """ + table = MagicMock() + table.get_item.return_value = {"Item": {"status": "complete"}} + resource = MagicMock() + resource.Table.return_value = table + return resource + + +async def _search(access, backend: RecordingBackend, top_k: int = 5): + with patch.dict("os.environ", {"DYNAMODB_ASSISTANTS_TABLE_NAME": TABLE_NAME}, clear=False), patch( + "apis.shared.assistants.rag_service.resolve_backend", return_value=backend + ), patch( + "apis.shared.assistants.rag_service.boto3.resource", + return_value=_complete_document_table(), + ), patch( + "apis.shared.assistants.rag_service.emit_count" + ) as emit: + results = await search_assistant_knowledgebase_with_formatting( + ASSISTANT_ID, "q", top_k, access=access + ) + return results, emit + + +# ── 1. The gate runs, and it runs first ────────────────────────────────────── +class TestAccessIsResolvedBeforeRetrieval: + @pytest.mark.asyncio + async def test_an_owner_reaches_the_backend(self): + """The permissive case, first — so every denial below means something.""" + backend = RecordingBackend([_chunk()]) + results, _ = await _search(granted(ASSISTANT_ID, USER_ID, "owner"), backend) + + assert len(results) == 1 + assert backend.calls, "an owner must reach the backend" + + @pytest.mark.asyncio + async def test_a_viewer_reads_through_the_agent(self): + """Sharing an agent is *for* reading, so a viewer retrieves (Req 25.2).""" + backend = RecordingBackend([_chunk()]) + results, _ = await _search(granted(ASSISTANT_ID, USER_ID, "viewer"), backend) + + assert len(results) == 1 + assert backend.calls + + @pytest.mark.asyncio + async def test_a_viewer_may_not_upgrade(self): + """Reading is not upgrading: an engine migration spends money.""" + access = granted(ASSISTANT_ID, USER_ID, "viewer") + assert access.may_read is True + assert access.may_upgrade is False + + for permission in ("owner", "editor"): + assert granted(ASSISTANT_ID, USER_ID, permission).may_upgrade is True + + @pytest.mark.asyncio + async def test_no_grant_never_reaches_the_backend(self): + """Requirement 25.1: resolved *before* retrieval is attempted. + + Asserting on ``backend.calls`` rather than on the return value, because + an empty list is also what an authorized user with no matches gets. + """ + backend = RecordingBackend([_chunk()]) + results, emit = await _search(None, backend) + + assert results == [] + assert backend.calls == [], ( + "the backend was queried despite there being no grant; the access " + "check is running after retrieval, which makes it an audit log" + ) + assert any(call.args and call.args[0] == "KbAccessDenied" for call in emit.call_args_list) + + @pytest.mark.asyncio + async def test_a_grant_for_another_assistant_is_refused(self): + """The copy-paste shape: permission resolved for one id, retrieval on another.""" + backend = RecordingBackend([_chunk()]) + other = granted("ast-somebody-else", USER_ID, "owner") + results, _ = await _search(other, backend) + + assert results == [] + assert backend.calls == [] + + def test_the_access_argument_is_required(self): + """Forgetting it must be a call-site failure, not a silent empty result. + + A default of ``None`` would fail closed too, but it would fail closed + *quietly* in a caller that never intended to deny anyone. + """ + import asyncio + + with pytest.raises(TypeError, match="access"): + asyncio.run(search_assistant_knowledgebase_with_formatting(ASSISTANT_ID, "q")) + + +# ── 2. Fail closed ─────────────────────────────────────────────────────────── +class TestAccessChecksFailClosed: + """Requirement 24.6. Contrast with the resolver, which treats an unreadable + KB_Record as legacy: there both answers serve the user's own documents, so + one of them is always safe. Here the answers are "yours" and "someone + else's".""" + + @pytest.mark.parametrize("permission", [None, "", "unknown", "auditor", "OWNER"]) + def test_anything_but_a_known_read_permission_denies(self, permission): + """Including a case-variant, which is how a refactor introduces this bug.""" + assert granted(ASSISTANT_ID, USER_ID, permission) is None + + @pytest.mark.asyncio + async def test_a_failing_permission_lookup_denies(self): + async def _boom(**kwargs): + raise RuntimeError("dynamodb unavailable") + + with patch( + "apis.shared.assistants.service.resolve_assistant_permission", side_effect=_boom + ): + assert await resolve_kb_access(ASSISTANT_ID, USER_ID, "a@b.test") is None + + @pytest.mark.asyncio + async def test_a_resolved_permission_becomes_a_grant(self): + async def _resolve(**kwargs): + return object(), "editor" + + with patch( + "apis.shared.assistants.service.resolve_assistant_permission", side_effect=_resolve + ): + access = await resolve_kb_access(ASSISTANT_ID, USER_ID, "a@b.test") + + assert access is not None + assert access.permission == "editor" + # 1:1 this phase: the knowledge base id is the assistant id. + assert access.app_kb_id == ASSISTANT_ID + + @pytest.mark.asyncio + async def test_no_permission_resolves_to_no_grant(self): + async def _resolve(**kwargs): + return object(), None + + with patch( + "apis.shared.assistants.service.resolve_assistant_permission", side_effect=_resolve + ): + assert await resolve_kb_access(ASSISTANT_ID, USER_ID) is None + + def test_the_permission_sets_do_not_overlap_wrongly(self): + """Write implies read; read does not imply write.""" + assert KB_WRITE_PERMISSIONS < KB_READ_PERMISSIONS + assert "viewer" in KB_READ_PERMISSIONS + assert "viewer" not in KB_WRITE_PERMISSIONS + + def test_a_grant_cannot_be_mutated_after_it_is_resolved(self): + access = granted(ASSISTANT_ID, USER_ID, "viewer") + with pytest.raises(Exception): + access.permission = "owner" + + +# ── 3. Filters are not the tenant boundary ─────────────────────────────────── +_KB_BACKEND_DIR = ( + Path(__file__).resolve().parent.parent.parent / "src" / "apis" / "shared" / "kb_backend" +) + + +class TestFiltersAreNotTheTenantBoundary: + def test_the_seam_does_not_adopt_acl_aware_retrieval(self): + """Requirement 25.5, asserted against the source. + + ACL-aware retrieval's identity is email only, with no alias resolution, + and a mismatch fails silently. On a platform that authenticates via OIDC + with claim mappings, that is a worse primitive than an explicit check — + so the configuration keys must not appear at all. A test on behaviour + could not see a config key that was set but never exercised in tests. + """ + forbidden = ("aclConfiguration", "userGroupFilter", "implicitFilterConfiguration") + offenders = [] + for pyfile in sorted(_KB_BACKEND_DIR.glob("*.py")): + body = pyfile.read_text(encoding="utf-8") + for key in forbidden: + if key in body: + offenders.append(f"{pyfile.name} mentions {key}") + assert offenders == [], ( + "ACL-aware retrieval is deliberately not adopted in this phase " + "(Requirement 25.5): " + "; ".join(offenders) + ) + + def test_retrieval_is_scoped_by_knowledge_base_not_by_filter(self): + """Requirement 25.4. The default retrieval configuration carries no filter. + + The boundary is one knowledge base per assistant, which holds because + this phase keeps ``App_KB_Id == assistant_id``. A filter is a query + argument, and anything that can issue a query can omit it — so if the + default config carried a tenancy filter, omitting it would be a + cross-tenant read. + """ + from apis.shared.kb_backend.managed_backend import retrieval_configuration + + config = retrieval_configuration(top_k=5) + managed = config["managedSearchConfiguration"] + assert "filter" not in managed, ( + "the default retrieval configuration carries a filter; if isolation " + "depended on it, omitting it would cross a tenant boundary" + ) + assert "vectorSearchConfiguration" not in config + + def test_an_isolation_critical_filter_must_be_exact_match(self): + """Requirement 11.5. Prefix and substring operators over-match silently: + a filter isolating ``ast-1`` also admits ``ast-10``.""" + from apis.shared.kb_backend.managed_backend import ( + UnsafeFilterOperator, + validate_isolation_filter, + ) + + validate_isolation_filter({"equals": {"key": "document_id", "value": "doc-1"}}) + with pytest.raises(UnsafeFilterOperator): + validate_isolation_filter({"startsWith": {"key": "document_id", "value": "doc-"}}) + with pytest.raises(UnsafeFilterOperator): + validate_isolation_filter( + {"andAll": [{"equals": {}}, {"stringContains": {}}]} + ) + + +# ── 4. Resource policies ───────────────────────────────────────────────────── +AWS_KB_ID = "KB1234567890" +NEW_AWS_KB_ID = "KB0987654321" +PRINCIPAL = "arn:aws:iam::123456789012:role/test-project-agentcore-runtime-role" +POLICY_ENV = { + "AWS_ACCOUNT_ID": "123456789012", + "AWS_REGION": "us-west-2", + "MANAGED_KB_RETRIEVAL_PRINCIPAL_ARNS": PRINCIPAL, +} + + +class TestResourcePolicyDocument: + def test_the_policy_names_principals_and_a_single_resource(self): + arn = knowledge_base_arn(AWS_KB_ID, "us-west-2", "123456789012") + document = retrieve_policy_document(arn, [PRINCIPAL]) + statement = document["Statement"][0] + + assert statement["Effect"] == "Allow" + assert statement["Principal"] == {"AWS": [PRINCIPAL]} + assert statement["Action"] == list(RETRIEVE_ACTIONS) + assert statement["Resource"] == arn + + def test_there_is_no_branch_that_produces_a_wildcard(self): + """The CDK grant cannot condition on a policy's contents, so this + function is the only thing standing between "narrow the blast radius" + and "widen it". Serialized and searched, because a wildcard could hide + in a nested structure a key-by-key assertion would miss.""" + arn = knowledge_base_arn(AWS_KB_ID, "us-west-2", "123456789012") + body = json.dumps(retrieve_policy_document(arn, [PRINCIPAL])) + assert '"*"' not in body + assert "arn:aws:iam::*" not in body + + def test_an_empty_principal_list_is_refused(self): + with pytest.raises(ResourcePolicyError): + retrieve_policy_document("arn:aws:bedrock:us-west-2:1:knowledge-base/x", []) + + def test_an_unknown_account_is_named_rather_than_guessed(self): + with patch.dict("os.environ", {}, clear=True): + with pytest.raises(ResourcePolicyError, match="AWS_ACCOUNT_ID"): + knowledge_base_arn(AWS_KB_ID) + + def test_principals_are_deduplicated_and_ordered(self): + """So the same configuration always produces the same document; otherwise + every call looks like a change and nothing can be compared.""" + with patch.dict( + "os.environ", + {"MANAGED_KB_RETRIEVAL_PRINCIPAL_ARNS": f" {PRINCIPAL} ,{PRINCIPAL},"}, + clear=False, + ): + assert retrieval_principals() == (PRINCIPAL,) + + +class TestResourcePolicyStaleness: + def test_a_new_aws_kb_id_makes_the_policy_stale(self): + """Requirement 25.7 / 24.12. A policy attaches to an ARN, so a + replacement identifier silently drops sharing.""" + assert policy_is_stale({"awsKbId": NEW_AWS_KB_ID, POLICY_KB_ID_ATTR: AWS_KB_ID}) is True + + def test_a_matching_target_is_not_stale(self): + assert policy_is_stale({"awsKbId": AWS_KB_ID, POLICY_KB_ID_ATTR: AWS_KB_ID}) is False + + def test_never_applied_is_stale(self): + assert policy_is_stale({"awsKbId": AWS_KB_ID}) is True + + def test_an_unprovisioned_record_is_not_stale(self): + """Nothing exists yet, so there is nothing to be stale against — lazy + provisioning makes this the ordinary state, not an error.""" + assert policy_is_stale({}) is False + assert policy_is_stale(None) is False + + +class TestEnsureRetrievePolicy: + @pytest.mark.asyncio + async def test_a_shared_knowledge_base_gets_a_policy(self): + client = MagicMock() + client.put_resource_policy.return_value = {"revisionId": "rev-1"} + record = {"awsKbId": AWS_KB_ID} + + with patch.dict("os.environ", POLICY_ENV, clear=False), patch( + "apis.shared.kb_backend.records.set_resource_policy_state" + ) as setter: + revision = await ensure_retrieve_policy( + ASSISTANT_ID, ASSISTANT_ID, shared=True, record=record, client=client + ) + + assert revision == "rev-1" + arn = client.put_resource_policy.call_args.kwargs["resourceArn"] + assert arn.endswith(f"knowledge-base/{AWS_KB_ID}") + setter.assert_called_once_with(ASSISTANT_ID, ASSISTANT_ID, AWS_KB_ID, "rev-1") + + @pytest.mark.asyncio + async def test_a_private_knowledge_base_gets_none(self): + """A policy on a single-owner corpus restricts nothing the assistant's + own access check does not already restrict.""" + client = MagicMock() + + with patch.dict("os.environ", POLICY_ENV, clear=False): + revision = await ensure_retrieve_policy( + ASSISTANT_ID, + ASSISTANT_ID, + shared=False, + record={"awsKbId": AWS_KB_ID}, + client=client, + ) + + assert revision is None + client.put_resource_policy.assert_not_called() + client.delete_resource_policy.assert_not_called() + + @pytest.mark.asyncio + async def test_unsharing_removes_the_policy_and_forgets_the_target(self): + client = MagicMock() + record = {"awsKbId": AWS_KB_ID, POLICY_KB_ID_ATTR: AWS_KB_ID} + + with patch.dict("os.environ", POLICY_ENV, clear=False), patch( + "apis.shared.kb_backend.records.set_resource_policy_state" + ) as setter: + await ensure_retrieve_policy( + ASSISTANT_ID, ASSISTANT_ID, shared=False, record=record, client=client + ) + + client.delete_resource_policy.assert_called_once() + setter.assert_called_once_with(ASSISTANT_ID, ASSISTANT_ID, None, None) + + @pytest.mark.asyncio + async def test_a_current_policy_makes_no_aws_call(self): + """The common path. Callers may invoke this freely only if it is cheap.""" + client = MagicMock() + record = {"awsKbId": AWS_KB_ID, POLICY_KB_ID_ATTR: AWS_KB_ID, POLICY_REVISION_ATTR: "rev-1"} + + with patch.dict("os.environ", POLICY_ENV, clear=False): + revision = await ensure_retrieve_policy( + ASSISTANT_ID, ASSISTANT_ID, shared=True, record=record, client=client + ) + + assert revision == "rev-1" + client.put_resource_policy.assert_not_called() + + @pytest.mark.asyncio + async def test_a_rehydration_reapplies_to_the_new_identifier(self): + """Requirement 24.12, the reason this whole mechanism is state-based. + + The record still names the *old* target. Nothing fired an event; the + mismatch alone is enough to repair it. + """ + client = MagicMock() + client.put_resource_policy.return_value = {"revisionId": "rev-2"} + record = {"awsKbId": NEW_AWS_KB_ID, POLICY_KB_ID_ATTR: AWS_KB_ID} + + with patch.dict("os.environ", POLICY_ENV, clear=False), patch( + "apis.shared.kb_backend.records.set_resource_policy_state" + ) as setter: + revision = await ensure_retrieve_policy( + ASSISTANT_ID, ASSISTANT_ID, shared=True, record=record, client=client + ) + + assert revision == "rev-2" + arn = client.put_resource_policy.call_args.kwargs["resourceArn"] + assert arn.endswith(f"knowledge-base/{NEW_AWS_KB_ID}"), ( + "the policy was re-applied to the old identifier, so sharing is " + "still attached to a knowledge base nobody reads" + ) + setter.assert_called_once_with(ASSISTANT_ID, ASSISTANT_ID, NEW_AWS_KB_ID, "rev-2") + + @pytest.mark.asyncio + async def test_an_unprovisioned_shared_knowledge_base_is_not_an_error(self): + client = MagicMock() + with patch.dict("os.environ", POLICY_ENV, clear=False): + assert ( + await ensure_retrieve_policy( + ASSISTANT_ID, ASSISTANT_ID, shared=True, record={}, client=client + ) + is None + ) + client.put_resource_policy.assert_not_called() + + @pytest.mark.asyncio + async def test_no_configured_principals_applies_nothing(self): + """Rather than inventing one, which would either widen access or lock + the platform out of its own corpus.""" + client = MagicMock() + with patch.dict( + "os.environ", + {**POLICY_ENV, "MANAGED_KB_RETRIEVAL_PRINCIPAL_ARNS": ""}, + clear=False, + ): + revision = await ensure_retrieve_policy( + ASSISTANT_ID, + ASSISTANT_ID, + shared=True, + record={"awsKbId": AWS_KB_ID}, + client=client, + ) + + assert revision is None + client.put_resource_policy.assert_not_called() + + +class TestSharedBeyondOwner: + @pytest.mark.asyncio + @pytest.mark.parametrize("visibility", ["PUBLIC", "SHARED"]) + async def test_visibility_alone_can_answer_yes(self, visibility): + assert await is_shared_beyond_owner(ASSISTANT_ID, USER_ID, visibility) is True + + @pytest.mark.asyncio + async def test_a_private_assistant_with_a_share_record_is_shared(self): + """The case visibility alone misses, and the one most likely to exist: + ``resolve_assistant_permission`` resolves an editor share on a PRIVATE + assistant to ``editor``.""" + + async def _shares(assistant_id, owner_id): + return [{"email": "someone@else.test", "permission": "editor"}] + + with patch( + "apis.shared.assistants.service.list_assistant_shares", side_effect=_shares + ): + assert await is_shared_beyond_owner(ASSISTANT_ID, USER_ID, "PRIVATE") is True + + @pytest.mark.asyncio + async def test_a_private_assistant_with_no_shares_is_not_shared(self): + async def _shares(assistant_id, owner_id): + return [] + + with patch( + "apis.shared.assistants.service.list_assistant_shares", side_effect=_shares + ): + assert await is_shared_beyond_owner(ASSISTANT_ID, USER_ID, "PRIVATE") is False + + @pytest.mark.asyncio + async def test_an_error_assumes_shared(self): + """Fails toward the narrowing policy: one extra control-plane call + against leaving a multi-user corpus account-readable.""" + + async def _boom(assistant_id, owner_id): + raise RuntimeError("dynamodb unavailable") + + with patch("apis.shared.assistants.service.list_assistant_shares", side_effect=_boom): + assert await is_shared_beyond_owner(ASSISTANT_ID, USER_ID, "PRIVATE") is True + + +# ── 5. Publication semantics ───────────────────────────────────────────────── +class TestPublishedAgentCorpusBehaviour: + """Requirement 24.14.""" + + def test_an_engine_swap_is_not_a_corpus_change(self): + """Requirement 25.8. Parity is the contract, so a swap needs no re-review. + If a future change makes engines return different results, this is where + the argument has to be had.""" + assert migration_requires_review("s3vectors", "managed") is False + assert migration_requires_review(None, "managed") is False + + def test_an_engine_swap_does_not_change_what_is_retrieved(self): + """Asserted where it is observable: the facade applies the same rules to + whatever comes back across the seam, so two backends returning the same + chunks produce the same response — the published-agent guarantee.""" + import asyncio + + chunks = [_chunk("doc-a"), _chunk("doc-b")] + legacy = RecordingBackend(chunks) + managed = RecordingBackend(chunks) + access = granted(ASSISTANT_ID, USER_ID, "viewer") + + before, _ = asyncio.run(_search(access, legacy)) + after, _ = asyncio.run(_search(access, managed)) + + assert before == after + assert [r["text"] for r in after] == [c.text for c in chunks] + + +class TestReclaimExemption: + """Requirements 25.9, 25.10.""" + + def test_an_agent_on_the_shelf_is_exempt(self): + assert is_reclaim_exempt({}, "published", 3) is True + + def test_a_live_listing_in_changes_requested_is_still_exempt(self): + """The trap ``is_on_shelf`` exists for: an admin requesting changes on a + live listing leaves it serving but moves its state out of + ``LISTED_STATES``. Keyed on ``is_listed``, a reclaim pass would delete + the corpus behind an agent users can still see.""" + assert is_reclaim_exempt({}, "changes_requested", 3) is True + + def test_a_taken_down_agent_is_not_exempt_by_state_alone(self): + """Requirement 25.10: reaching reclaim eligibility requires the listing + machine's explicit ``taken_down`` edge, which clears + ``published_version`` in the same breath. Nothing infers a takedown.""" + assert is_reclaim_exempt({}, "taken_down", None) is False + + def test_a_private_agent_is_not_exempt(self): + assert is_reclaim_exempt({}, None, None) is False + assert is_reclaim_exempt({}, "private", None) is False + + @pytest.mark.parametrize("attribute", ["exemptFromReclaim", "pinned"]) + def test_an_explicit_hold_is_exempt_regardless_of_listing(self, attribute): + assert is_reclaim_exempt({attribute: True}, None, None) is True + + def test_an_unreadable_record_is_exempt(self): + """Fail-closed applied to deletion, where it matters more than anywhere + else: reclaim acts on knowledge bases it can describe.""" + assert is_reclaim_exempt(None, "private", None) is True + + def test_the_reason_is_actionable(self): + """A report-only pass that says "skipped 400" tells an operator nothing.""" + assert "shelf" in reclaim_exemption_reason({}, "published", 1) + assert "pinned" in reclaim_exemption_reason({"pinned": True}) + assert reclaim_exemption_reason({}, "private", None) is None diff --git a/backend/tests/shared/test_kb_backend_parity.py b/backend/tests/shared/test_kb_backend_parity.py new file mode 100644 index 000000000..f1dec3154 --- /dev/null +++ b/backend/tests/shared/test_kb_backend_parity.py @@ -0,0 +1,620 @@ +""" +Parity contract tests: the rules hold on the managed path, not just legacy. + +Requirement 3 says a migrated knowledge base must behave exactly as it did +before, parser quality aside. Four properties carry that promise, and all four +are owned by the **facade** in ``rag_service`` rather than by either adapter — +which is the point of these tests. A rule implemented inside the legacy adapter +would be a rule the managed adapter silently lacks, and the symptom would be a +quality difference that looks like the engine swap's fault. + +The managed backend itself is task 8.3. These tests use a protocol-conforming +stand-in, which is sufficient and in fact preferable: what is under test is +whether the *facade* applies ``top_k``, the status filter, the 2,000-character +cap and the 500-character citation clip to whatever comes back across the seam. +A stand-in proves that without letting Bedrock's wire format into the assertion. + +Feature: managed-kb-migration +Requirements: 3.1, 3.2, 3.3, 3.4 +""" + +import asyncio +from typing import Any, Dict, List +from unittest.mock import MagicMock, patch + +import pytest + +from apis.shared.assistants.kb_access import granted +from apis.shared.assistants.rag_service import ( + MAX_CONTEXT_CHARS, + augment_prompt_with_context, + search_assistant_knowledgebase_with_formatting, +) +from apis.shared.kb_backend.protocol import ( + DEFAULT_TOP_K, + Chunk, + KnowledgeBaseBackend, +) +from apis.shared.kb_backend.records import ENGINE_MANAGED +from apis.shared.kb_backend.resolver import register_backend, unregister_backend + +ASSISTANT_ID = "ast-parity-001" +TABLE_NAME = "test-table" + +#: Every parity test runs as a user who is allowed to read, because parity is a +#: claim about *authorized* retrieval on both backends. The access gate itself is +#: covered by ``test_kb_authorization.py``; going through ``granted`` rather than +#: constructing a ``KbAccess`` directly keeps these tests honest about the one +#: door into the type. +OWNER_ACCESS = granted(ASSISTANT_ID, "user-parity", "owner") + +#: The clip applied when the inference API turns chunks into citation events +#: (``inference_api/chat/routes.py``: ``chunk.get("text", "")[:500]``). +#: Duplicated as a constant here deliberately — asserting the number the route +#: uses is what makes a change to it visible (Requirement 3.4). +CITATION_EXCERPT_CHARS = 500 + + +class FakeManagedBackend: + """Protocol-conforming stand-in for the managed backend (task 8.3). + + Records the ``top_k`` it was asked for, so the facade's request can be + asserted rather than assumed, and returns however many chunks the test + wants — including more than ``top_k``, which is how "the facade narrows" + becomes an observable claim. + """ + + def __init__(self, chunks: List[Chunk]): + self._chunks = chunks + self.calls: List[Dict[str, Any]] = [] + + async def search(self, kb_ref: str, query: str, top_k: int = DEFAULT_TOP_K) -> List[Chunk]: + self.calls.append({"kb_ref": kb_ref, "query": query, "top_k": top_k}) + return list(self._chunks) + + async def ingest(self, kb_ref: str, source) -> None: # pragma: no cover - unused + raise NotImplementedError + + async def delete_document(self, kb_ref: str, document_id: str) -> None: # pragma: no cover + raise NotImplementedError + + +def _managed_chunk(document_id: str, index: int = 0, text: str = None, relevance: float = None) -> Chunk: + """A chunk as the managed backend would produce it: native relevance.""" + return Chunk( + text=text if text is not None else f"passage from {document_id}", + # Higher is better, descending with rank. + relevance=relevance if relevance is not None else 1.0 - (index * 0.01), + document_id=document_id, + metadata={"document_id": document_id, "text": text if text is not None else f"passage from {document_id}"}, + key=f"{document_id}#{index}", + ) + + +@pytest.fixture +def managed_kb(request): + """Route ``ASSISTANT_ID`` to a fake managed backend for one test. + + Registers the stand-in under the ``managed`` engine and makes the KB_Record + lookup report that engine, so resolution takes the managed branch for real + rather than being bypassed. + """ + created: List[FakeManagedBackend] = [] + + def _install(chunks: List[Chunk]) -> FakeManagedBackend: + backend = FakeManagedBackend(chunks) + register_backend(ENGINE_MANAGED, backend) + created.append(backend) + return backend + + yield _install + + unregister_backend(ENGINE_MANAGED) + assert created, "fixture used without installing a backend" + + +def _patch_record_and_statuses(status_map: Dict[str, str]): + """Patch KB_Record resolution to 'managed' and DOC# statuses to *status_map*. + + Both reads hit the same assistants table, so one mock serves both: ``KB#`` + returns the managed engine and ``DOC#`` returns the requested status. + """ + def _get_item(**kwargs): + sk = kwargs["Key"]["SK"] + if sk.startswith("KB#"): + return {"Item": {"retrievalEngine": ENGINE_MANAGED}} + doc_id = sk.replace("DOC#", "") + if doc_id in status_map: + return {"Item": {"status": status_map[doc_id]}} + return {} + + table = MagicMock() + table.get_item = MagicMock(side_effect=_get_item) + resource = MagicMock() + resource.Table.return_value = table + return patch("boto3.resource", return_value=resource), table + + +def _patch_legacy_record_and_statuses(status_map: Dict[str, str]): + """As above, but the KB_Record is absent — so resolution must pick legacy. + + Absence, not an explicit ``"s3vectors"`` value: that is the shape every + knowledge base predating this feature has, and the one the resolver's default + exists for. + """ + def _get_item(**kwargs): + sk = kwargs["Key"]["SK"] + if sk.startswith("KB#"): + return {} + doc_id = sk.replace("DOC#", "") + if doc_id in status_map: + return {"Item": {"status": status_map[doc_id]}} + return {} + + table = MagicMock() + table.get_item = MagicMock(side_effect=_get_item) + resource = MagicMock() + resource.Table.return_value = table + return patch("boto3.resource", return_value=resource), table + + +# --------------------------------------------------------------------------- +# Requirement 3.1 — top_k = 5 on both backends +# --------------------------------------------------------------------------- + + +def test_managed_path_requests_top_k_five(managed_kb): + """The facade asks the managed backend for five chunks, as it does legacy.""" + backend = managed_kb([_managed_chunk("doc-a", i) for i in range(5)]) + boto_patch, _ = _patch_record_and_statuses({"doc-a": "complete"}) + + with patch.dict("os.environ", {"DYNAMODB_ASSISTANTS_TABLE_NAME": TABLE_NAME}), boto_patch: + asyncio.run(search_assistant_knowledgebase_with_formatting(ASSISTANT_ID, "q", access=OWNER_ACCESS)) + + assert backend.calls, "the managed backend was never reached" + assert backend.calls[0]["top_k"] == DEFAULT_TOP_K + assert DEFAULT_TOP_K == 5, "the parity contract pins top_k at 5" + + +def test_managed_path_narrows_results_to_top_k(managed_kb): + """An over-generous backend is narrowed by the facade, not trusted. + + The stand-in returns nine chunks. If the facade stopped slicing, all nine + would reach the model and the context budget would be spent on four chunks + nobody asked for. + """ + managed_kb([_managed_chunk(f"doc-{i}", 0) for i in range(9)]) + statuses = {f"doc-{i}": "complete" for i in range(9)} + boto_patch, _ = _patch_record_and_statuses(statuses) + + with patch.dict("os.environ", {"DYNAMODB_ASSISTANTS_TABLE_NAME": TABLE_NAME}), boto_patch: + results = asyncio.run( + search_assistant_knowledgebase_with_formatting(ASSISTANT_ID, "q", access=OWNER_ACCESS) + ) + + assert len(results) == DEFAULT_TOP_K + + +def test_managed_path_narrowing_happens_after_the_status_filter(managed_kb): + """Filter first, slice second — so an incomplete doc cannot shrink the answer. + + Six chunks come back and the first is from a deleted document. Slicing before + filtering would yield four usable chunks; filtering first yields five. + """ + chunks = [_managed_chunk("doc-gone", 0)] + [_managed_chunk(f"doc-{i}", 0) for i in range(5)] + managed_kb(chunks) + statuses = {f"doc-{i}": "complete" for i in range(5)} + statuses["doc-gone"] = "deleting" + boto_patch, _ = _patch_record_and_statuses(statuses) + + with patch.dict("os.environ", {"DYNAMODB_ASSISTANTS_TABLE_NAME": TABLE_NAME}), boto_patch: + results = asyncio.run( + search_assistant_knowledgebase_with_formatting(ASSISTANT_ID, "q", access=OWNER_ACCESS) + ) + + assert len(results) == DEFAULT_TOP_K + assert "doc-gone" not in {r["metadata"]["document_id"] for r in results} + + +# --------------------------------------------------------------------------- +# Requirement 3.3 — the document status filter runs on the managed path +# --------------------------------------------------------------------------- + + +def test_managed_path_applies_document_status_filter(managed_kb): + """Chunks from non-complete documents are dropped on the managed path too.""" + managed_kb([ + _managed_chunk("doc-ok", 0), + _managed_chunk("doc-deleting", 0), + _managed_chunk("doc-failed", 0), + ]) + boto_patch, _ = _patch_record_and_statuses({ + "doc-ok": "complete", + "doc-deleting": "deleting", + "doc-failed": "failed", + }) + + with patch.dict("os.environ", {"DYNAMODB_ASSISTANTS_TABLE_NAME": TABLE_NAME}), boto_patch: + results = asyncio.run( + search_assistant_knowledgebase_with_formatting(ASSISTANT_ID, "q", access=OWNER_ACCESS) + ) + + assert [r["metadata"]["document_id"] for r in results] == ["doc-ok"] + + +def test_managed_path_drops_chunks_for_missing_document_records(managed_kb): + """A document with no DOC# record is not retrievable on the managed path.""" + managed_kb([_managed_chunk("doc-ok", 0), _managed_chunk("doc-absent", 0)]) + boto_patch, _ = _patch_record_and_statuses({"doc-ok": "complete"}) + + with patch.dict("os.environ", {"DYNAMODB_ASSISTANTS_TABLE_NAME": TABLE_NAME}), boto_patch: + results = asyncio.run( + search_assistant_knowledgebase_with_formatting(ASSISTANT_ID, "q", access=OWNER_ACCESS) + ) + + assert [r["metadata"]["document_id"] for r in results] == ["doc-ok"] + + +# --------------------------------------------------------------------------- +# Requirement 3.4 — result shape, score direction, and the citation clip +# --------------------------------------------------------------------------- + + +def test_managed_path_emits_the_legacy_result_shape(managed_kb): + """The keys callers read are unchanged, including the derived ``distance``. + + ``app_api/assistants/routes.py`` puts ``chunk.get("distance")`` in a response + body, so the field survives the rename to ``relevance`` at the seam. Managed + relevance of ``1.0`` must present as distance ``-1.0`` — the exact negation, + the same transform the legacy adapter inverts. + """ + managed_kb([_managed_chunk("doc-ok", 0, text="hello", relevance=1.0)]) + boto_patch, _ = _patch_record_and_statuses({"doc-ok": "complete"}) + + with patch.dict("os.environ", {"DYNAMODB_ASSISTANTS_TABLE_NAME": TABLE_NAME}), boto_patch: + results = asyncio.run( + search_assistant_knowledgebase_with_formatting(ASSISTANT_ID, "q", access=OWNER_ACCESS) + ) + + assert len(results) == 1 + assert set(results[0]) == {"text", "distance", "metadata", "key"} + assert results[0]["text"] == "hello" + assert results[0]["key"] == "doc-ok#0" + assert results[0]["distance"] == -1.0 + + +def test_managed_path_preserves_backend_ranking_order(managed_kb): + """The facade does not reorder; the backend's ranking is what callers see.""" + managed_kb([ + _managed_chunk("doc-best", 0, relevance=0.9), + _managed_chunk("doc-mid", 0, relevance=0.5), + _managed_chunk("doc-worst", 0, relevance=0.1), + ]) + boto_patch, _ = _patch_record_and_statuses({ + "doc-best": "complete", + "doc-mid": "complete", + "doc-worst": "complete", + }) + + with patch.dict("os.environ", {"DYNAMODB_ASSISTANTS_TABLE_NAME": TABLE_NAME}), boto_patch: + results = asyncio.run( + search_assistant_knowledgebase_with_formatting(ASSISTANT_ID, "q", access=OWNER_ACCESS) + ) + + assert [r["metadata"]["document_id"] for r in results] == [ + "doc-best", + "doc-mid", + "doc-worst", + ] + + +def test_citation_excerpt_clip_holds_on_managed_results(managed_kb): + """Citations built from managed chunks clip the excerpt at 500 characters. + + Mirrors ``inference_api/chat/routes.py``'s citation construction. Both + backends feed the same ``context_chunks`` structure into it, so the clip is + a property of the shared shape rather than of either engine. + """ + long_text = "x" * 2000 + managed_kb([_managed_chunk("doc-long", 0, text=long_text)]) + boto_patch, _ = _patch_record_and_statuses({"doc-long": "complete"}) + + with patch.dict("os.environ", {"DYNAMODB_ASSISTANTS_TABLE_NAME": TABLE_NAME}), boto_patch: + results = asyncio.run( + search_assistant_knowledgebase_with_formatting(ASSISTANT_ID, "q", access=OWNER_ACCESS) + ) + + excerpt = results[0].get("text", "")[:CITATION_EXCERPT_CHARS] + assert len(excerpt) == CITATION_EXCERPT_CHARS + assert CITATION_EXCERPT_CHARS == 500, "the parity contract pins the clip at 500" + + +# --------------------------------------------------------------------------- +# Requirement 3.2 — the 2,000-character context cap +# --------------------------------------------------------------------------- + + +def test_context_cap_is_two_thousand_characters(): + """The cap is a named constant pinned at 2,000, on every backend.""" + assert MAX_CONTEXT_CHARS == 2000 + + +def test_context_cap_bounds_augmented_prompt_from_managed_chunks(): + """Managed chunks are capped by the facade exactly as legacy chunks are. + + Five 1,000-character chunks total 5,000 characters of context. The augmented + prompt must carry at most 2,000 of them, so the cap is observable as the + difference between the prompt's length and the corpus's. + """ + chunks = [ + {"text": "y" * 1000, "distance": -0.9, "metadata": {"document_id": f"doc-{i}"}, "key": f"doc-{i}#0"} + for i in range(5) + ] + user_message = "what does the corpus say?" + + augmented = augment_prompt_with_context(user_message=user_message, context_chunks=chunks) + + context_only = augmented.split("---\nUser Question:")[0] + body_length = len(context_only) - len( + "The following context is retrieved from the assistant's knowledge base. " + "Use this information to answer the user's question accurately and " + "comprehensively.\n\n" + ) + assert body_length <= MAX_CONTEXT_CHARS + 64, ( + f"context body is {body_length} chars; the 2,000-character cap is not " + f"being applied to managed chunks" + ) + assert augmented.count("y") < 5000, "all five chunks were included uncapped" + assert user_message in augmented + + +def test_context_cap_default_is_not_overridable_by_callers_accidentally(): + """An explicit larger cap is honoured, proving the default is what binds. + + Guards against the cap appearing to hold because the text was short: with the + cap raised, the same corpus produces a longer prompt. + """ + chunks = [{"text": "z" * 1000, "metadata": {"document_id": "d"}, "key": "d#0"} for _ in range(5)] + + capped = augment_prompt_with_context("q", chunks) + raised = augment_prompt_with_context("q", chunks, max_context_length=10000) + + assert len(raised) > len(capped) + assert capped.count("z") <= MAX_CONTEXT_CHARS + + +# --------------------------------------------------------------------------- +# The same rules, on the legacy path — the refactor's regression guard +# --------------------------------------------------------------------------- + + +def _s3_response(hits: List[Dict[str, Any]]) -> Dict[str, Any]: + return {"vectors": hits} + + +def test_legacy_path_emits_identical_values_to_the_pre_seam_formatter(): + """The legacy path's output is unchanged by the refactor, value for value. + + This is the guard on "zero behaviour change". The facade now receives + ``relevance`` and derives ``distance`` from it, and + ``app_api/assistants/routes.py`` forwards that number into an HTTP response + body — so it must be the *same* float, not a nearby one. ``0.1`` is chosen + deliberately: a ``1.0 - x`` conversion round-trips it to + ``0.09999999999999998`` and this assertion is what notices. + """ + hits = [ + { + "key": "doc-a#0", + "distance": 0.1, + "metadata": {"document_id": "doc-a", "text": "alpha", "source": "a.pdf"}, + }, + { + "key": "doc-a#1", + "distance": 0.30000000000000004, + "metadata": {"document_id": "doc-a", "text": "beta", "source": "a.pdf"}, + }, + ] + boto_patch, _ = _patch_legacy_record_and_statuses({"doc-a": "complete"}) + + with ( + patch.dict("os.environ", {"DYNAMODB_ASSISTANTS_TABLE_NAME": TABLE_NAME}), + boto_patch, + patch( + "apis.shared.embeddings.bedrock_embeddings.search_assistant_knowledgebase", + return_value=_s3_response(hits), + ), + ): + results = asyncio.run( + search_assistant_knowledgebase_with_formatting(ASSISTANT_ID, "q", access=OWNER_ACCESS) + ) + + expected = [ + { + "text": hit["metadata"]["text"], + "distance": hit["distance"], + "metadata": hit["metadata"], + "key": hit["key"], + } + for hit in hits + ] + assert results == expected + + +def test_legacy_path_applies_the_same_filter_and_top_k(): + """The legacy path goes through the same facade rules, not a bypass.""" + hits = [ + {"key": f"doc-{i}#0", "distance": i / 10, "metadata": {"document_id": f"doc-{i}", "text": f"t{i}"}} + for i in range(8) + ] + statuses = {f"doc-{i}": "complete" for i in range(8)} + statuses["doc-0"] = "deleting" + boto_patch, _ = _patch_legacy_record_and_statuses(statuses) + + with ( + patch.dict("os.environ", {"DYNAMODB_ASSISTANTS_TABLE_NAME": TABLE_NAME}), + boto_patch, + patch( + "apis.shared.embeddings.bedrock_embeddings.search_assistant_knowledgebase", + return_value=_s3_response(hits), + ), + ): + results = asyncio.run( + search_assistant_knowledgebase_with_formatting(ASSISTANT_ID, "q", access=OWNER_ACCESS) + ) + + assert len(results) == DEFAULT_TOP_K + assert "doc-0" not in {r["metadata"]["document_id"] for r in results} + + +def test_legacy_path_preserves_a_missing_distance_as_none(): + """A hit without a distance still surfaces ``distance: None``, as before. + + Fabricating a score would be worse than reporting none: ``0.0`` is a perfect + match, so a default would promote an unscored chunk to best in the list. + """ + hits = [{"key": "doc-a#0", "metadata": {"document_id": "doc-a", "text": "alpha"}}] + boto_patch, _ = _patch_legacy_record_and_statuses({"doc-a": "complete"}) + + with ( + patch.dict("os.environ", {"DYNAMODB_ASSISTANTS_TABLE_NAME": TABLE_NAME}), + boto_patch, + patch( + "apis.shared.embeddings.bedrock_embeddings.search_assistant_knowledgebase", + return_value=_s3_response(hits), + ), + ): + results = asyncio.run( + search_assistant_knowledgebase_with_formatting(ASSISTANT_ID, "q", access=OWNER_ACCESS) + ) + + assert results[0]["distance"] is None + + +def test_empty_backend_result_returns_empty_list(): + """No hits ⇒ empty list, without reaching the status filter, as before.""" + boto_patch, table = _patch_legacy_record_and_statuses({}) + + with ( + patch.dict("os.environ", {"DYNAMODB_ASSISTANTS_TABLE_NAME": TABLE_NAME}), + boto_patch, + patch( + "apis.shared.embeddings.bedrock_embeddings.search_assistant_knowledgebase", + return_value=_s3_response([]), + ), + ): + results = asyncio.run( + search_assistant_knowledgebase_with_formatting(ASSISTANT_ID, "q", access=OWNER_ACCESS) + ) + + assert results == [] + doc_lookups = [c for c in table.get_item.call_args_list if c.kwargs["Key"]["SK"].startswith("DOC#")] + assert doc_lookups == [], "the status filter ran on an empty result set" + + +def test_backend_failure_degrades_to_empty_list(): + """A backend exception still yields an empty list, never a 500.""" + boto_patch, _ = _patch_legacy_record_and_statuses({}) + + with ( + patch.dict("os.environ", {"DYNAMODB_ASSISTANTS_TABLE_NAME": TABLE_NAME}), + boto_patch, + patch( + "apis.shared.embeddings.bedrock_embeddings.search_assistant_knowledgebase", + side_effect=RuntimeError("S3 Vectors unavailable"), + ), + ): + results = asyncio.run( + search_assistant_knowledgebase_with_formatting(ASSISTANT_ID, "q", access=OWNER_ACCESS) + ) + + assert results == [] + + +def test_absent_kb_record_resolves_to_the_legacy_backend(): + """No KB_Record ⇒ legacy. The zero-backfill invariant, seen from the facade. + + The managed stand-in is registered but must not be reached: a knowledge base + that has never been enrolled has no opinion, and no opinion means legacy. + """ + unreachable = FakeManagedBackend([_managed_chunk("doc-managed", 0)]) + register_backend(ENGINE_MANAGED, unreachable) + try: + table = MagicMock() + # No Item for either KB# or DOC#: an un-enrolled, un-documented assistant. + table.get_item = MagicMock(return_value={}) + resource = MagicMock() + resource.Table.return_value = table + + with ( + patch.dict("os.environ", {"DYNAMODB_ASSISTANTS_TABLE_NAME": TABLE_NAME}), + patch("boto3.resource", return_value=resource), + patch( + "apis.shared.embeddings.bedrock_embeddings.search_assistant_knowledgebase", + return_value=_s3_response( + [{"key": "doc-a#0", "distance": 0.2, "metadata": {"document_id": "doc-a", "text": "legacy"}}] + ), + ), + ): + results = asyncio.run( + search_assistant_knowledgebase_with_formatting(ASSISTANT_ID, "q", access=OWNER_ACCESS) + ) + finally: + unregister_backend(ENGINE_MANAGED) + + assert unreachable.calls == [], "an un-enrolled knowledge base reached the managed backend" + # doc-a has no DOC# record, so the filter drops it — the point is which + # backend ran, which the empty `calls` above establishes. + assert results == [] + + +def test_existing_record_without_an_engine_attribute_resolves_to_legacy(): + """A *present* KB_Record carrying no ``retrievalEngine`` still means legacy. + + Distinct from the absent-record case above, and the one that matters during + migration: a knowledge base enrolled but not yet promoted has a real record + with real fields and no engine opinion. It must keep reading legacy until the + single promotion write lands, or a half-migrated corpus starts serving from a + managed index that catch-up has not finished filling. + """ + unreachable = FakeManagedBackend([_managed_chunk("doc-managed", 0)]) + register_backend(ENGINE_MANAGED, unreachable) + try: + def _get_item(**kwargs): + sk = kwargs["Key"]["SK"] + if sk.startswith("KB#"): + # Enrolled, provisioning done, engine not yet promoted. + return {"Item": {"appKbId": ASSISTANT_ID, "provisioningState": "active"}} + return {"Item": {"status": "complete"}} + + table = MagicMock() + table.get_item = MagicMock(side_effect=_get_item) + resource = MagicMock() + resource.Table.return_value = table + + with ( + patch.dict("os.environ", {"DYNAMODB_ASSISTANTS_TABLE_NAME": TABLE_NAME}), + patch("boto3.resource", return_value=resource), + patch( + "apis.shared.embeddings.bedrock_embeddings.search_assistant_knowledgebase", + return_value=_s3_response( + [{"key": "doc-a#0", "distance": 0.2, "metadata": {"document_id": "doc-a", "text": "legacy"}}] + ), + ), + ): + results = asyncio.run( + search_assistant_knowledgebase_with_formatting(ASSISTANT_ID, "q", access=OWNER_ACCESS) + ) + finally: + unregister_backend(ENGINE_MANAGED) + + assert unreachable.calls == [], "an unpromoted knowledge base reached the managed backend" + assert [r["metadata"]["document_id"] for r in results] == ["doc-a"] + + +# --------------------------------------------------------------------------- +# The stand-in really does conform to the seam +# --------------------------------------------------------------------------- + + +def test_fake_managed_backend_conforms_to_protocol(): + assert isinstance(FakeManagedBackend([]), KnowledgeBaseBackend) diff --git a/backend/tests/shared/test_kb_binding_freeze.py b/backend/tests/shared/test_kb_binding_freeze.py new file mode 100644 index 000000000..66b79049a --- /dev/null +++ b/backend/tests/shared/test_kb_binding_freeze.py @@ -0,0 +1,130 @@ +""" +The agent-to-knowledge-base relationship stays 1:1 during this migration. + +Requirements 6.7, 6.8. This feature does *not* change cardinality: one agent, one +knowledge base, ``App_KB_Id == assistant_id``. That constraint is what lets the +access check be as simple as it is ("can this user invoke this agent" answers "may +this turn retrieve"), and what lets the data model key a knowledge base off the +assistant id with no join. + +The rejections that enforce it already exist — ``binding_validation`` refuses an +explicit ``knowledge_base`` binding and ``bindable_catalog`` serves an empty +palette for the kind. Nothing in this feature touches either. That is exactly why +they are tested here: an untested constraint that some *other* feature relies on +is a constraint that gets relaxed by a plausible-looking change, and the symptom +would surface in this feature's code rather than in the change that caused it. +0..N bindings are F4's problem, and F4 is a separate spec (evaluation §10.6 +forbids coupling the two). + +Feature: managed-kb-migration +Requirements: 6.7, 6.8 +""" + +import pytest + +from apis.app_api.agent_designer.services.bindable_catalog import BINDABLE_KINDS, list_bindable +from apis.app_api.agent_designer.services.binding_validation import ( + BindingValidationError, + validate_agent_write, +) +from apis.shared.assistants.models import AgentBinding +from apis.shared.auth.models import User + + +def _user() -> User: + return User( + user_id="user-binding-freeze", + email="author@example.test", + name="Binding Freeze Author", + roles=["default"], + ) + + +class TestAnExplicitKnowledgeBaseBindingIsRejected: + @pytest.mark.asyncio + async def test_a_knowledge_base_binding_is_refused(self): + """Requirement 6.8. The Designer never offers this, so reaching it means + a hand-rolled request — which is precisely the path that must not open a + second knowledge base onto an agent.""" + with pytest.raises(BindingValidationError) as exc: + await validate_agent_write( + _user(), + bindings=[AgentBinding(kind="knowledge_base", ref="ast-somebody-else")], + ) + + assert exc.value.status_code == 400 + assert "knowledge_base" in str(exc.value) + + @pytest.mark.asyncio + async def test_it_is_refused_even_when_the_ref_is_the_agents_own(self): + """The self-referential case is the tempting one to allow: it looks like a + no-op. It is not — it would make the binding author-settable, and the + moment it is settable the cardinality is no longer enforced by the + absence of a mechanism.""" + with pytest.raises(BindingValidationError): + await validate_agent_write( + _user(), + bindings=[AgentBinding(kind="knowledge_base", ref="ast-self")], + ) + + @pytest.mark.asyncio + async def test_an_empty_binding_list_is_accepted(self): + """Sanity: the rejection above is about the kind, not about validation + refusing everything — otherwise both tests would pass with the + knowledge_base branch deleted.""" + await validate_agent_write(_user(), bindings=[]) + + +class TestTheBindablePaletteOffersNoKnowledgeBases: + @pytest.mark.asyncio + async def test_the_knowledge_base_palette_is_empty(self): + """Requirement 6.8. Welded to the agent, synthesized on read.""" + assert await list_bindable("knowledge_base", _user()) == [] + + def test_the_kind_is_still_a_known_kind(self): + """Empty because it is welded, **not** because it is unrecognized. If the + kind were simply removed, this test would pass while + ``compat.effective_bindings`` — which synthesizes a ``knowledge_base`` + binding on read — kept emitting a kind nothing recognized. + """ + assert "knowledge_base" in BINDABLE_KINDS + + @pytest.mark.asyncio + async def test_an_unknown_kind_still_raises(self): + """So "returns an empty list" cannot be mistaken for "swallows anything".""" + with pytest.raises(ValueError): + await list_bindable("knowledge_bases", _user()) + + +class TestTheSynthesizedBindingStaysOneToOne: + def test_the_synthesized_binding_refs_the_assistant_itself(self): + """``App_KB_Id == assistant_id`` (Requirement 6.5), read off the one + function that produces the binding. When F4 lands, ``ref`` becomes a real + knowledge base id with no shape change — and this test is what will say + so out loud.""" + from apis.shared.assistants.compat import effective_bindings + from apis.shared.assistants.models import Assistant + + assistant = Assistant.model_validate( + { + "assistantId": "ast-one-to-one", + "ownerId": "user-binding-freeze", + "ownerName": "Binding Freeze Author", + "name": "One to one", + "description": "", + "instructions": "", + "vectorIndexId": "idx-1", + "visibility": "PRIVATE", + "createdAt": "2026-07-01T00:00:00Z", + "updatedAt": "2026-07-01T00:00:00Z", + "status": "COMPLETE", + } + ) + + kb_bindings = [b for b in effective_bindings(assistant) if b.kind == "knowledge_base"] + + assert len(kb_bindings) == 1, ( + "an agent must synthesize exactly one knowledge base binding; more " + "than one is the 0..N model this phase deliberately does not build" + ) + assert kb_bindings[0].ref == assistant.assistant_id diff --git a/backend/tests/shared/test_kb_dual_read.py b/backend/tests/shared/test_kb_dual_read.py new file mode 100644 index 000000000..9e373078c --- /dev/null +++ b/backend/tests/shared/test_kb_dual_read.py @@ -0,0 +1,498 @@ +""" +Dual-read pilot: legacy serves, managed is observed, nobody waits. + +Requirement 18. The pilot's value depends entirely on it being safe to leave +switched on, so the tests here are mostly about what it must *not* do. + +The load-bearing one is latency (18.5). Managed ``Retrieve`` measured a +662–695 ms p50 against legacy's 257 ms, so a pilot that awaited both would nearly +triple the retrieval leg of every piloted turn — and it would do so while +producing correct results and passing every other test in this file. It is caught +here by making the managed backend sleep far longer than any test tolerance and +asserting that the facade still returns promptly, which fails if the two calls +are ever gathered instead of detached. + +Feature: managed-kb-migration +Requirements: 18.1, 18.2, 18.3, 18.4, 18.5 +""" + +import asyncio +import time +from typing import Any, Dict, List +from unittest.mock import MagicMock, patch + +import pytest + +from apis.shared.assistants.kb_access import granted +from apis.shared.assistants.rag_service import search_assistant_knowledgebase_with_formatting +from apis.shared.kb_backend.dual_read import ( + DUAL_READ_ATTR, + compare, + is_pilot_enabled, + start_managed_read, +) +from apis.shared.kb_backend.protocol import DEFAULT_TOP_K, Chunk +from apis.shared.kb_backend.records import ENGINE_MANAGED +from apis.shared.kb_backend.resolver import register_backend, unregister_backend + +ASSISTANT_ID = "ast-dualread-001" +TABLE_NAME = "test-table" +ACCESS = granted(ASSISTANT_ID, "user-dualread", "owner") + +#: Longer than any assertion below tolerates. If the facade ever awaits the +#: managed read, the latency test fails on the clock rather than on a value. +MANAGED_DELAY_SECONDS = 3.0 + + +class StubBackend: + """Records its calls; optionally sleeps, or raises, or returns nothing.""" + + def __init__( + self, + chunks: List[Chunk] = None, + delay: float = 0.0, + error: Exception = None, + ): + self._chunks = chunks or [] + self._delay = delay + self._error = error + self.calls: List[Dict[str, Any]] = [] + + async def search(self, kb_ref: str, query: str, top_k: int = DEFAULT_TOP_K) -> List[Chunk]: + self.calls.append({"kb_ref": kb_ref, "query": query, "top_k": top_k}) + if self._delay: + await asyncio.sleep(self._delay) + if self._error: + raise self._error + return list(self._chunks) + + async def ingest(self, kb_ref: str, source) -> None: # pragma: no cover - unused + raise NotImplementedError + + async def delete_document(self, kb_ref: str, document_id: str) -> None: # pragma: no cover + raise NotImplementedError + + +def _chunk(document_id: str, index: int = 0) -> Chunk: + return Chunk( + text=f"passage from {document_id}", + relevance=1.0 - index * 0.01, + document_id=document_id, + metadata={"document_id": document_id}, + key=f"{document_id}#{index}", + ) + + +def _all_documents_complete() -> MagicMock: + table = MagicMock() + table.get_item.return_value = {"Item": {"status": "complete"}} + resource = MagicMock() + resource.Table.return_value = table + return resource + + +def _real_managed_backend(): + """The adapter the resolver installs at import. + + Used to put the registry back after a test deliberately empties it, so one + test cannot leave the process in a state the shipped code never has. + """ + from apis.shared.kb_backend.managed_backend import ManagedKbBackend + + return ManagedKbBackend() + + +@pytest.fixture +def managed_registered(): + """Replace the registered managed backend with a stub for one test. + + The real adapter is installed at import (see the resolver's docstring), so this + fixture *substitutes* rather than introduces — and restores the real one + afterwards, because leaving a stub behind would make every later test in the + process quietly pass against a fake. + """ + def _register(backend: StubBackend) -> StubBackend: + register_backend(ENGINE_MANAGED, backend) + return backend + + yield _register + register_backend(ENGINE_MANAGED, _real_managed_backend()) + + +async def _search(legacy: StubBackend, record: Dict[str, Any], top_k: int = 5): + """Run the facade with ``record`` as the knowledge base's KB_Record.""" + with patch.dict( + "os.environ", {"DYNAMODB_ASSISTANTS_TABLE_NAME": TABLE_NAME}, clear=False + ), patch( + "apis.shared.assistants.rag_service.load_record", return_value=record + ), patch( + "apis.shared.assistants.rag_service.resolve_backend", return_value=legacy + ), patch( + "apis.shared.assistants.rag_service.boto3.resource", + return_value=_all_documents_complete(), + ), patch( + "apis.shared.assistants.rag_service.emit_count" + ), patch( + "apis.shared.kb_backend.dual_read.emit_value" + ) as emit_value, patch( + "apis.shared.kb_backend.dual_read.emit_count" + ) as emit_count: + results = await search_assistant_knowledgebase_with_formatting( + ASSISTANT_ID, "q", top_k, access=ACCESS + ) + # Let the detached comparison run to completion before asserting on it. + # Not a sleep in the facade — a sleep in the *test*, which is the whole + # point: the caller never waits, the assertion does. + for _ in range(200): + await asyncio.sleep(0) + if emit_value.called or emit_count.called: + break + return results, emit_value, emit_count + + +# ── Opt-in, default off ────────────────────────────────────────────────────── +class TestThePilotIsOptIn: + def test_absence_means_off(self): + """Requirement 18.4. Same convention as ``retrievalEngine``: the default + costs nothing to express and nothing to revert.""" + assert is_pilot_enabled(None) is False + assert is_pilot_enabled({}) is False + assert is_pilot_enabled({"appKbId": "x"}) is False + + @pytest.mark.parametrize("value", ["true", "True", 1, "yes", [1]]) + def test_only_a_real_boolean_true_enrols(self, value): + """A truthy value left by a hand-edited record must not enrol a knowledge + base into paying for a second retrieval on every turn. This is the shape + of the reconciler-arming defect: a permissive read of a flag turned a + report-only job into a deleting one.""" + assert is_pilot_enabled({DUAL_READ_ATTR: value}) is False + + def test_boolean_true_enrols(self): + assert is_pilot_enabled({DUAL_READ_ATTR: True}) is True + + @pytest.mark.asyncio + async def test_no_managed_call_when_not_enrolled(self, managed_registered): + managed = managed_registered(StubBackend([_chunk("doc-m")])) + legacy = StubBackend([_chunk("doc-a")]) + + results, _, _ = await _search(legacy, {}) + + assert len(results) == 1 + assert managed.calls == [], "an unenrolled knowledge base was dual-read" + + @pytest.mark.asyncio + async def test_no_managed_call_when_no_managed_backend_is_registered(self): + """A build that cannot serve managed cannot pilot it either. + + The managed backend is registered at import, so this condition has to be + *created* rather than assumed — before that it was assumed, and once + registration landed the test kept passing for the wrong reason: a real + backend was starting, failing against no AWS, and the assertion on + ``emit_value`` could not tell that apart from never starting. + """ + from apis.shared.kb_backend.resolver import backend_for_engine + + legacy = StubBackend([_chunk("doc-a")]) + unregister_backend(ENGINE_MANAGED) + try: + assert backend_for_engine(ENGINE_MANAGED) is None + results, emit_value, emit_count = await _search(legacy, {DUAL_READ_ATTR: True}) + finally: + register_backend(ENGINE_MANAGED, _real_managed_backend()) + + assert len(results) == 1 + assert emit_value.called is False + assert emit_count.called is False, ( + "a failure was counted, which means a managed read was attempted" + ) + + @pytest.mark.asyncio + async def test_the_managed_backend_is_registered_by_default(self): + """The registration this feature would otherwise have shipped without. + + ``register_backend`` existed and nothing called it, so every promoted + knowledge base would have raised ``BackendUnavailable`` — a correct + fail-safe and a useless signal, visible only to the one migrated user. + """ + from apis.shared.kb_backend.resolver import ( + backend_for_engine, + registered_engines, + resolve_backend, + ) + + assert ENGINE_MANAGED in registered_engines() + assert backend_for_engine(ENGINE_MANAGED) is not None + + served = resolve_backend( + ASSISTANT_ID, record={"retrievalEngine": ENGINE_MANAGED} + ) + assert type(served).__name__ == "ManagedKbBackend" + + @pytest.mark.asyncio + async def test_no_dual_read_for_an_already_promoted_knowledge_base(self, managed_registered): + """Comparing managed against managed is the same call twice at twice the + price.""" + managed = managed_registered(StubBackend([_chunk("doc-m")])) + legacy = StubBackend([_chunk("doc-a")]) + + await _search( + legacy, {DUAL_READ_ATTR: True, "retrievalEngine": ENGINE_MANAGED} + ) + + assert managed.calls == [] + + def test_start_managed_read_outside_an_event_loop_returns_none(self): + """Rather than raising. A synchronous caller getting no dual read is a + missing observation; a synchronous caller getting a RuntimeError is a + broken retrieval.""" + assert start_managed_read({DUAL_READ_ATTR: True}, ASSISTANT_ID, "q") is None + + +# ── Legacy is what is served ───────────────────────────────────────────────── +class TestLegacyIsAlwaysServed: + @pytest.mark.asyncio + async def test_the_managed_result_never_reaches_the_caller(self, managed_registered): + """Requirement 18.2. The managed backend returns *different* documents, so + a facade that served them would be caught by identity rather than by + count.""" + managed_registered(StubBackend([_chunk("doc-managed-1"), _chunk("doc-managed-2")])) + legacy = StubBackend([_chunk("doc-legacy-1")]) + + results, _, _ = await _search(legacy, {DUAL_READ_ATTR: True}) + + assert [r["metadata"]["document_id"] for r in results] == ["doc-legacy-1"] + + @pytest.mark.asyncio + async def test_an_empty_legacy_result_is_served_as_empty(self, managed_registered): + """An empty legacy result is a *finding*. Substituting the other engine's + answer would destroy the measurement and change what users see in the same + move.""" + managed_registered(StubBackend([_chunk("doc-managed-1")])) + legacy = StubBackend([]) + + results, _, _ = await _search(legacy, {DUAL_READ_ATTR: True}) + + assert results == [] + + @pytest.mark.asyncio + async def test_a_managed_failure_does_not_fail_the_turn(self, managed_registered): + managed_registered(StubBackend(error=RuntimeError("bedrock threw"))) + legacy = StubBackend([_chunk("doc-legacy-1")]) + + results, _, emit_count = await _search(legacy, {DUAL_READ_ATTR: True}) + + assert len(results) == 1 + emit_count.assert_called() + assert emit_count.call_args.args[0] == "KbDualReadFailed" + + @pytest.mark.asyncio + async def test_a_legacy_failure_still_fails_closed(self, managed_registered): + """And does not leave the managed task orphaned with an unretrieved + exception. + + ``managed.calls`` is empty because the facade cancels the task it started + when legacy raised before the comparison was detached. Without that + cancel the task runs to completion — paying for a Retrieve nobody will + ever read — and Python reports a task whose exception was never + retrieved. + """ + managed = managed_registered(StubBackend([_chunk("doc-managed-1")])) + legacy = StubBackend(error=RuntimeError("s3 vectors threw")) + + results, _, _ = await _search(legacy, {DUAL_READ_ATTR: True}) + + assert results == [] + assert managed.calls == [], ( + "the orphaned managed read was left running after legacy failed" + ) + + +# ── Latency ────────────────────────────────────────────────────────────────── +class TestThePilotAddsNoLatency: + @pytest.mark.asyncio + async def test_the_caller_does_not_wait_for_the_managed_read(self, managed_registered): + """Requirement 18.5, asserted on the clock. + + The managed backend sleeps for three seconds. If the facade gathers the + two calls instead of detaching the comparison, this takes three seconds + and fails; the results themselves would still be correct. + """ + managed_registered(StubBackend([_chunk("doc-managed-1")], delay=MANAGED_DELAY_SECONDS)) + legacy = StubBackend([_chunk("doc-legacy-1")]) + + with patch.dict( + "os.environ", {"DYNAMODB_ASSISTANTS_TABLE_NAME": TABLE_NAME}, clear=False + ), patch( + "apis.shared.assistants.rag_service.load_record", + return_value={DUAL_READ_ATTR: True}, + ), patch( + "apis.shared.assistants.rag_service.resolve_backend", return_value=legacy + ), patch( + "apis.shared.assistants.rag_service.boto3.resource", + return_value=_all_documents_complete(), + ), patch( + "apis.shared.assistants.rag_service.emit_count" + ), patch( + "apis.shared.kb_backend.dual_read.emit_value" + ): + started = time.perf_counter() + results = await search_assistant_knowledgebase_with_formatting( + ASSISTANT_ID, "q", 5, access=ACCESS + ) + elapsed = time.perf_counter() - started + + assert len(results) == 1 + assert elapsed < MANAGED_DELAY_SECONDS / 3, ( + f"the facade took {elapsed:.2f}s while the managed backend slept " + f"{MANAGED_DELAY_SECONDS}s; the dual read is being awaited, so every " + f"piloted turn now pays the slower backend's latency" + ) + + @pytest.mark.asyncio + async def test_the_managed_read_starts_before_legacy_resolves(self, managed_registered): + """Concurrency is not just about the total: the two latencies are only + comparable if both calls were in flight at the same time.""" + order: List[str] = [] + + class OrderedLegacy(StubBackend): + async def search(self, kb_ref, query, top_k=DEFAULT_TOP_K): + order.append("legacy-start") + await asyncio.sleep(0.05) + order.append("legacy-end") + return [_chunk("doc-legacy-1")] + + class OrderedManaged(StubBackend): + async def search(self, kb_ref, query, top_k=DEFAULT_TOP_K): + order.append("managed-start") + return [_chunk("doc-managed-1")] + + managed_registered(OrderedManaged()) + await _search(OrderedLegacy(), {DUAL_READ_ATTR: True}) + + assert "managed-start" in order + assert order.index("managed-start") < order.index("legacy-end"), ( + f"the managed read started only after legacy finished: {order}" + ) + + +# ── The comparison itself ──────────────────────────────────────────────────── +class TestComparisonMetrics: + @pytest.mark.asyncio + async def test_overlap_rank_correlation_and_both_latencies_are_recorded( + self, managed_registered + ): + """Requirement 18.3, all three named measures.""" + shared = [_chunk("doc-a", 0), _chunk("doc-b", 1)] + managed_registered(StubBackend(shared)) + legacy = StubBackend(shared) + + _, emit_value, _ = await _search(legacy, {DUAL_READ_ATTR: True}) + + emitted = [call.args[0] for call in emit_value.call_args_list] + assert "KbDualReadOverlap" in emitted + assert "KbDualReadRankCorrelation" in emitted + assert emitted.count("KbDualReadLatency") == 2 + + backends = { + call.kwargs["dimensions"]["backend"] + for call in emit_value.call_args_list + if call.args[0] == "KbDualReadLatency" + } + assert backends == {"s3vectors", ENGINE_MANAGED} + + @pytest.mark.asyncio + async def test_latency_is_published_in_milliseconds(self, managed_registered): + """A latency published as ``Count`` is not merely mislabelled — CloudWatch + graphs and alarms on it as a rate.""" + managed_registered(StubBackend([_chunk("doc-a")])) + _, emit_value, _ = await _search(StubBackend([_chunk("doc-a")]), {DUAL_READ_ATTR: True}) + + for call in emit_value.call_args_list: + if call.args[0] == "KbDualReadLatency": + assert call.kwargs["unit"] == "Milliseconds" + + +class TestCompareIsPureArithmetic: + def test_identical_results_agree_completely(self): + chunks = [_chunk("doc-a", 0), _chunk("doc-b", 1), _chunk("doc-c", 2)] + result = compare(chunks, chunks, 100.0, 700.0) + + assert result.overlap_count == 3 + assert result.overlap_ratio == 1.0 + assert result.rank_correlation == pytest.approx(1.0) + assert result.legacy_ms == 100.0 + assert result.managed_ms == 700.0 + + def test_a_reversed_ranking_correlates_negatively(self): + forward = [_chunk("doc-a", 0), _chunk("doc-b", 1), _chunk("doc-c", 2)] + result = compare(forward, list(reversed(forward)), 1.0, 1.0) + + assert result.overlap_ratio == 1.0 + assert result.rank_correlation == pytest.approx(-1.0) + + def test_disjoint_results_have_no_overlap_and_no_correlation(self): + result = compare([_chunk("doc-a")], [_chunk("doc-z")], 1.0, 1.0) + + assert result.overlap_count == 0 + assert result.overlap_ratio == 0.0 + assert result.rank_correlation is None + + def test_overlap_is_symmetric(self): + """Jaccard, not a ratio against one side's length — which would read as + agreement when one backend simply returned fewer documents, the case most + likely to occur while the managed corpus is still catching up.""" + many = [_chunk("doc-a", 0), _chunk("doc-b", 1), _chunk("doc-c", 2)] + few = [_chunk("doc-a", 0)] + + forward = compare(many, few, 1.0, 1.0) + backward = compare(few, many, 1.0, 1.0) + + assert forward.overlap_ratio == pytest.approx(1 / 3) + assert forward.overlap_ratio == backward.overlap_ratio + + def test_one_shared_document_yields_no_correlation(self): + """A correlation over a single point is undefined, not 1.0 — reporting 1.0 + would make a pilot on small corpora look like perfect agreement.""" + result = compare( + [_chunk("doc-a", 0), _chunk("doc-b", 1)], + [_chunk("doc-a", 0), _chunk("doc-z", 1)], + 1.0, + 1.0, + ) + + assert result.overlap_count == 1 + assert result.rank_correlation is None + + def test_a_repeated_document_is_ranked_by_its_best_chunk(self): + """Not by its last. Both are one-line implementations and both produce a + number, so the difference is invisible without an input where they + disagree: here ``doc-a`` appears at ranks 0 and 2 in legacy and at rank 0 + in managed, and the two backends agree perfectly on best-rank while + appearing to disagree perfectly on last-rank. + """ + legacy = [_chunk("doc-a", 0), _chunk("doc-b", 1), _chunk("doc-a", 2)] + managed = [_chunk("doc-a", 0), _chunk("doc-b", 1)] + + result = compare(legacy, managed, 1.0, 1.0) + + assert result.rank_correlation == pytest.approx(1.0), ( + "ranking by the document's last chunk instead of its best inverts " + "the measured agreement" + ) + + def test_two_empty_results_do_not_divide_by_zero(self): + result = compare([], [], 1.0, 1.0) + + assert result.overlap_ratio == 0.0 + assert result.rank_correlation is None + + def test_a_document_with_several_chunks_counts_once(self): + """Otherwise the correlation measures chunking rather than agreement: a + document contributing four passages would dominate one contributing a + single passage.""" + repeated = [_chunk("doc-a", 0), _chunk("doc-a", 1), _chunk("doc-a", 2)] + result = compare(repeated, [_chunk("doc-a", 0)], 1.0, 1.0) + + assert result.overlap_count == 1 + assert result.overlap_ratio == 1.0 + assert result.legacy_count == 3 diff --git a/backend/tests/shared/test_kb_idleness.py b/backend/tests/shared/test_kb_idleness.py new file mode 100644 index 000000000..3ef5bba9a --- /dev/null +++ b/backend/tests/shared/test_kb_idleness.py @@ -0,0 +1,326 @@ +""" +Idleness and the fleet gauges: measuring "nobody needs this" without lying. + +Requirements 22.1, 22.5, 22.6. Two rules carry real risk and both are tested here +against the mistake they exist to prevent, not merely for their happy path. + +**Idleness is not retrieval (22.5).** The tempting implementation reads +``lastRetrievedAt`` and stops. It is wrong in the one direction that destroys data: +an agent can be invoked hundreds of times a day and retrieve nothing, because +retrieval only fires when the query matches. So a corpus judged by retrieval alone +looks most abandoned exactly when its agent is busiest with questions its documents +do not answer — and the follow-up spec's eviction pass would delete it. + +**Never a write per retrieval (22.6).** The write is conditional on a throttle +floor, so at most one lands per window however many turns race. + +Also asserted: an unmeasured knowledge base is **not** counted as idle. Reporting +"very idle" for every freshly provisioned corpus is precisely the training signal +that makes operators stop reading a metric. + +Feature: managed-kb-migration +Requirements: 22.1, 22.5, 22.6 +""" + +import json +from unittest.mock import MagicMock, patch + +import pytest + +from apis.shared.kb_backend import idleness +from apis.shared.kb_backend.metrics import ( + BYTES_PER_GB, + IDLE_THRESHOLD_DAYS, + METRIC_KB_COUNT, + METRIC_KB_IDLE_GB, + METRIC_KB_STORAGE_GB, + emit_fleet_gauges, +) + +ASSISTANT_ID = "ast-idle-001" +NOW = "2026-08-25T12:00:00Z" +TABLE = "test-assistants" +ENV = {"DYNAMODB_ASSISTANTS_TABLE_NAME": TABLE, "AWS_REGION": "us-west-2"} + + +def _days_ago(days: int) -> str: + from datetime import datetime, timedelta, timezone + + moment = datetime(2026, 8, 25, 12, 0, 0, tzinfo=timezone.utc) - timedelta(days=days) + return moment.strftime("%Y-%m-%dT%H:%M:%SZ") + + +# ── The rule that stops a busy agent's corpus looking abandoned ────────────── +class TestIdlenessIsNotRetrievalAlone: + def test_agent_use_counts_even_when_nothing_was_ever_retrieved(self): + """Requirement 22.5, stated as the failure it prevents. + + The knowledge base has never served a retrieval — no ``lastRetrievedAt`` at + all — but its agent was used today. Judged by retrieval alone this is + maximally idle; judged correctly it is active. + """ + days = idleness.idle_days( + ASSISTANT_ID, record={}, agent_timestamps=[_days_ago(0)], now=NOW + ) + + assert days == pytest.approx(0.0, abs=0.01) + + def test_the_more_recent_of_the_two_wins(self): + record = {idleness.LAST_RETRIEVED_ATTR: _days_ago(90)} + + days = idleness.idle_days( + ASSISTANT_ID, record=record, agent_timestamps=[_days_ago(2)], now=NOW + ) + + assert days == pytest.approx(2.0, abs=0.01) + + def test_retrieval_wins_when_it_is_the_more_recent(self): + record = {idleness.LAST_RETRIEVED_ATTR: _days_ago(1)} + + days = idleness.idle_days( + ASSISTANT_ID, record=record, agent_timestamps=[_days_ago(200)], now=NOW + ) + + assert days == pytest.approx(1.0, abs=0.01) + + def test_a_genuinely_idle_knowledge_base_reads_as_idle(self): + """The permissive cases above would all pass if this returned 0 for + everything, so the negative case is what makes them mean something.""" + record = {idleness.LAST_RETRIEVED_ATTR: _days_ago(120)} + + days = idleness.idle_days( + ASSISTANT_ID, record=record, agent_timestamps=[_days_ago(115)], now=NOW + ) + + assert days == pytest.approx(115.0, abs=0.01) + assert days >= IDLE_THRESHOLD_DAYS + + def test_no_signal_at_all_is_unknown_not_ancient(self): + """A knowledge base provisioned an hour ago looks exactly like this. If it + returned a large number, every new corpus would report as idle.""" + assert idleness.idle_days(ASSISTANT_ID, record={}, agent_timestamps=[], now=NOW) is None + assert ( + idleness.idle_days(ASSISTANT_ID, record={}, agent_timestamps=[None], now=NOW) is None + ) + + def test_activity_is_a_maximum_over_a_set_of_agents(self): + """One bound agent this phase, written as a maximum over a set so F4 making + the set larger is not a rewrite of the module whose whole job is not to + under-report activity.""" + latest = idleness.last_activity_at( + ASSISTANT_ID, + record={idleness.LAST_RETRIEVED_ATTR: _days_ago(50)}, + agent_timestamps=[_days_ago(40), _days_ago(3), _days_ago(60)], + ) + + assert latest == _days_ago(3) + + def test_the_bound_agent_this_phase_is_the_assistant_itself(self): + assert idleness.bound_agent_ids(ASSISTANT_ID) == [ASSISTANT_ID] + + def test_an_unparseable_timestamp_is_unknown_rather_than_ancient(self): + record = {idleness.LAST_RETRIEVED_ATTR: "last Tuesday"} + assert idleness.idle_days(ASSISTANT_ID, record=record, agent_timestamps=[], now=NOW) is None + + def test_an_iso_offset_timestamp_still_parses(self): + """The table holds timestamps written by several generations of code.""" + record = {idleness.LAST_RETRIEVED_ATTR: "2026-08-23T12:00:00+00:00"} + days = idleness.idle_days(ASSISTANT_ID, record=record, agent_timestamps=[], now=NOW) + assert days == pytest.approx(2.0, abs=0.01) + + def test_a_future_timestamp_clamps_to_zero_rather_than_going_negative(self): + record = {idleness.LAST_RETRIEVED_ATTR: _days_ago(-5)} + days = idleness.idle_days(ASSISTANT_ID, record=record, agent_timestamps=[], now=NOW) + assert days == 0.0 + + def test_the_agent_timestamp_falls_back_through_updated_and_created(self): + """An assistant that has never been used still has a creation date, which is + a better idleness floor than nothing.""" + table = MagicMock() + table.get_item.return_value = {"Item": {"createdAt": _days_ago(10)}} + resource = MagicMock() + resource.Table.return_value = table + + with patch.dict("os.environ", ENV, clear=True), patch( + "boto3.resource", return_value=resource + ): + assert idleness.agent_last_used_at(ASSISTANT_ID) == _days_ago(10) + + +# ── The throttled write ────────────────────────────────────────────────────── +class TestTheWriteIsThrottled: + def _table(self, error_code=None): + table = MagicMock() + if error_code: + table.update_item.side_effect = _client_error(error_code) + resource = MagicMock() + resource.Table.return_value = table + return resource, table + + def test_the_write_is_conditional_on_a_freshness_floor(self): + """Requirement 22.6. The condition is what makes calling this on every + retrieval acceptable: a conditional write that loses is not a write.""" + resource, table = self._table() + + with patch.dict("os.environ", ENV, clear=True), patch( + "boto3.resource", return_value=resource + ): + assert idleness.touch_last_retrieved(ASSISTANT_ID, ASSISTANT_ID) is True + + kwargs = table.update_item.call_args.kwargs + assert idleness.LAST_RETRIEVED_ATTR in kwargs["ConditionExpression"] + assert ":floor" in kwargs["ConditionExpression"] + assert ":floor" in kwargs["ExpressionAttributeValues"] + + def test_the_condition_refuses_to_create_a_record(self): + """A legacy knowledge base has no KB_Record and must keep having none: the + migration's zero-backfill property across 1,692 rows is that nothing writes + to them until their owner opts in. A metrics side effect that created rows + would break it while looking harmless.""" + resource, table = self._table() + + with patch.dict("os.environ", ENV, clear=True), patch( + "boto3.resource", return_value=resource + ): + idleness.touch_last_retrieved(ASSISTANT_ID, ASSISTANT_ID) + + condition = table.update_item.call_args.kwargs["ConditionExpression"] + assert "attribute_exists(SK)" in condition + + def test_a_rejected_write_is_not_an_error(self): + resource, _ = self._table(error_code="ConditionalCheckFailedException") + + with patch.dict("os.environ", ENV, clear=True), patch( + "boto3.resource", return_value=resource + ): + assert idleness.touch_last_retrieved(ASSISTANT_ID, ASSISTANT_ID) is False + + def test_any_other_failure_is_swallowed(self): + resource, _ = self._table(error_code="ProvisionedThroughputExceededException") + + with patch.dict("os.environ", ENV, clear=True), patch( + "boto3.resource", return_value=resource + ): + assert idleness.touch_last_retrieved(ASSISTANT_ID, ASSISTANT_ID) is False + + def test_no_table_configured_is_a_no_op(self): + with patch.dict("os.environ", {}, clear=True): + assert idleness.touch_last_retrieved(ASSISTANT_ID, ASSISTANT_ID) is False + + def test_the_throttle_window_is_resolved_at_call_time(self): + """Never bound as a default argument, which is captured once at import and + makes a test's override silently ineffective.""" + with patch.dict("os.environ", {}, clear=True): + assert idleness.throttle_hours() == idleness.THROTTLE_HOURS + with patch.dict("os.environ", {"KB_LAST_RETRIEVED_THROTTLE_HOURS": "6"}, clear=True): + assert idleness.throttle_hours() == 6 + with patch.dict("os.environ", {"KB_LAST_RETRIEVED_THROTTLE_HOURS": "0"}, clear=True): + # Zero would mean a write per retrieval, which is the thing forbidden. + assert idleness.throttle_hours() == 1 + + @pytest.mark.asyncio + async def test_the_touch_is_detached_from_the_caller(self): + """Retrieval must wait for neither the write nor its rejection — and + rejection is the common case.""" + import asyncio + + started = asyncio.Event() + + def _slow(assistant_id, app_kb_id): + started.set() + return False + + with patch.object(idleness, "touch_last_retrieved", side_effect=_slow): + idleness.schedule_activity_touch(ASSISTANT_ID, ASSISTANT_ID) + # Nothing was awaited above; the work happens after we yield. + assert started.is_set() is False + for _ in range(100): + await asyncio.sleep(0) + if started.is_set(): + break + + assert started.is_set() is True + + def test_scheduling_outside_an_event_loop_does_nothing(self): + """A missing idleness sample is a gap in a baseline metric; an exception + would be a failed retrieval.""" + idleness.schedule_activity_touch(ASSISTANT_ID, ASSISTANT_ID) + + +# ── The gauges ─────────────────────────────────────────────────────────────── +class TestFleetGauges: + def _emit(self, **kwargs): + records = [] + with patch("apis.shared.observability.emf._emf_logger") as raw: + emit_fleet_gauges(**kwargs) + for call in raw.info.call_args_list: + records.append(json.loads(call.args[0])) + return records + + def test_all_three_gauges_are_emitted_with_storage_units(self): + records = self._emit(kb_count=4, stored_bytes=2 * BYTES_PER_GB, idle_bytes=BYTES_PER_GB) + + assert len(records) == 1 + record = records[0] + assert record[METRIC_KB_COUNT] == 4 + assert record[METRIC_KB_STORAGE_GB] == 2.0 + assert record[METRIC_KB_IDLE_GB] == 1.0 + + units = { + metric["Name"]: metric["Unit"] + for metric in record["_aws"]["CloudWatchMetrics"][0]["Metrics"] + } + assert units[METRIC_KB_STORAGE_GB] == "Gigabytes" + assert units[METRIC_KB_IDLE_GB] == "Gigabytes" + assert units[METRIC_KB_COUNT] == "Count" + + def test_the_namespace_is_the_one_the_iam_grant_allows(self): + with patch.dict("os.environ", {"PROJECT_PREFIX": "bsu-agentcore"}, clear=True): + records = self._emit(kb_count=1, stored_bytes=0, idle_bytes=0) + + namespace = records[0]["_aws"]["CloudWatchMetrics"][0]["Namespace"] + assert namespace == "bsu-agentcore/ManagedKb" + assert not namespace.startswith("AWS"), ( + "CloudWatch reserves every namespace beginning with AWS and rejects " + "writes to them, so this would publish nothing forever" + ) + + def test_gigabytes_are_decimal_so_a_dashboard_matches_an_invoice(self): + """AWS bills $5.00/GB-month on decimal gigabytes. Using 2**30 here would + make every dashboard number 7% smaller than the bill it is meant to + explain.""" + records = self._emit(kb_count=1, stored_bytes=1_500_000_000, idle_bytes=0) + assert records[0][METRIC_KB_STORAGE_GB] == 1.5 + + def test_unmeasured_rides_along_as_a_property_not_a_metric(self): + """Legitimately large the day this ships and near zero a month later, so an + alarm on it would be noise. Context for reading KbIdleGB, not a target.""" + records = self._emit(kb_count=3, stored_bytes=0, idle_bytes=0, unmeasured=2) + + record = records[0] + assert record["unmeasuredKnowledgeBases"] == 2 + names = {m["Name"] for m in record["_aws"]["CloudWatchMetrics"][0]["Metrics"]} + assert "unmeasuredKnowledgeBases" not in names + + def test_no_dimensions_so_the_alarm_target_is_one_fleet_sum(self): + records = self._emit(kb_count=1, stored_bytes=0, idle_bytes=0) + assert records[0]["_aws"]["CloudWatchMetrics"][0]["Dimensions"] == [[]] + + def test_a_reclaimed_gigabytes_metric_is_never_emitted(self): + """Requirement 22.1 forbids it: nothing reclaims in this phase, and a + structurally-always-zero metric trains operators to ignore the board.""" + records = self._emit(kb_count=1, stored_bytes=0, idle_bytes=0) + assert not any("Reclaim" in key for key in records[0]) + + def test_emission_never_raises(self): + with patch( + "apis.shared.observability.emf.emit_emf_metrics", + side_effect=RuntimeError("logging broke"), + ): + emit_fleet_gauges(kb_count=1, stored_bytes=0, idle_bytes=0) + + +def _client_error(code: str): + from botocore.exceptions import ClientError + + return ClientError({"Error": {"Code": code}}, "UpdateItem") diff --git a/backend/tests/shared/test_kb_records.py b/backend/tests/shared/test_kb_records.py new file mode 100644 index 000000000..b7644baae --- /dev/null +++ b/backend/tests/shared/test_kb_records.py @@ -0,0 +1,390 @@ +"""Conditional-transition tests for the KB_Record data layer. + +These assert the *persistence* behaviour, not the logic: the failures this layer +exists to prevent — two workers both promoting, a finished knowledge base left in +the dispatcher's queue, a rollback that writes a legacy value instead of removing +an attribute — are all invisible in a unit test that stubs DynamoDB out. So every +test here drives real condition expressions against moto. + +Concurrency is simulated deterministically rather than with threads. Two workers +racing on a conditional write is, from DynamoDB's point of view, simply two +sequential writes where the second one's guard no longer holds. Issuing them in +order and asserting the second is rejected tests exactly the property that +matters and does it without a flaky sleep. +""" + +import boto3 +import pytest +from moto import mock_aws + +from apis.shared.kb_backend import records as r + +REGION = "us-east-1" +TABLE = "test-kb-records" +ASSISTANT_ID = "ast-kb0001" +APP_KB_ID = ASSISTANT_ID # App_KB_Id == assistant_id in this phase +NOW = "2026-08-24T12:00:00Z" +LATER = "2026-08-24T13:00:00Z" + + +@pytest.fixture() +def table(monkeypatch): + """The assistants table including the GSI7 work index. + + Built here rather than reused from the shared ``assistants_table`` fixture, + which predates GSI7 and only defines GSI1-4. The listing tests set the same + precedent of building a table when they need an index the shared fixture + lacks. + """ + monkeypatch.setenv("AWS_DEFAULT_REGION", REGION) + monkeypatch.setenv("AWS_ACCESS_KEY_ID", "testing") + monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "testing") + monkeypatch.setenv("AWS_SESSION_TOKEN", "testing") + monkeypatch.setenv("DYNAMODB_ASSISTANTS_TABLE_NAME", TABLE) + + with mock_aws(): + ddb = boto3.client("dynamodb", region_name=REGION) + ddb.create_table( + TableName=TABLE, + KeySchema=[ + {"AttributeName": "PK", "KeyType": "HASH"}, + {"AttributeName": "SK", "KeyType": "RANGE"}, + ], + AttributeDefinitions=[ + {"AttributeName": "PK", "AttributeType": "S"}, + {"AttributeName": "SK", "AttributeType": "S"}, + {"AttributeName": "GSI7_PK", "AttributeType": "S"}, + {"AttributeName": "GSI7_SK", "AttributeType": "S"}, + ], + GlobalSecondaryIndexes=[ + { + "IndexName": "KbWorkIndex", + "KeySchema": [ + {"AttributeName": "GSI7_PK", "KeyType": "HASH"}, + {"AttributeName": "GSI7_SK", "KeyType": "RANGE"}, + ], + "Projection": {"ProjectionType": "ALL"}, + } + ], + BillingMode="PAY_PER_REQUEST", + ) + yield boto3.resource("dynamodb", region_name=REGION).Table(TABLE) + + +def _record(**overrides) -> r.KbRecord: + base = dict(app_kb_id=APP_KB_ID, owner_user_id="owner-opaque-1") + base.update(overrides) + return r.KbRecord(**base) + + +def _raw(table): + return table.get_item( + Key={"PK": r.kb_pk(ASSISTANT_ID), "SK": r.kb_sk(APP_KB_ID)} + )["Item"] + + +def _seed(table, **overrides): + """A record already provisioned and mid-migration.""" + r.create_provisioning(ASSISTANT_ID, _record(**overrides)) + return _raw(table) + + +# ── create_provisioning ────────────────────────────────────────────────────── +class TestCreateProvisioning: + def test_creates_the_record(self, table): + r.create_provisioning(ASSISTANT_ID, _record()) + item = _raw(table) + assert item["appKbId"] == APP_KB_ID + assert item["provisioningState"] == r.PROVISIONING + + def test_concurrent_create_yields_exactly_one_winner(self, table): + """Two enrolments race; one record exists and the loser is told so. + + Without the ``attribute_not_exists`` guard the second write would clobber + the first, replacing a ``clientToken`` that may already have been used to + create a real AWS knowledge base — orphaning it with no record pointing + at it. + """ + r.create_provisioning(ASSISTANT_ID, _record(client_token="a" * 33)) + + with pytest.raises(r.TransitionLost): + r.create_provisioning(ASSISTANT_ID, _record(client_token="b" * 33)) + + assert _raw(table)["clientToken"] == "a" * 33 + + def test_a_new_record_carries_no_engine_attribute(self, table): + """Absence means legacy, so a fresh record must not name an engine. + + A record created with ``retrievalEngine`` already set would be served by + the managed backend before its knowledge base exists. + """ + r.create_provisioning(ASSISTANT_ID, _record()) + assert "retrievalEngine" not in _raw(table) + + def test_a_new_record_carries_no_work_keys(self, table): + """Enrolment is not the same as being queued for work.""" + r.create_provisioning(ASSISTANT_ID, _record()) + item = _raw(table) + assert "GSI7_PK" not in item + assert "GSI7_SK" not in item + + +# ── attach_aws_ids ─────────────────────────────────────────────────────────── +class TestAttachAwsIds: + def test_attaches_and_activates(self, table): + _seed(table) + r.attach_aws_ids(ASSISTANT_ID, APP_KB_ID, "kb-123", "ds-456", NOW) + item = _raw(table) + assert item["awsKbId"] == "kb-123" + assert item["awsDataSourceId"] == "ds-456" + assert item["provisioningState"] == r.ACTIVE + + def test_a_late_duplicate_cannot_overwrite_the_identifiers(self, table): + """Guarded on still provisioning, so a slow retry cannot rebind the record. + + Rebinding would strand the first AWS knowledge base: still billed, no + longer referenced. + """ + _seed(table) + r.attach_aws_ids(ASSISTANT_ID, APP_KB_ID, "kb-first", "ds-first", NOW) + + with pytest.raises(r.TransitionLost): + r.attach_aws_ids(ASSISTANT_ID, APP_KB_ID, "kb-second", "ds-second", LATER) + + assert _raw(table)["awsKbId"] == "kb-first" + + +# ── set_migration_state ────────────────────────────────────────────────────── +class TestSetMigrationState: + def test_entering_an_eligible_state_writes_the_work_keys(self, table): + _seed(table) + r.set_migration_state(ASSISTANT_ID, APP_KB_ID, r.SHADOW, 0, due_at=NOW) + item = _raw(table) + assert item["GSI7_PK"] == "KBWORK#shadow" + assert item["GSI7_SK"] == NOW + + @pytest.mark.parametrize("terminal", [r.RETAIN, r.MIGRATION_FAILED]) + def test_a_terminal_state_removes_the_work_keys(self, table, terminal): + """The removal is what takes the record out of the dispatcher's queue. + + Asserted for both terminal states because ``failed`` is the easy one to + forget, and a failed migration left in the queue would be retried + forever against a knowledge base its owner was told had stopped. + """ + _seed(table) + r.set_migration_state(ASSISTANT_ID, APP_KB_ID, r.SHADOW, 0, due_at=NOW) + assert "GSI7_PK" in _raw(table) + + r.set_migration_state(ASSISTANT_ID, APP_KB_ID, terminal, 0) + + item = _raw(table) + assert item["migrationState"] == terminal + assert "GSI7_PK" not in item, "a terminal record kept its work partition key" + assert "GSI7_SK" not in item, "a terminal record kept its work sort key" + + def test_a_terminal_record_is_invisible_to_the_work_query(self, table): + """The physics claim, tested end to end rather than by attribute check.""" + _seed(table) + r.set_migration_state(ASSISTANT_ID, APP_KB_ID, r.SHADOW, 0, due_at=NOW) + assert len(r.query_due_work(r.SHADOW, LATER)) == 1 + + r.set_migration_state(ASSISTANT_ID, APP_KB_ID, r.RETAIN, 0) + assert r.query_due_work(r.SHADOW, LATER) == [] + + def test_a_stale_generation_cannot_move_the_state(self, table): + _seed(table, migration_generation=2) + with pytest.raises(r.TransitionLost): + r.set_migration_state(ASSISTANT_ID, APP_KB_ID, r.SHADOW, 1, due_at=NOW) + + def test_expected_states_guards_against_a_concurrent_mover(self, table): + _seed(table) + r.set_migration_state(ASSISTANT_ID, APP_KB_ID, r.SHADOW, 0, due_at=NOW) + + with pytest.raises(r.TransitionLost): + r.set_migration_state( + ASSISTANT_ID, APP_KB_ID, r.PROMOTE, 0, + due_at=NOW, expected_states=[r.VERIFY], + ) + + def test_reclaim_is_refused(self, table): + """Reserved in the enum, never entered. Entering it would delete data + this phase has promised to retain for the rollback window.""" + _seed(table) + with pytest.raises(r.ReclaimNotSupported): + r.set_migration_state(ASSISTANT_ID, APP_KB_ID, r.RECLAIM, 0, due_at=NOW) + + def test_an_eligible_state_requires_a_due_time(self, table): + """A work-eligible record with no dueAt would be unschedulable, and a + partial GSI key silently fails to index rather than erroring.""" + _seed(table) + with pytest.raises(ValueError): + r.set_migration_state(ASSISTANT_ID, APP_KB_ID, r.SHADOW, 0) + + +# ── promote_engine ─────────────────────────────────────────────────────────── +class TestPromoteEngine: + def _ready(self, table, migrated=3, total=3, generation=0): + _seed( + table, + migration_generation=generation, + migration_progress={"migrated": migrated, "total": total}, + ) + r.set_migration_state( + ASSISTANT_ID, APP_KB_ID, r.PROMOTE, generation, due_at=NOW + ) + + def test_promotes_when_catch_up_has_converged(self, table): + self._ready(table) + r.promote_engine(ASSISTANT_ID, APP_KB_ID, 0, NOW) + item = _raw(table) + assert item["retrievalEngine"] == r.ENGINE_MANAGED + assert item["promotedAt"] == NOW + + def test_concurrent_promote_yields_exactly_one_winner(self, table): + """Both workers think they should promote; only one write lands. + + The second is rejected because the first bumped nothing it could reuse — + it is the guard, not luck, that stops a double promotion. + """ + self._ready(table) + r.promote_engine(ASSISTANT_ID, APP_KB_ID, 0, NOW) + + # The winner moves the record on; the loser's guard no longer holds. + r.set_migration_state(ASSISTANT_ID, APP_KB_ID, r.RETAIN, 0) + + with pytest.raises(r.TransitionLost): + r.promote_engine(ASSISTANT_ID, APP_KB_ID, 0, LATER) + + assert _raw(table)["promotedAt"] == NOW + + def test_refuses_to_promote_before_catch_up_converges(self, table): + """The guard that stops documents being stranded on an unread backend.""" + self._ready(table, migrated=2, total=5) + with pytest.raises(r.TransitionLost): + r.promote_engine(ASSISTANT_ID, APP_KB_ID, 0, NOW) + assert "retrievalEngine" not in _raw(table) + + def test_a_stale_generation_cannot_promote(self, table): + self._ready(table, generation=2) + with pytest.raises(r.TransitionLost): + r.promote_engine(ASSISTANT_ID, APP_KB_ID, 1, NOW) + assert "retrievalEngine" not in _raw(table) + + def test_cannot_promote_from_a_non_promote_state(self, table): + _seed(table, migration_progress={"migrated": 3, "total": 3}) + r.set_migration_state(ASSISTANT_ID, APP_KB_ID, r.SHADOW, 0, due_at=NOW) + with pytest.raises(r.TransitionLost): + r.promote_engine(ASSISTANT_ID, APP_KB_ID, 0, NOW) + + def test_an_already_promoted_record_cannot_be_promoted_again(self, table): + """The guard that ``test_concurrent_promote_yields_exactly_one_winner`` + cannot see, because that test moves the record on between the two attempts. + + Here **nothing** changes between them: same state, same generation, same + converged progress. Every guard except ``attribute_not_exists`` still + holds, so without it the second write lands — overwriting the real cutover + moment with a later ``promotedAt`` and letting two genuinely concurrent + workers both succeed, which is exactly what Requirement 15.10 forbids. + Found by the convergence property test crashing between the promotion and + the state transition. + """ + self._ready(table) + r.promote_engine(ASSISTANT_ID, APP_KB_ID, 0, NOW) + + with pytest.raises(r.TransitionLost): + r.promote_engine(ASSISTANT_ID, APP_KB_ID, 0, LATER) + + assert _raw(table)["promotedAt"] == NOW, ( + "a second promotion overwrote the original cutover timestamp" + ) + + def test_a_rolled_back_record_can_be_promoted_again(self, table): + """So the not-already-promoted guard does not make rollback one-way. + + Rollback ``REMOVE``s the attribute, which is precisely what restores + eligibility — another consequence of rollback restoring the original + *shape* rather than writing a legacy value. + """ + self._ready(table) + r.promote_engine(ASSISTANT_ID, APP_KB_ID, 0, NOW) + r.rollback_engine(ASSISTANT_ID, APP_KB_ID, LATER) + + r.promote_engine(ASSISTANT_ID, APP_KB_ID, 0, LATER) + + assert _raw(table)["retrievalEngine"] == r.ENGINE_MANAGED + assert _raw(table)["promotedAt"] == LATER + + +# ── rollback_engine ────────────────────────────────────────────────────────── +class TestRollbackEngine: + def test_rollback_removes_the_attribute_rather_than_writing_legacy(self, table): + """The invariant that makes rollback a pointer flip. + + A rolled-back record must be indistinguishable from one that never + migrated. Writing the literal legacy value would pass a naive test and + quietly convert every future rollback into a data migration. + """ + _seed(table, migration_progress={"migrated": 1, "total": 1}) + r.set_migration_state(ASSISTANT_ID, APP_KB_ID, r.PROMOTE, 0, due_at=NOW) + r.promote_engine(ASSISTANT_ID, APP_KB_ID, 0, NOW) + assert _raw(table)["retrievalEngine"] == r.ENGINE_MANAGED + + r.rollback_engine(ASSISTANT_ID, APP_KB_ID, LATER) + + item = _raw(table) + assert "retrievalEngine" not in item, "rollback left an engine attribute behind" + assert r.resolve_engine(item) == r.ENGINE_LEGACY + assert item["rolledBackAt"] == LATER + + def test_rolling_back_a_legacy_record_is_rejected(self, table): + _seed(table) + with pytest.raises(r.TransitionLost): + r.rollback_engine(ASSISTANT_ID, APP_KB_ID, NOW) + + +# ── acquire_lease ──────────────────────────────────────────────────────────── +class TestAcquireLease: + def test_takes_a_free_lease(self, table): + _seed(table) + r.acquire_lease(ASSISTANT_ID, APP_KB_ID, LATER, NOW) + assert _raw(table)["migrationLeaseUntil"] == LATER + + def test_a_live_lease_admits_exactly_one_holder(self, table): + """Two workers cannot migrate the same corpus concurrently. + + Double ingestion is not merely wasteful: it is billed per gigabyte and + would double the owner's stored bytes against their cap. + """ + _seed(table) + r.acquire_lease(ASSISTANT_ID, APP_KB_ID, "2026-08-24T14:00:00Z", NOW) + + with pytest.raises(r.TransitionLost): + r.acquire_lease(ASSISTANT_ID, APP_KB_ID, "2026-08-24T15:00:00Z", LATER) + + assert _raw(table)["migrationLeaseUntil"] == "2026-08-24T14:00:00Z" + + def test_an_expired_lease_can_be_taken_over(self, table): + """Otherwise a worker that died holding a lease would strand the record.""" + _seed(table) + r.acquire_lease(ASSISTANT_ID, APP_KB_ID, "2026-08-24T12:30:00Z", NOW) + + r.acquire_lease(ASSISTANT_ID, APP_KB_ID, "2026-08-24T14:00:00Z", LATER) + + assert _raw(table)["migrationLeaseUntil"] == "2026-08-24T14:00:00Z" + + +# ── tombstones ─────────────────────────────────────────────────────────────── +class TestTombstoneKeys: + def test_the_two_tombstone_shapes_are_distinct(self): + assert r.kb_tombstone_sk(APP_KB_ID) == f"KBTOMB#{APP_KB_ID}" + assert ( + r.document_tombstone_sk(APP_KB_ID, "doc-9") + == f"KBTOMB#{APP_KB_ID}#DOC#doc-9" + ) + + def test_a_whole_kb_tombstone_does_not_prefix_match_a_document_one(self): + """They share a prefix, so a query for one must not sweep up the other.""" + whole = r.kb_tombstone_sk(APP_KB_ID) + doc = r.document_tombstone_sk(APP_KB_ID, "doc-9") + assert doc.startswith(whole) + assert doc != whole diff --git a/backend/tests/shared/test_kb_tombstones.py b/backend/tests/shared/test_kb_tombstones.py new file mode 100644 index 000000000..6c5749a55 --- /dev/null +++ b/backend/tests/shared/test_kb_tombstones.py @@ -0,0 +1,749 @@ +"""Tombstoned deletion sagas — ordering, confirmation, and refusal. + +Feature: managed-kb-migration, task 10.1. +Requirements: 13.1-13.8, 24.11. + +What these tests are actually defending +--------------------------------------- +A managed knowledge base is a *billed* resource with no CloudFormation parent, so a +delete that half-finishes is not a crash — it is a recurring charge with no owner +and no error anywhere. Every assertion here exists because the natural +implementation of one step produces exactly that outcome: + +* **Tombstone before AWS.** Asserted *causally*, not by call order in a mock: the + stubbed client reads DynamoDB at the moment it is called and records whether the + tombstone was already there. Reversing the two steps in the source makes that + observation ``False`` and the test fails, which a plain "was write called before + delete" assertion on two independent mocks would not reliably catch. +* **Clear only after confirmed absence.** Same technique from the other side: the + stub records, on every list poll, whether the tombstone still exists. Clearing + early makes one of those observations ``False``. +* **"Accepted" is not "gone".** The stub's ``delete_knowledge_base`` succeeds and + then keeps listing the knowledge base, which is exactly what AWS does for 2-6 + minutes. + +No test here contacts AWS. DynamoDB is moto; every ``bedrock-agent`` call goes to +:class:`FakeBedrockAgent` (Requirement 24.11). +""" + +from datetime import datetime, timezone + +import boto3 +import pytest +from botocore.exceptions import ClientError +from moto import mock_aws + +from apis.shared.kb_backend import tags as kb_tags +from apis.shared.kb_backend import tombstones as tomb + +REGION = "us-east-1" +TABLE = "test-kb-tombstones" +ASSISTANT_ID = "ast-tomb01" +APP_KB_ID = ASSISTANT_ID +AWS_KB_ID = "KBAAAA1111" +AWS_DS_ID = "DSAAAA1111" +DOCUMENT_ID = "doc-tomb01" +ARN = f"arn:aws:bedrock:{REGION}:123456789012:knowledge-base/{AWS_KB_ID}" +PROJECT_TAGS = kb_tags.build_tags(APP_KB_ID, "u-1", "testprefix", "testenv") + + +# ── Fixtures ───────────────────────────────────────────────────────────────── +@pytest.fixture() +def table(monkeypatch): + monkeypatch.setenv("AWS_DEFAULT_REGION", REGION) + monkeypatch.setenv("AWS_ACCESS_KEY_ID", "testing") + monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "testing") + monkeypatch.setenv("AWS_SESSION_TOKEN", "testing") + monkeypatch.setenv("DYNAMODB_ASSISTANTS_TABLE_NAME", TABLE) + monkeypatch.setenv(kb_tags.ENV_TAG_VALUE_PREFIX, "testprefix") + monkeypatch.setenv(kb_tags.ENV_TAG_VALUE_ENVIRONMENT, "testenv") + + with mock_aws(): + boto3.client("dynamodb", region_name=REGION).create_table( + TableName=TABLE, + KeySchema=[ + {"AttributeName": "PK", "KeyType": "HASH"}, + {"AttributeName": "SK", "KeyType": "RANGE"}, + ], + AttributeDefinitions=[ + {"AttributeName": "PK", "AttributeType": "S"}, + {"AttributeName": "SK", "AttributeType": "S"}, + ], + BillingMode="PAY_PER_REQUEST", + ) + yield boto3.resource("dynamodb", region_name=REGION).Table(TABLE) + + +@pytest.fixture(autouse=True) +def no_metrics(monkeypatch): + """Metrics are observability; stub them so nothing reaches CloudWatch.""" + monkeypatch.setattr(tomb, "emit_count", lambda *a, **k: None) + + +def _not_found(operation): + return ClientError( + {"Error": {"Code": "ResourceNotFoundException", "Message": "not found"}}, + operation, + ) + + +class FakeBedrockAgent: + """A ``bedrock-agent`` stand-in that behaves the way the real API does. + + Two behaviours are modelled deliberately because they are what the sagas are + built against: + + * ``delete_knowledge_base`` **succeeds** and the knowledge base keeps + appearing in ``list_knowledge_bases`` for ``polls_before_gone`` further + polls, mirroring the measured 2-6 minute asynchronous deletion. + * every call runs ``probe`` first, so a test can capture the state of + DynamoDB *at the moment of the AWS call* and assert ordering causally + rather than by mock call sequence. + """ + + def __init__( + self, + knowledge_bases=None, + tags=None, + polls_before_gone=0, + probe=None, + page_size=None, + delete_raises=None, + document_statuses=None, + ): + self.kbs = {kb["knowledgeBaseId"]: dict(kb) for kb in (knowledge_bases or [])} + self.tags = dict(tags or {}) + self.polls_before_gone = polls_before_gone + self.probe = probe + self.page_size = page_size + self.delete_raises = delete_raises + + #: Successive ``GetKnowledgeBaseDocuments`` statuses to hand back, so a + #: test can model DELETING → NOT_FOUND. + self.document_statuses = list(document_statuses or ["NOT_FOUND"]) + + self.list_calls = 0 + self.delete_calls = [] + self.delete_document_calls = [] + self.get_document_calls = 0 + #: One entry per AWS call: (operation, probe_result). + self.observations = [] + self._pending = {} + + # ── probe plumbing ────────────────────────────────────────────────────── + def _observe(self, operation): + result = self.probe() if self.probe else None + self.observations.append((operation, result)) + return result + + def probes_for(self, operation): + return [result for op, result in self.observations if op == operation] + + # ── control plane ─────────────────────────────────────────────────────── + def list_knowledge_bases(self, **kwargs): + self._observe("list_knowledge_bases") + self.list_calls += 1 + + # Age out anything whose delete has been accepted. + for kb_id in list(self._pending): + self._pending[kb_id] -= 1 + if self._pending[kb_id] <= 0: + self._pending.pop(kb_id) + self.kbs.pop(kb_id, None) + + summaries = [ + { + "knowledgeBaseId": kb["knowledgeBaseId"], + "name": kb.get("name", ""), + "status": kb.get("status", "ACTIVE"), + "updatedAt": kb.get("updatedAt", datetime.now(timezone.utc)), + } + for kb in self.kbs.values() + ] + + page_size = self.page_size or kwargs.get("maxResults") or len(summaries) or 1 + start = int(kwargs.get("nextToken") or 0) + page = summaries[start : start + page_size] + response = {"knowledgeBaseSummaries": page} + if start + page_size < len(summaries): + response["nextToken"] = str(start + page_size) + return response + + def get_knowledge_base(self, knowledgeBaseId): # noqa: N803 - AWS parameter name + self._observe("get_knowledge_base") + kb = self.kbs.get(knowledgeBaseId) + if kb is None: + raise _not_found("GetKnowledgeBase") + return {"knowledgeBase": kb} + + def list_tags_for_resource(self, resourceArn): # noqa: N803 - AWS parameter name + self._observe("list_tags_for_resource") + return {"tags": dict(self.tags.get(resourceArn, {}))} + + def delete_knowledge_base(self, knowledgeBaseId): # noqa: N803 - AWS parameter name + self._observe("delete_knowledge_base") + self.delete_calls.append(knowledgeBaseId) + if self.delete_raises is not None: + raise self.delete_raises + if knowledgeBaseId not in self.kbs: + raise _not_found("DeleteKnowledgeBase") + if self.polls_before_gone <= 0: + self.kbs.pop(knowledgeBaseId, None) + else: + # A knowledge base already in DELETE_UNSUCCESSFUL stays there: a second + # delete call is accepted and changes nothing, which is exactly why the + # state needs an operator rather than a retry. + if self.kbs[knowledgeBaseId].get("status") != "DELETE_UNSUCCESSFUL": + self.kbs[knowledgeBaseId]["status"] = "DELETING" + self._pending[knowledgeBaseId] = self.polls_before_gone + return {"knowledgeBaseId": knowledgeBaseId, "status": "DELETING"} + + # ── documents ─────────────────────────────────────────────────────────── + def delete_knowledge_base_documents(self, **kwargs): + self._observe("delete_knowledge_base_documents") + self.delete_document_calls.append(kwargs) + return {"documentDetails": [{"status": "DELETING"}]} + + def get_knowledge_base_documents(self, **kwargs): + self._observe("get_knowledge_base_documents") + index = min(self.get_document_calls, len(self.document_statuses) - 1) + self.get_document_calls += 1 + status = self.document_statuses[index] + return {"documentDetails": [{"status": status, "identifier": {}}]} + + +def _kb(kb_id=AWS_KB_ID, status="ACTIVE", name="testprefix-kb-x", role_arn="role/a"): + return { + "knowledgeBaseId": kb_id, + "name": name, + "status": status, + "knowledgeBaseArn": f"arn:aws:bedrock:{REGION}:123456789012:knowledge-base/{kb_id}", + "roleArn": role_arn, + "createdAt": datetime(2026, 1, 1, tzinfo=timezone.utc), + } + + +def _tomb_item(table, app_kb_id=APP_KB_ID): + return table.get_item( + Key={"PK": f"AST#{ASSISTANT_ID}", "SK": f"KBTOMB#{app_kb_id}"} + ).get("Item") + + +def _tomb_exists(table, app_kb_id=APP_KB_ID): + return _tomb_item(table, app_kb_id) is not None + + +# ── Requirement 13.1: tombstone before AWS ─────────────────────────────────── +class TestTombstoneIsWrittenBeforeAws: + def test_tombstone_already_exists_when_aws_delete_is_called(self, table): + """The tombstone must be durable *before* the delete call, not after. + + Asserted from inside the AWS call itself. If the saga were reordered to + call AWS first, the probe recorded at ``delete_knowledge_base`` would be + ``False`` — a crash at that instant would leave a billed knowledge base + that no record and no tombstone points at. + """ + client = FakeBedrockAgent( + knowledge_bases=[_kb()], probe=lambda: _tomb_exists(table) + ) + + tomb.delete_knowledge_base( + ASSISTANT_ID, APP_KB_ID, AWS_KB_ID, AWS_DS_ID, client=client + ) + + probes = client.probes_for("delete_knowledge_base") + assert probes == [True], ( + "the tombstone was not present in DynamoDB at the moment " + "DeleteKnowledgeBase was called" + ) + + def test_tombstone_carries_the_aws_identifiers_and_intent(self, table): + tomb.write_kb_tombstone(ASSISTANT_ID, APP_KB_ID, AWS_KB_ID, AWS_DS_ID) + + item = _tomb_item(table) + assert item["intent"] == tomb.INTENT_DELETE_KB + assert item["awsKbId"] == AWS_KB_ID + assert item["awsDataSourceId"] == AWS_DS_ID + assert item["appKbId"] == APP_KB_ID + assert item["createdAt"] + + def test_retry_preserves_created_at_and_counts_attempts(self, table): + """A retried saga must not restart the clock on a stuck delete.""" + tomb.write_kb_tombstone(ASSISTANT_ID, APP_KB_ID, AWS_KB_ID, AWS_DS_ID) + first = _tomb_item(table) + + tomb.write_kb_tombstone(ASSISTANT_ID, APP_KB_ID, AWS_KB_ID, AWS_DS_ID) + second = _tomb_item(table) + + assert second["createdAt"] == first["createdAt"] + assert int(second["attempts"]) == 2 + + +# ── No TTL, ever (Requirement 13.6) ────────────────────────────────────────── +class TestTombstonesHaveNoTtl: + def test_kb_tombstone_carries_no_expiry_attribute(self, table): + """A TTL would let DynamoDB delete the evidence of an unfinished delete. + + That is the precise silent-leak class this design closes, so the absence + of *any* expiry-shaped attribute is asserted rather than assumed. + """ + tomb.write_kb_tombstone(ASSISTANT_ID, APP_KB_ID, AWS_KB_ID, AWS_DS_ID) + + item = _tomb_item(table) + expiry_attrs = {"ttl", "TTL", "expiresAt", "expires_at", "expireAt", "ttlEpoch"} + found = expiry_attrs & set(item) + assert not found, ( + f"tombstone carries expiry attribute(s) {sorted(found)}; a tombstone " + f"must be cleared by confirmed deletion or persist as a work item" + ) + + def test_document_tombstone_carries_no_expiry_attribute(self, table): + tomb.write_document_tombstone( + ASSISTANT_ID, APP_KB_ID, DOCUMENT_ID, AWS_KB_ID, AWS_DS_ID + ) + + item = table.get_item( + Key={ + "PK": f"AST#{ASSISTANT_ID}", + "SK": f"KBTOMB#{APP_KB_ID}#DOC#{DOCUMENT_ID}", + } + )["Item"] + expiry_attrs = {"ttl", "TTL", "expiresAt", "expires_at", "expireAt", "ttlEpoch"} + assert not (expiry_attrs & set(item)) + + +# ── Requirement 13.2, 13.3, 13.4: confirmation by polling ──────────────────── +class TestClearOnlyAfterConfirmedAbsence: + def test_tombstone_survives_every_poll_and_is_gone_only_at_the_end(self, table): + """The tombstone must outlive every poll in which AWS still lists the KB. + + ``polls_before_gone=3`` models the measured asynchronous deletion. The + probe on each list call captures whether the tombstone was still there; + clearing on the accepted delete call instead would make the later + observations ``False``. + """ + client = FakeBedrockAgent( + knowledge_bases=[_kb()], + polls_before_gone=3, + probe=lambda: _tomb_exists(table), + ) + slept = [] + + outcome = tomb.delete_knowledge_base( + ASSISTANT_ID, + APP_KB_ID, + AWS_KB_ID, + AWS_DS_ID, + client=client, + interval_seconds=0.0, + sleep=slept.append, + ) + + list_probes = client.probes_for("list_knowledge_bases") + assert list_probes, "the saga never polled ListKnowledgeBases" + assert all(list_probes), ( + f"the tombstone was cleared before AWS confirmed absence; " + f"per-poll presence was {list_probes}" + ) + assert outcome.confirmed is True + assert outcome.tombstone_cleared is True + assert outcome.polls == 3 + assert not _tomb_exists(table), "tombstone survived a confirmed deletion" + assert len(slept) == 2 + + def test_accepted_delete_alone_does_not_clear_the_tombstone(self, table): + """A knowledge base that never disappears must leave a work item. + + The stub's delete call succeeds — exactly like the real API — so anything + that treated the accepted call as completion would pass. The knowledge + base is still listed, so the saga must refuse to confirm. + """ + client = FakeBedrockAgent(knowledge_bases=[_kb()], polls_before_gone=10_000) + + with pytest.raises(tomb.DeleteNotConfirmed): + tomb.delete_knowledge_base( + ASSISTANT_ID, + APP_KB_ID, + AWS_KB_ID, + AWS_DS_ID, + client=client, + timeout_seconds=0.0, + interval_seconds=0.0, + sleep=lambda _s: None, + ) + + assert client.delete_calls == [AWS_KB_ID], "the delete call was never made" + item = _tomb_item(table) + assert item is not None, "an unconfirmed delete cleared its tombstone" + assert "still present" in item["lastError"] + + def test_clear_refuses_without_confirmation(self, table): + tomb.write_kb_tombstone(ASSISTANT_ID, APP_KB_ID, AWS_KB_ID, AWS_DS_ID) + + with pytest.raises(tomb.TombstoneError, match="has not confirmed"): + tomb.clear_kb_tombstone(ASSISTANT_ID, APP_KB_ID, False) + + assert _tomb_exists(table), "the refused clear removed the tombstone anyway" + + def test_clear_document_tombstone_refuses_without_confirmation(self, table): + tomb.write_document_tombstone(ASSISTANT_ID, APP_KB_ID, DOCUMENT_ID) + + with pytest.raises(tomb.TombstoneError): + tomb.clear_document_tombstone(ASSISTANT_ID, APP_KB_ID, DOCUMENT_ID, False) + + def test_poll_window_tolerates_at_least_six_minutes(self): + """Requirement 13.4. Deletion was measured at 2-6 minutes.""" + assert tomb.KB_DELETE_POLL_TIMEOUT_SECONDS >= 360.0, ( + f"the poll window is {tomb.KB_DELETE_POLL_TIMEOUT_SECONDS}s, below the " + f"360s floor Requirement 13.4 sets from measured deletions" + ) + + def test_timeout_is_read_at_call_time_not_bound_at_import(self, monkeypatch): + """Patching the module constant must actually change the behaviour. + + A default argument is evaluated once at import, so binding the timeout + that way makes it unpatchable: a test that shortens it appears to pass + while waiting the full production window. Here the constant is lowered to + zero, so a knowledge base that never disappears must fail immediately and + without sleeping. + """ + monkeypatch.setattr(tomb, "KB_DELETE_POLL_TIMEOUT_SECONDS", 0.0) + client = FakeBedrockAgent(knowledge_bases=[_kb()], polls_before_gone=10_000) + slept = [] + + with pytest.raises(tomb.DeleteNotConfirmed): + tomb.confirm_knowledge_base_absent(client, AWS_KB_ID, sleep=slept.append) + + assert slept == [], "the patched timeout was ignored; the poll slept anyway" + assert client.list_calls == 1 + + def test_already_absent_is_a_completed_delete(self, table): + """``ResourceNotFoundException`` means an earlier attempt finally landed.""" + client = FakeBedrockAgent(knowledge_bases=[]) + + outcome = tomb.delete_knowledge_base( + ASSISTANT_ID, APP_KB_ID, AWS_KB_ID, AWS_DS_ID, client=client + ) + + assert outcome.already_absent is True + assert outcome.confirmed is True + assert not _tomb_exists(table) + + +# ── Requirement 13.7: DELETE_UNSUCCESSFUL ──────────────────────────────────── +class TestDeleteUnsuccessfulIsAnOperatorState: + def test_delete_unsuccessful_raises_and_keeps_the_tombstone(self, table): + """The dev account has held one of these since 2025-11-24. + + It does not clear by waiting and it does not clear by retrying, so it must + surface as its own state rather than as a timeout or a success. + """ + client = FakeBedrockAgent( + knowledge_bases=[_kb(status="DELETE_UNSUCCESSFUL")], + polls_before_gone=10_000, + ) + + with pytest.raises(tomb.DeleteUnsuccessful): + tomb.delete_knowledge_base( + ASSISTANT_ID, + APP_KB_ID, + AWS_KB_ID, + AWS_DS_ID, + client=client, + interval_seconds=0.0, + sleep=lambda _s: None, + ) + + item = _tomb_item(table) + assert item is not None + assert item["awsStatus"] == tomb.KB_STATUS_DELETE_UNSUCCESSFUL + + def test_delete_unsuccessful_is_not_a_timeout(self, table): + """It must stop polling at once, not burn the window and misreport.""" + client = FakeBedrockAgent( + knowledge_bases=[_kb(status="DELETE_UNSUCCESSFUL")], + polls_before_gone=10_000, + ) + + with pytest.raises(tomb.DeleteUnsuccessful): + tomb.confirm_knowledge_base_absent( + client, AWS_KB_ID, interval_seconds=0.0, sleep=lambda _s: None + ) + + assert client.list_calls == 1 + + +# ── Requirement 13.6: the record outlives the delete ───────────────────────── +class TestRecordRemoval: + def test_refuses_to_remove_the_record_without_confirmation(self, table): + table.put_item( + Item={"PK": f"AST#{ASSISTANT_ID}", "SK": f"KB#{APP_KB_ID}", "appKbId": APP_KB_ID} + ) + + with pytest.raises(tomb.RecordRemovalRefused): + tomb.remove_kb_record(ASSISTANT_ID, APP_KB_ID, False) + + assert table.get_item( + Key={"PK": f"AST#{ASSISTANT_ID}", "SK": f"KB#{APP_KB_ID}"} + ).get("Item") + + def test_removes_the_record_once_confirmed(self, table): + table.put_item( + Item={"PK": f"AST#{ASSISTANT_ID}", "SK": f"KB#{APP_KB_ID}", "appKbId": APP_KB_ID} + ) + + tomb.remove_kb_record(ASSISTANT_ID, APP_KB_ID, True) + + assert not table.get_item( + Key={"PK": f"AST#{ASSISTANT_ID}", "SK": f"KB#{APP_KB_ID}"} + ).get("Item") + + def test_saga_removes_the_record_only_after_confirmation(self, table): + table.put_item( + Item={"PK": f"AST#{ASSISTANT_ID}", "SK": f"KB#{APP_KB_ID}", "appKbId": APP_KB_ID} + ) + client = FakeBedrockAgent(knowledge_bases=[_kb()], polls_before_gone=2) + + tomb.delete_knowledge_base( + ASSISTANT_ID, + APP_KB_ID, + AWS_KB_ID, + AWS_DS_ID, + client=client, + remove_record=True, + interval_seconds=0.0, + sleep=lambda _s: None, + ) + + assert not table.get_item( + Key={"PK": f"AST#{ASSISTANT_ID}", "SK": f"KB#{APP_KB_ID}"} + ).get("Item") + + def test_unconfirmed_saga_leaves_the_record_alone(self, table): + table.put_item( + Item={"PK": f"AST#{ASSISTANT_ID}", "SK": f"KB#{APP_KB_ID}", "appKbId": APP_KB_ID} + ) + client = FakeBedrockAgent(knowledge_bases=[_kb()], polls_before_gone=10_000) + + with pytest.raises(tomb.DeleteNotConfirmed): + tomb.delete_knowledge_base( + ASSISTANT_ID, + APP_KB_ID, + AWS_KB_ID, + AWS_DS_ID, + client=client, + remove_record=True, + timeout_seconds=0.0, + interval_seconds=0.0, + sleep=lambda _s: None, + ) + + assert table.get_item( + Key={"PK": f"AST#{ASSISTANT_ID}", "SK": f"KB#{APP_KB_ID}"} + ).get("Item"), "an unconfirmed delete removed the KB_Record" + + +# ── Requirement 13.8: a survivor is discoverable work ──────────────────────── +class TestSurvivingTombstonesAreDiscoverable: + def test_iter_tombstones_returns_kb_and_document_survivors(self, table): + tomb.write_kb_tombstone(ASSISTANT_ID, APP_KB_ID, AWS_KB_ID, AWS_DS_ID) + tomb.write_document_tombstone(ASSISTANT_ID, APP_KB_ID, DOCUMENT_ID) + table.put_item( + Item={"PK": f"AST#{ASSISTANT_ID}", "SK": f"KB#{APP_KB_ID}", "appKbId": APP_KB_ID} + ) + + found = tomb.iter_tombstones(ASSISTANT_ID) + + assert [item["SK"] for item in found] == [ + f"KBTOMB#{APP_KB_ID}", + f"KBTOMB#{APP_KB_ID}#DOC#{DOCUMENT_ID}", + ] + + def test_confirmed_delete_leaves_no_work_item(self, table): + client = FakeBedrockAgent(knowledge_bases=[_kb()]) + + tomb.delete_knowledge_base( + ASSISTANT_ID, APP_KB_ID, AWS_KB_ID, AWS_DS_ID, client=client + ) + + assert tomb.iter_tombstones(ASSISTANT_ID) == [] + + +# ── Requirement 13.5: the service role goes last ───────────────────────────── +class TestServiceRoleGuard: + def test_refuses_while_a_knowledge_base_still_uses_the_role(self, table): + client = FakeBedrockAgent( + knowledge_bases=[_kb(role_arn="arn:aws:iam::1:role/kb")], + tags={ARN: PROJECT_TAGS}, + ) + + with pytest.raises(tomb.ServiceRoleStillInUse, match=AWS_KB_ID): + tomb.assert_service_role_deletable(client, "arn:aws:iam::1:role/kb") + + def test_a_deleting_knowledge_base_still_blocks_the_role(self, table): + """Mid-``DELETING`` counts as present: it needs the role to finish.""" + client = FakeBedrockAgent( + knowledge_bases=[_kb(status="DELETING", role_arn="arn:aws:iam::1:role/kb")], + tags={ARN: PROJECT_TAGS}, + ) + + with pytest.raises(tomb.ServiceRoleStillInUse): + tomb.assert_service_role_deletable(client, "arn:aws:iam::1:role/kb") + + def test_allows_deletion_once_all_are_absent(self, table): + client = FakeBedrockAgent(knowledge_bases=[], tags={}) + + tomb.assert_service_role_deletable(client, "arn:aws:iam::1:role/kb") + + def test_another_projects_knowledge_base_does_not_block_the_role(self, table): + """The guard is scoped by tag, like everything else here.""" + client = FakeBedrockAgent( + knowledge_bases=[_kb(role_arn="arn:aws:iam::1:role/kb")], + tags={ARN: kb_tags.build_tags("ast-x", "u-9", "someone-else", "prod")}, + ) + + tomb.assert_service_role_deletable(client, "arn:aws:iam::1:role/kb") + + +# ── Requirement 14.1: paginated, tag-filtered listing ──────────────────────── +class TestListingIsPaginatedAndTagFiltered: + def test_every_page_is_read(self, table): + kbs = [_kb(kb_id=f"KB{i:04d}") for i in range(7)] + client = FakeBedrockAgent(knowledge_bases=kbs, page_size=2) + + seen = [s["knowledgeBaseId"] for s in tomb.iter_knowledge_base_summaries(client)] + + assert len(seen) == 7, f"paging stopped early: {seen}" + assert client.list_calls == 4 + + def test_tag_filter_keeps_ours_and_drops_everything_else(self, table): + mine, theirs, untagged = _kb("KBMINE"), _kb("KBTHEIRS"), _kb("KBNOTAGS") + client = FakeBedrockAgent( + knowledge_bases=[mine, theirs, untagged], + tags={ + mine["knowledgeBaseArn"]: PROJECT_TAGS, + theirs["knowledgeBaseArn"]: kb_tags.build_tags("ast-y", "u-9", "other", "testenv"), + }, + ) + + found = list(tomb.iter_project_knowledge_bases(client)) + + assert [f.kb_id for f in found] == ["KBMINE"] + + def test_created_at_is_carried_through_from_aws(self, table): + """The Reconciler's age gate depends on this provenance.""" + created = datetime(2025, 3, 4, 5, 6, 7, tzinfo=timezone.utc) + kb = _kb() + kb["createdAt"] = created + client = FakeBedrockAgent(knowledge_bases=[kb], tags={ARN: PROJECT_TAGS}) + + found = list(tomb.iter_project_knowledge_bases(client)) + + assert found[0].created_at == created + + def test_untagged_never_matches(self): + assert tomb.matches_project_tags(None, {kb_tags.TAG_KEY_PREFIX: "p"}) is False + assert tomb.matches_project_tags({}, {kb_tags.TAG_KEY_PREFIX: "p"}) is False + assert tomb.matches_project_tags({kb_tags.TAG_KEY_PREFIX: "p"}, {kb_tags.TAG_KEY_PREFIX: "p"}) is True + assert tomb.matches_project_tags({kb_tags.TAG_KEY_PREFIX: "q"}, {kb_tags.TAG_KEY_PREFIX: "p"}) is False + + def test_a_knowledge_base_that_vanishes_between_list_and_describe_is_skipped(self, table): + client = FakeBedrockAgent(knowledge_bases=[_kb()], tags={ARN: PROJECT_TAGS}) + # Model the race: the summary is produced, then the resource is gone. + summaries = list(tomb.iter_knowledge_base_summaries(client)) + assert summaries + client.kbs.clear() + + assert list(tomb.iter_project_knowledge_bases(client)) == [] + + +# ── Document saga ──────────────────────────────────────────────────────────── +class TestDocumentSaga: + def test_tombstone_precedes_the_aws_document_delete(self, table): + def probe(): + return ( + table.get_item( + Key={ + "PK": f"AST#{ASSISTANT_ID}", + "SK": f"KBTOMB#{APP_KB_ID}#DOC#{DOCUMENT_ID}", + } + ).get("Item") + is not None + ) + + client = FakeBedrockAgent(probe=probe, document_statuses=["NOT_FOUND"]) + + tomb.delete_document( + ASSISTANT_ID, APP_KB_ID, DOCUMENT_ID, AWS_KB_ID, AWS_DS_ID, client=client + ) + + assert client.probes_for("delete_knowledge_base_documents") == [True] + + def test_deleting_status_is_not_absent(self, table): + """``DELETING`` is present. Treating it as gone is the document-scale + version of trusting the accepted delete call.""" + client = FakeBedrockAgent(document_statuses=["DELETING"]) + + with pytest.raises(tomb.DeleteNotConfirmed): + tomb.delete_document( + ASSISTANT_ID, + APP_KB_ID, + DOCUMENT_ID, + AWS_KB_ID, + AWS_DS_ID, + client=client, + timeout_seconds=0.0, + interval_seconds=0.0, + sleep=lambda _s: None, + ) + + assert table.get_item( + Key={ + "PK": f"AST#{ASSISTANT_ID}", + "SK": f"KBTOMB#{APP_KB_ID}#DOC#{DOCUMENT_ID}", + } + ).get("Item"), "an unconfirmed document delete cleared its tombstone" + + def test_cleared_once_not_found(self, table): + client = FakeBedrockAgent(document_statuses=["DELETING", "NOT_FOUND"]) + + outcome = tomb.delete_document( + ASSISTANT_ID, + APP_KB_ID, + DOCUMENT_ID, + AWS_KB_ID, + AWS_DS_ID, + client=client, + interval_seconds=0.0, + sleep=lambda _s: None, + ) + + assert outcome.confirmed is True + assert outcome.polls == 2 + assert not table.get_item( + Key={ + "PK": f"AST#{ASSISTANT_ID}", + "SK": f"KBTOMB#{APP_KB_ID}#DOC#{DOCUMENT_ID}", + } + ).get("Item") + + +# ── Failure annotation ─────────────────────────────────────────────────────── +class TestFailureAnnotation: + def test_a_transport_failure_leaves_an_annotated_tombstone(self, table): + client = FakeBedrockAgent( + knowledge_bases=[_kb()], + delete_raises=ClientError( + {"Error": {"Code": "ThrottlingException", "Message": "slow down"}}, + "DeleteKnowledgeBase", + ), + ) + + with pytest.raises(ClientError): + tomb.delete_knowledge_base( + ASSISTANT_ID, APP_KB_ID, AWS_KB_ID, AWS_DS_ID, client=client + ) + + item = _tomb_item(table) + assert item is not None + assert "Throttling" in item["lastError"] diff --git a/backend/tests/shared/test_kb_work_index_guard.py b/backend/tests/shared/test_kb_work_index_guard.py new file mode 100644 index 000000000..1a5959f96 --- /dev/null +++ b/backend/tests/shared/test_kb_work_index_guard.py @@ -0,0 +1,131 @@ +"""``GSI7_*`` (KbWorkIndex) is never written by the generic assistant update. + +Managed-KB migration work is discovered through a sparse index whose keys are written +only while a knowledge base is actually eligible for background work +(.kiro/specs/managed-kb-migration, Requirements 15.13/15.14). The dispatcher that reads +it creates and deletes billed AWS resources, so a stray key is not a cosmetic bug: it +hands the dispatcher a knowledge base nobody asked to migrate. + +``Assistant`` is ``extra="allow"`` and reads hydrate straight from the raw DynamoDB +item, so any attribute present on the row round-trips as an extra model field and the +generic update path would write it back. That is exactly how ``GSI5_*`` could +re-publish a taken-down agent before it was listed immutable — see +``test_stale_edit_racing_a_takedown_does_not_resurrect_the_directory_key``. + +``GSI7_*`` differs from ``GSI5_*`` in one respect worth stating, because it changes what +this test is actually worth: the work keys live on a **separate** item +(``SK = KB#{app_kb_id}``), not on the ``METADATA`` row this path writes, so today the +generic update cannot reach them by the normal route. These tests therefore pin the +invariant rather than reproduce a live bug — they fail if someone removes ``GSI7_*`` +from ``immutable_fields``, or if a future refactor moves work state onto the assistant +row without re-establishing the guard. +""" + +import pytest + + +async def _create(owner_id: str = "u1"): + from apis.shared.assistants.service import create_assistant + + return await create_assistant( + owner_id=owner_id, + owner_name="Alice", + name="Bot", + description="d", + instructions="hi", + ) + + +def _item(table_name: str, assistant_id: str): + import boto3 + + table = boto3.resource("dynamodb", region_name="us-east-1").Table(table_name) + return table.get_item(Key={"PK": f"AST#{assistant_id}", "SK": "METADATA"})["Item"] + + +class TestKbWorkIndexGuard: + @pytest.fixture(autouse=True) + def _set_env(self, monkeypatch): + monkeypatch.setenv("S3_ASSISTANTS_VECTOR_STORE_INDEX_NAME", "test-index") + + @pytest.mark.asyncio + async def test_a_stale_edit_cannot_resurrect_a_cleared_work_key(self, assistants_table): + """The headline guard, exercised through the real mechanism and the real sequence. + + Two details make or break this test: + + 1. The keys must be read back through ``get_assistant`` so they hydrate into + ``__pydantic_extra__`` and reach ``model_dump``. Simulating that with + ``object.__setattr__`` bypasses the extras dict, so the attribute never reaches + the update payload and the test passes even with the guard removed. + 2. The keys must be **cleared from the row** before the stale write lands. + ``immutable_fields`` prevents the generic update from *writing* an attribute; + it does not delete one that is already there. Without the clear, the key is + still on the item afterwards for the trivial reason that nothing removed it, + and the assertion passes or fails for reasons unrelated to the guard. + + That is the true shape of the exposure: work finishes and the dispatcher clears + the key, an author edit that began earlier still carries it, and the write lands. + """ + import boto3 + + from apis.shared.assistants.service import _update_assistant_cloud, get_assistant + + created = await _create() + table = boto3.resource("dynamodb", region_name="us-east-1").Table("test-assistants") + key = {"PK": f"AST#{created.assistant_id}", "SK": "METADATA"} + + # 1. KB is enrolled: work keys present on the row. + table.update_item( + Key=key, + UpdateExpression="SET GSI7_PK = :p, GSI7_SK = :s", + ExpressionAttributeValues={":p": "KBWORK#shadow", ":s": "2026-08-17T00:00:00Z"}, + ) + + # 2. An author's request reads it while still enrolled. + stale = await get_assistant(created.assistant_id, "u1") + assert stale.model_dump(by_alias=True).get("GSI7_PK") == "KBWORK#shadow", ( + "precondition: the key must hydrate as an extra field, or this test is vacuous" + ) + + # 3. Work completes and the dispatcher clears the key. + table.update_item(Key=key, UpdateExpression="REMOVE GSI7_PK, GSI7_SK") + assert "GSI7_PK" not in _item("test-assistants", created.assistant_id) + + # 4. The in-flight author edit lands, still carrying the stale key. + stale.description = "An unrelated tweak" + await _update_assistant_cloud(stale, "test-assistants") + + item = _item("test-assistants", created.assistant_id) + assert item["description"] == "An unrelated tweak", "the legitimate edit must land" + assert "GSI7_PK" not in item, "a stale edit re-enrolled a KB that had left the queue" + assert "GSI7_SK" not in item, "a stale edit restored a work-discovery sort key" + + @pytest.mark.asyncio + async def test_the_guard_lists_both_work_key_attributes(self): + """Pins the guard itself, so removing one half of the pair fails loudly. + + Asserted against the source set rather than through behaviour because a missing + *sort* key is invisible to a query-free test: DynamoDB accepts a partial GSI key + by simply not indexing the item. + """ + import inspect + + from apis.shared.assistants import service + + src = inspect.getsource(service._update_assistant_cloud) + immutable = src.split("immutable_fields = {", 1)[1].split("}", 1)[0] + assert '"GSI7_PK"' in immutable + assert '"GSI7_SK"' in immutable + + @pytest.mark.asyncio + async def test_a_normal_create_writes_no_work_key(self, assistants_table): + """Absence of the key is the "not enrolled" state, so a fresh agent must have none. + + This is what makes the index sparse: enrolment is an explicit write, never a + side effect of creating an assistant. + """ + created = await _create() + item = _item("test-assistants", created.assistant_id) + assert "GSI7_PK" not in item + assert "GSI7_SK" not in item diff --git a/backend/tests/shared/test_managed_kb_backend.py b/backend/tests/shared/test_managed_kb_backend.py new file mode 100644 index 000000000..db6ac3411 --- /dev/null +++ b/backend/tests/shared/test_managed_kb_backend.py @@ -0,0 +1,1398 @@ +"""Managed knowledge base provisioning, retrieval, ingestion and deletion. + +Every AWS interaction here is stubbed. The ``bedrock-agent`` and +``bedrock-agent-runtime`` clients are hand-rolled fakes (no network client is ever +constructed) and DynamoDB is moto, so the conditional writes the provisioning saga +depends on are exercised for real rather than mocked into always succeeding. +Nothing in this file can reach AWS: there is no ``boto3.client`` call for either +Bedrock service, and moto intercepts the DynamoDB ones. + +Two things are asserted here that no other test in the suite can see: + +* **Order.** The KB_Record must exist, in ``provisioning``, *before* + ``CreateKnowledgeBase`` is called. The fake client reads the record from moto at + the moment it is called, which turns "written first" from a code-reading + exercise into an assertion. +* **Shape.** The managed API shapes contradict the documentation in several + places — a 33-character ``clientToken`` minimum, a 10-document batch cap where + the user guide says 25, a nested connector type, ``managedSearchConfiguration`` + instead of ``vectorSearchConfiguration``. Each is pinned by a test whose failure + message says why, because each looks like a mistake to anyone who checks the + docs. + +Feature: managed-kb-migration +Requirements: 7.1-7.8, 8.1-8.8, 9.1-9.6, 11.1-11.5, 20.7, 24.2, 24.3, 24.11 +""" + +import ast +import subprocess +import sys +import threading +from pathlib import Path +from typing import Any, Dict, List, Optional, Tuple + +import boto3 +import pytest +from moto import mock_aws + +from apis.shared.kb_backend import tags as kb_tags +from apis.shared.kb_backend import managed_backend as mb +from apis.shared.kb_backend import provisioning as p +from apis.shared.kb_backend import records as r +from apis.shared.kb_backend.protocol import DEFAULT_TOP_K, Chunk, DocumentSource +from apis.shared.kb_backend.protocol import KnowledgeBaseBackend + +REGION = "us-east-1" +TABLE = "test-managed-kb" +ASSISTANT_ID = "ast-managed-001" +APP_KB_ID = ASSISTANT_ID # App_KB_Id == assistant_id in this phase +OWNER = "owner-opaque-42" +ROLE_ARN = "arn:aws:iam::123456789012:role/test-managed-kb-role" +AWS_KB_ID = "KBAAAAAAAA" +AWS_DS_ID = "DSAAAAAAAA" + + +# --------------------------------------------------------------------------- +# Stubs +# --------------------------------------------------------------------------- + + +class FakeBedrockAgent: + """A ``bedrock-agent`` control-plane stub with real idempotency semantics. + + ``clientToken`` deduplication is modelled rather than ignored, because it is + the mechanism the crash-recovery path relies on: a retry that reuses the token + must receive the *same* knowledge base, not a second one. A stub that minted a + fresh id per call would let a duplicate-creating bug pass. + + ``on_create`` runs at the moment ``create_knowledge_base`` is entered, which is + how the record-before-AWS ordering is observed. + """ + + def __init__( + self, + *, + on_create=None, + create_failures: Optional[List[Exception]] = None, + ) -> None: + self.create_kb_calls: List[Dict[str, Any]] = [] + self.create_ds_calls: List[Dict[str, Any]] = [] + self.ingest_calls: List[Dict[str, Any]] = [] + self.delete_calls: List[Dict[str, Any]] = [] + self.start_ingestion_job_calls: List[Dict[str, Any]] = [] + self.thread_idents: List[int] = [] + self.observed_records: List[Optional[Dict[str, Any]]] = [] + self._by_token: Dict[str, str] = {} + self._ds_by_token: Dict[str, str] = {} + self._on_create = on_create + self._create_failures = list(create_failures or []) + self._counter = 0 + self._lock = threading.Lock() + self._in_flight = 0 + self.max_in_flight = 0 + + # -- provisioning ------------------------------------------------------ + def create_knowledge_base(self, **kwargs): + self.thread_idents.append(threading.get_ident()) + if self._on_create is not None: + self.observed_records.append(self._on_create()) + self.create_kb_calls.append(kwargs) + + if self._create_failures: + raise self._create_failures.pop(0) + + token = kwargs["clientToken"] + if token not in self._by_token: + self._counter += 1 + self._by_token[token] = f"KB{self._counter:08d}" + return {"knowledgeBase": {"knowledgeBaseId": self._by_token[token], "status": "ACTIVE"}} + + def create_data_source(self, **kwargs): + self.thread_idents.append(threading.get_ident()) + self.create_ds_calls.append(kwargs) + token = kwargs["clientToken"] + if token not in self._ds_by_token: + self._ds_by_token[token] = f"DS{len(self._ds_by_token) + 1:08d}" + return {"dataSource": {"dataSourceId": self._ds_by_token[token], "status": "AVAILABLE"}} + + @property + def distinct_knowledge_base_ids(self) -> set: + return set(self._by_token.values()) + + # -- documents --------------------------------------------------------- + def _enter(self): + with self._lock: + self._in_flight += 1 + self.max_in_flight = max(self.max_in_flight, self._in_flight) + + def _exit(self): + with self._lock: + self._in_flight -= 1 + + def ingest_knowledge_base_documents(self, **kwargs): + self._enter() + try: + self.thread_idents.append(threading.get_ident()) + self.ingest_calls.append(kwargs) + # Real calls take time; without a pause every coroutine would finish + # before the next started and the concurrency bound would be + # untestable (max in flight would read 1 no matter what it was). + threading.Event().wait(0.02) + return {"documentDetails": []} + finally: + self._exit() + + def delete_knowledge_base_documents(self, **kwargs): + self._enter() + try: + self.delete_calls.append(kwargs) + threading.Event().wait(0.02) + return {"documentDetails": []} + finally: + self._exit() + + def start_ingestion_job(self, **kwargs): # pragma: no cover - must never run + self.start_ingestion_job_calls.append(kwargs) + raise AssertionError( + "StartIngestionJob must never be called (Requirement 9.2): it is " + "0.1 RPS account-wide and not adjustable" + ) + + +class FakeBedrockAgentRuntime: + """A ``bedrock-agent-runtime`` stub returning canned ``Retrieve`` results.""" + + def __init__(self, results: Optional[List[Dict[str, Any]]] = None) -> None: + self.results = results if results is not None else [] + self.calls: List[Dict[str, Any]] = [] + self.thread_idents: List[int] = [] + + def retrieve(self, **kwargs): + self.thread_idents.append(threading.get_ident()) + self.calls.append(kwargs) + return {"retrievalResults": self.results} + + +def _client_error(code: str, message: str): + """A ``ClientError`` shaped exactly as botocore raises one.""" + from botocore.exceptions import ClientError + + return ClientError({"Error": {"Code": code, "Message": message}}, "CreateKnowledgeBase") + + +def _result(document_id: str, score: float, text: str = "passage") -> Dict[str, Any]: + """One ``Retrieve`` result, as the service returns it for a CUSTOM connector.""" + return { + "content": {"text": text}, + "score": score, + "location": { + "type": "CUSTOM", + "customDocumentLocation": {"id": document_id}, + }, + "metadata": {"filename": f"{document_id}.pdf"}, + } + + +# --------------------------------------------------------------------------- +# Fixtures +# --------------------------------------------------------------------------- + + +@pytest.fixture() +def table(monkeypatch): + """The assistants table, including the GSI7 work index. + + Built here rather than taken from the shared ``assistants_table`` fixture, + which predates GSI7 — the same reason ``test_kb_records.py`` builds its own. + """ + monkeypatch.setenv("AWS_DEFAULT_REGION", REGION) + monkeypatch.setenv("AWS_ACCESS_KEY_ID", "testing") + monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "testing") + monkeypatch.setenv("AWS_SESSION_TOKEN", "testing") + monkeypatch.setenv("DYNAMODB_ASSISTANTS_TABLE_NAME", TABLE) + monkeypatch.setenv(kb_tags.ENV_TAG_VALUE_PREFIX, "test-prefix") + monkeypatch.setenv(kb_tags.ENV_TAG_VALUE_ENVIRONMENT, "test") + + with mock_aws(): + ddb = boto3.client("dynamodb", region_name=REGION) + ddb.create_table( + TableName=TABLE, + KeySchema=[ + {"AttributeName": "PK", "KeyType": "HASH"}, + {"AttributeName": "SK", "KeyType": "RANGE"}, + ], + AttributeDefinitions=[ + {"AttributeName": "PK", "AttributeType": "S"}, + {"AttributeName": "SK", "AttributeType": "S"}, + {"AttributeName": "GSI7_PK", "AttributeType": "S"}, + {"AttributeName": "GSI7_SK", "AttributeType": "S"}, + ], + GlobalSecondaryIndexes=[ + { + "IndexName": "KbWorkIndex", + "KeySchema": [ + {"AttributeName": "GSI7_PK", "KeyType": "HASH"}, + {"AttributeName": "GSI7_SK", "KeyType": "RANGE"}, + ], + "Projection": {"ProjectionType": "ALL"}, + } + ], + BillingMode="PAY_PER_REQUEST", + ) + yield boto3.resource("dynamodb", region_name=REGION).Table(TABLE) + + +def _record(table) -> Optional[Dict[str, Any]]: + return table.get_item( + Key={"PK": r.kb_pk(ASSISTANT_ID), "SK": r.kb_sk(APP_KB_ID)} + ).get("Item") + + +async def _no_sleep(_seconds: float) -> None: + """Retry backoff, without the wait.""" + return None + + +async def _provision(client, **overrides): + kwargs: Dict[str, Any] = dict( + assistant_id=ASSISTANT_ID, + app_kb_id=APP_KB_ID, + owner_user_id=OWNER, + role_arn=ROLE_ARN, + client=client, + region=REGION, + sleep=_no_sleep, + ) + kwargs.update(overrides) + return await p.provision_managed_kb(**kwargs) + + +def _locator(kb_id: str = AWS_KB_ID, ds_id: str = AWS_DS_ID): + def _locate(_kb_ref: str) -> Tuple[str, str]: + return kb_id, ds_id + + return _locate + + +# =========================================================================== +# 8.1 — clientToken construction +# =========================================================================== + + +class TestClientToken: + """The 33-character minimum, which the natural template silently violates.""" + + def test_built_token_clears_the_minimum(self): + token = p.build_client_token(APP_KB_ID, "knowledge-base") + assert len(token) >= p.CLIENT_TOKEN_MIN_LENGTH + + def test_the_natural_interpolated_template_is_too_short_and_is_rejected(self): + """This is the bug the builder exists to prevent, spelled out. + + ``{id}-{variant}-kb`` reads as obviously fine and is 31 characters for a + realistic id, four short of the API's minimum. It fails in botocore's + client-side validation, so there is no service error and no request id to + search for. + """ + naive = f"{APP_KB_ID}-managed-kb" + assert len(naive) < p.CLIENT_TOKEN_MIN_LENGTH, ( + "the premise of this test is that the natural template is too short; " + f"{naive!r} is {len(naive)} characters" + ) + with pytest.raises(ValueError, match="at least 33 characters"): + p.validate_client_token(naive) + + @pytest.mark.parametrize("app_kb_id", ["a", "ast-1", "x" * 300, "ast_with_underscores"]) + def test_tokens_are_valid_for_any_input_length(self, app_kb_id): + """Short, long and illegal-alphabet inputs all produce a legal token.""" + token = p.build_client_token(app_kb_id, "knowledge-base") + p.validate_client_token(token) # raises if not + assert p.CLIENT_TOKEN_MIN_LENGTH <= len(token) <= p.CLIENT_TOKEN_MAX_LENGTH + assert p.CLIENT_TOKEN_PATTERN.match(token) + + def test_token_is_deterministic_so_a_retry_reuses_it(self): + """Determinism is what makes AWS deduplicate a retried create.""" + assert p.build_client_token(APP_KB_ID, "knowledge-base") == p.build_client_token( + APP_KB_ID, "knowledge-base" + ) + + def test_different_knowledge_bases_get_different_tokens(self): + assert p.build_client_token("ast-a", "knowledge-base") != p.build_client_token( + "ast-b", "knowledge-base" + ) + + def test_kb_and_data_source_tokens_differ(self): + """Two idempotency scopes, two tokens; sharing one would conflate them.""" + assert p.build_client_token(APP_KB_ID, "knowledge-base") != p.build_client_token( + APP_KB_ID, "data-source" + ) + + def test_a_token_may_not_end_with_a_hyphen(self): + with pytest.raises(ValueError, match="pattern"): + p.validate_client_token("a" * 40 + "-") + + def test_an_over_long_token_is_rejected(self): + with pytest.raises(ValueError, match="at most 256"): + p.validate_client_token("a" * 257) + + +# =========================================================================== +# 8.1 / 8.2 — payload shapes +# =========================================================================== + + +class TestKnowledgeBasePayload: + def _payload(self): + return p.knowledge_base_payload( + name="test-prefix-kb-ast", + role_arn=ROLE_ARN, + client_token=p.build_client_token(APP_KB_ID, "knowledge-base"), + region=REGION, + ) + + def test_type_is_managed(self): + assert self._payload()["knowledgeBaseConfiguration"]["type"] == "MANAGED" + + def test_managed_configuration_is_present(self): + assert "managedKnowledgeBaseConfiguration" in self._payload()["knowledgeBaseConfiguration"] + + def test_storage_configuration_is_omitted_entirely(self): + """Requirement 8.2. There is no vector store, and sending one is rejected.""" + assert "storageConfiguration" not in self._payload() + + def test_role_arn_is_passed(self): + assert self._payload()["roleArn"] == ROLE_ARN + + def test_embedding_is_pinned_to_titan_v2_float32_1024(self): + """Requirement 8.5, and immutable from here on (8.8). + + A drift in any of these three values is not a migration but a rebuild, so + the numbers are asserted rather than merely present. + """ + managed = self._payload()["knowledgeBaseConfiguration"][ + "managedKnowledgeBaseConfiguration" + ] + assert managed["embeddingModelType"] == "CUSTOM" + assert managed["embeddingModelArn"].endswith("amazon.titan-embed-text-v2:0") + bedrock_config = managed["embeddingModelConfiguration"][ + "bedrockEmbeddingModelConfiguration" + ] + assert bedrock_config["dimensions"] == 1024 + assert bedrock_config["embeddingDataType"] == "FLOAT32" + + def test_kms_key_is_only_sent_when_supplied(self): + assert "serverSideEncryptionConfiguration" not in self._payload()[ + "knowledgeBaseConfiguration" + ]["managedKnowledgeBaseConfiguration"] + + with_kms = p.knowledge_base_payload( + name="n", + role_arn=ROLE_ARN, + client_token=p.build_client_token(APP_KB_ID, "kb"), + kms_key_arn="arn:aws:kms:us-east-1:123456789012:key/abc", + ) + assert ( + with_kms["knowledgeBaseConfiguration"]["managedKnowledgeBaseConfiguration"][ + "serverSideEncryptionConfiguration" + ]["kmsKeyArn"] + == "arn:aws:kms:us-east-1:123456789012:key/abc" + ) + + def test_a_short_client_token_is_refused_before_the_request_is_built(self): + with pytest.raises(ValueError, match="at least 33"): + p.knowledge_base_payload(name="n", role_arn=ROLE_ARN, client_token="short") + + +class TestDataSourcePayload: + def _payload(self): + return p.data_source_payload( + knowledge_base_id=AWS_KB_ID, + name="test-prefix-kb-ast", + client_token=p.build_client_token(APP_KB_ID, "data-source"), + ) + + def test_data_source_type_is_the_managed_connector(self): + """Requirement 8.3. A top-level ``CUSTOM`` is rejected outright.""" + config = self._payload()["dataSourceConfiguration"] + assert config["type"] == "MANAGED_KNOWLEDGE_BASE_CONNECTOR" + assert config["type"] != "CUSTOM", ( + "the classic top-level CUSTOM/S3/WEB types are rejected for managed " + "knowledge bases with 'Unsupported data source type'" + ) + + def test_the_real_connector_type_is_nested_in_connector_parameters(self): + """Requirement 8.4 — ``CUSTOM`` lives one level down, not at the top.""" + connector = self._payload()["dataSourceConfiguration"][ + "managedKnowledgeBaseConnectorConfiguration" + ] + assert connector["connectorParameters"] == {"type": "CUSTOM", "version": "1"} + + def test_image_extraction_is_enabled(self): + """Requirement 8.6. Opt-in: left default, no image or chart content is + indexed at all, silently, while the bill is unchanged.""" + connector = self._payload()["dataSourceConfiguration"][ + "managedKnowledgeBaseConnectorConfiguration" + ] + status = connector["mediaExtractionConfiguration"]["imageExtractionConfiguration"][ + "imageExtractionStatus" + ] + assert status == "ENABLED" + + def test_data_deletion_policy_is_retain_at_creation(self): + """Requirement 8.7 — the documented remedy for DELETE_UNSUCCESSFUL, and it + must be set at creation because it cannot rescue a stuck knowledge base + afterwards. The dev account has one stuck since 2025-11-24.""" + assert self._payload()["dataDeletionPolicy"] == "RETAIN" + + def test_a_short_client_token_is_refused(self): + with pytest.raises(ValueError, match="at least 33"): + p.data_source_payload(knowledge_base_id=AWS_KB_ID, name="n", client_token="short") + + +class TestTags: + def test_the_resource_name_uses_the_same_prefix_as_the_tags(self, monkeypatch): + """A name that says one deployment while the tag says another is the kind + of thing an operator reads once and trusts. + + Every filter in this feature matches on tags, so the name is only a + convention — but it is resolved through the same helper precisely so the + two cannot disagree. Asserted because the docstring claims it. + """ + monkeypatch.setenv(kb_tags.ENV_TAG_VALUE_PREFIX, "from-tag-var") + monkeypatch.delenv("PROJECT_PREFIX", raising=False) + + name = p._resource_name("ast-1") + tags = p.build_tags("ast-1", "u-1") + + assert name.startswith(f"{tags[kb_tags.TAG_KEY_PREFIX]}-kb-") + assert name == "from-tag-var-kb-ast-1" + + def test_tags_carry_prefix_env_kb_and_owner(self): + tags = p.build_tags(APP_KB_ID, OWNER, project_prefix="pfx", environment="dev") + assert tags == { + kb_tags.TAG_KEY_PREFIX: "pfx", + kb_tags.TAG_KEY_ENVIRONMENT: "dev", + kb_tags.TAG_KEY_APP_KB_ID: APP_KB_ID, + kb_tags.TAG_KEY_OWNER_USER_ID: OWNER, + } + + def test_an_email_owner_tag_is_refused(self): + """Requirement 20.12. Tags are readable by anyone with ListKnowledgeBases.""" + with pytest.raises(ValueError, match="opaque identifier"): + p.build_tags(APP_KB_ID, "student@example.edu") + + +# =========================================================================== +# 8.1 — the saga +# =========================================================================== + + +class TestProvisioningSaga: + @pytest.mark.asyncio + async def test_happy_path_creates_and_attaches(self, table): + client = FakeBedrockAgent() + result = await _provision(client) + + assert result.created is True + assert len(client.create_kb_calls) == 1 + assert len(client.create_ds_calls) == 1 + + item = _record(table) + assert item["awsKbId"] == result.aws_kb_id + assert item["awsDataSourceId"] == result.aws_data_source_id + assert item["provisioningState"] == r.ACTIVE + + @pytest.mark.asyncio + async def test_the_record_exists_in_provisioning_before_aws_is_called(self, table): + """Requirement 7.3, and the reason this whole ordering exists. + + The fake client reads the record at the instant ``CreateKnowledgeBase`` is + entered. Written afterwards instead, a crash in between would leave a + billed knowledge base that no record points at and nothing can find — no + exception, no alarm, just an invoice. + """ + client = FakeBedrockAgent(on_create=lambda: _record(table)) + await _provision(client) + + assert client.observed_records, "the create hook never ran" + seen = client.observed_records[0] + assert seen is not None, ( + "CreateKnowledgeBase was called before the KB_Record existed: a crash " + "at that point would strand an untraceable paying resource" + ) + assert seen["provisioningState"] == r.PROVISIONING + assert "awsKbId" not in seen + + @pytest.mark.asyncio + async def test_the_client_token_is_persisted_and_is_the_one_sent(self, table): + """Persisted so a *later* process, with no memory of this one, can retry.""" + client = FakeBedrockAgent() + await _provision(client) + + stored = _record(table)["clientToken"] + assert len(stored) >= p.CLIENT_TOKEN_MIN_LENGTH + assert client.create_kb_calls[0]["clientToken"] == stored + + @pytest.mark.asyncio + async def test_tags_are_applied_at_creation(self, table): + client = FakeBedrockAgent() + await _provision(client) + tags = client.create_kb_calls[0]["tags"] + assert tags[kb_tags.TAG_KEY_APP_KB_ID] == APP_KB_ID + assert tags[kb_tags.TAG_KEY_OWNER_USER_ID] == OWNER + + @pytest.mark.asyncio + async def test_a_second_call_creates_nothing(self, table): + """Requirement 7.4 — idempotent once the record carries both identifiers.""" + client = FakeBedrockAgent() + first = await _provision(client) + second = await _provision(client) + + assert second.created is False + assert (second.aws_kb_id, second.aws_data_source_id) == ( + first.aws_kb_id, + first.aws_data_source_id, + ) + assert len(client.create_kb_calls) == 1 + + @pytest.mark.asyncio + async def test_the_loser_of_the_enrolment_race_does_not_create_a_second_kb(self, table): + """Requirement 7.4. The loser must wait, not proceed. + + A loser that carried on with its own token would create a second knowledge + base, and only one of the two could ever be recorded — the other would be + billed forever with nothing pointing at it. + """ + r.create_provisioning( + ASSISTANT_ID, + r.KbRecord(app_kb_id=APP_KB_ID, owner_user_id=OWNER, client_token="z" * 40), + ) + # Simulate the race: the winner's record is already there, so the loser's + # own create_provisioning is rejected by the attribute_not_exists guard. + client = FakeBedrockAgent() + table.delete_item(Key={"PK": r.kb_pk(ASSISTANT_ID), "SK": r.kb_sk(APP_KB_ID)}) + + original = r.create_provisioning + + def _lose(assistant_id, record): + original(assistant_id, record) # the "winner" writes first + raise r.TransitionLost("simulated race loss") + + r.create_provisioning = _lose + try: + with pytest.raises(p.ProvisioningInProgress): + await _provision(client) + finally: + r.create_provisioning = original + + assert client.create_kb_calls == [], ( + "the losing worker called CreateKnowledgeBase anyway, which is how a " + "second knowledge base gets created for one record" + ) + + @pytest.mark.asyncio + async def test_provisioning_requires_a_service_role(self, table): + with pytest.raises(p.ProvisioningError, match="service role"): + await _provision(FakeBedrockAgent(), role_arn=None) + + +class TestRetryableEmbeddingVerification: + """Requirement 7.7 — the failure that looks fatal and is not.""" + + @pytest.mark.asyncio + async def test_embedding_verification_failure_is_retried_and_succeeds(self, table): + """It is IAM eventual consistency against a model confirmed ACTIVE. + + Treated as fatal, lazy provisioning fails intermittently and the message + sends the operator to check a model that is demonstrably fine. + """ + client = FakeBedrockAgent( + create_failures=[ + _client_error( + "ValidationException", + "Unable to verify the specified embedding model", + ) + ] + ) + result = await _provision(client) + + assert result.created is True + assert len(client.create_kb_calls) == 2, ( + "the embedding-verification failure was not retried" + ) + assert _record(table)["provisioningState"] == r.ACTIVE + + @pytest.mark.asyncio + async def test_retries_reuse_the_same_client_token(self, table): + """Otherwise a retry is a second create, not a retry.""" + client = FakeBedrockAgent( + create_failures=[ + _client_error("ValidationException", "unable to verify the specified embedding model") + ] + ) + await _provision(client) + tokens = {call["clientToken"] for call in client.create_kb_calls} + assert len(tokens) == 1 + assert client.distinct_knowledge_base_ids == {"KB00000001"} + + @pytest.mark.asyncio + async def test_a_genuine_validation_error_is_not_retried(self, table): + """A malformed request must fail fast: retrying only delays the report.""" + client = FakeBedrockAgent( + create_failures=[_client_error("ValidationException", "roleArn is invalid")] + ) + with pytest.raises(Exception, match="roleArn is invalid"): + await _provision(client) + assert len(client.create_kb_calls) == 1 + + def test_classification_of_the_embedding_message(self): + assert p.is_retryable_error( + _client_error("ValidationException", "Unable to verify the specified embedding model") + ) + assert p.is_retryable_error(_client_error("ThrottlingException", "slow down")) + assert not p.is_retryable_error(_client_error("AccessDeniedException", "no")) + + +# =========================================================================== +# 8.6 — crash between the AWS create and the record update +# =========================================================================== + + +class TestCrashBetweenCreateAndRecordUpdate: + """Requirements 7.8, 24.3. + + The window that record-first ordering exists to make survivable: the AWS + knowledge base exists, the record does not yet name it. + """ + + @pytest.mark.asyncio + async def test_the_record_survives_as_a_discoverable_retry_anchor(self, table): + client = FakeBedrockAgent() + crashed = [] + + def _crash(*_args, **_kwargs): + crashed.append(True) + raise RuntimeError("worker died before attaching identifiers") + + original = r.attach_aws_ids + r.attach_aws_ids = _crash + try: + with pytest.raises(RuntimeError, match="worker died"): + await _provision(client) + finally: + r.attach_aws_ids = original + + assert crashed, "the crash never happened; the test proves nothing" + assert len(client.create_kb_calls) == 1 + + anchor = _record(table) + assert anchor is not None, ( + "the KB_Record vanished, so the created knowledge base is now an " + "orphan nothing can find (Requirement 7.8)" + ) + assert anchor["provisioningState"] == r.PROVISIONING + assert "awsKbId" not in anchor + assert anchor["clientToken"], ( + "the anchor carries no clientToken, so a retry cannot be deduplicated " + "and would create a second knowledge base" + ) + + @pytest.mark.asyncio + async def test_the_retry_does_not_create_a_second_knowledge_base(self, table): + """The whole point: one knowledge base across a crash and a retry.""" + client = FakeBedrockAgent() + + original = r.attach_aws_ids + r.attach_aws_ids = lambda *a, **k: (_ for _ in ()).throw(RuntimeError("crash")) + try: + with pytest.raises(RuntimeError): + await _provision(client) + finally: + r.attach_aws_ids = original + + result = await _provision(client) # the retry + + assert len(client.create_kb_calls) == 2, "the retry did not re-issue the create" + assert client.distinct_knowledge_base_ids == {"KB00000001"}, ( + "the retry created a SECOND knowledge base: the persisted clientToken " + "was not reused, so AWS did not deduplicate" + ) + assert result.aws_kb_id == "KB00000001" + assert _record(table)["provisioningState"] == r.ACTIVE + + @pytest.mark.asyncio + async def test_a_crash_after_the_data_source_still_converges(self, table): + """Both AWS resources exist; only the final write was lost.""" + client = FakeBedrockAgent() + + original = r.attach_aws_ids + r.attach_aws_ids = lambda *a, **k: (_ for _ in ()).throw(RuntimeError("crash")) + try: + with pytest.raises(RuntimeError): + await _provision(client) + finally: + r.attach_aws_ids = original + + assert len(client.create_ds_calls) == 1 + await _provision(client) + + # One data source, because the data-source token is reused too. + assert len({call["clientToken"] for call in client.create_ds_calls}) == 1 + item = _record(table) + assert item["awsDataSourceId"] == "DS00000001" + + +# =========================================================================== +# 20.7 — synchronous boto3 calls run off the event loop +# =========================================================================== + + +class TestOffEventLoop: + @pytest.mark.asyncio + async def test_create_knowledge_base_runs_in_a_worker_thread(self, table): + """``CreateKnowledgeBase`` blocks for 47-124 s; on the loop thread that + would stall every other coroutine, including the health check.""" + client = FakeBedrockAgent() + loop_thread = threading.get_ident() + await _provision(client) + + assert client.thread_idents, "no AWS call was recorded" + assert all(ident != loop_thread for ident in client.thread_idents) + + @pytest.mark.asyncio + async def test_retrieve_runs_in_a_worker_thread(self): + runtime = FakeBedrockAgentRuntime([_result("doc-a", 0.9)]) + backend = mb.ManagedKbBackend(runtime_client=runtime, locator=_locator()) + loop_thread = threading.get_ident() + + await backend.search(APP_KB_ID, "query") + + assert runtime.thread_idents and runtime.thread_idents[0] != loop_thread + + @pytest.mark.asyncio + async def test_ingest_runs_in_a_worker_thread(self): + agent = FakeBedrockAgent() + backend = mb.ManagedKbBackend(agent_client=agent, locator=_locator()) + loop_thread = threading.get_ident() + + await backend.ingest(APP_KB_ID, DocumentSource("doc-a", "a.pdf", chunks=["x"])) + + assert agent.thread_idents and agent.thread_idents[0] != loop_thread + + +# =========================================================================== +# 8.3 — managed retrieval +# =========================================================================== + + +class TestRetrievalConfiguration: + def test_uses_managed_search_configuration(self): + """Requirement 11.1. ``vectorSearchConfiguration`` is a valid *member* of + the request shape, so it passes client-side validation and is then rejected + by the service for managed knowledge bases — every retrieval fails.""" + config = mb.retrieval_configuration() + assert "managedSearchConfiguration" in config + assert "vectorSearchConfiguration" not in config + + def test_number_of_results_defaults_to_the_parity_five(self): + assert mb.retrieval_configuration()["managedSearchConfiguration"][ + "numberOfResults" + ] == 5 + assert DEFAULT_TOP_K == 5 + + def test_reranking_is_managed_not_none(self): + """Requirement 11.2. Managed reranking separates the scores + (0.89/0.38/0.25/0.21/0.19 versus a nearly flat 1.00/0.84/0.78/0.77/0.77), + and that separation is what makes the 2,000-character cap defensible.""" + managed = mb.retrieval_configuration()["managedSearchConfiguration"] + assert managed["rerankingModelType"] == "MANAGED" + assert managed["rerankingModelType"] != "NONE" + + def test_no_hybrid_search_key_is_sent(self): + """Requirement 11.3 — not toggleable, and not attempted.""" + managed = mb.retrieval_configuration()["managedSearchConfiguration"] + assert not any("hybrid" in key.lower() or key == "overrideSearchType" for key in managed) + + def test_top_k_is_honoured(self): + assert mb.retrieval_configuration(3)["managedSearchConfiguration"][ + "numberOfResults" + ] == 3 + + @pytest.mark.parametrize("operator", ["equals", "in"]) + def test_exact_match_filters_are_permitted(self, operator): + config = mb.retrieval_configuration( + 5, {operator: {"key": "assistant_id", "value": APP_KB_ID}} + ) + assert operator in config["managedSearchConfiguration"]["filter"] + + @pytest.mark.parametrize("operator", ["startsWith", "stringContains", "notEquals"]) + def test_non_exact_filters_are_refused(self, operator): + """Requirement 11.5. A prefix filter isolating 'ast-1' also admits + 'ast-10', and the over-match looks like an ordinary result.""" + with pytest.raises(mb.UnsafeFilterOperator, match=operator): + mb.retrieval_configuration(5, {operator: {"key": "k", "value": "v"}}) + + def test_an_unsafe_operator_nested_in_a_compound_filter_is_refused(self): + """A compound filter is only as safe as its least safe leaf.""" + with pytest.raises(mb.UnsafeFilterOperator, match="startsWith"): + mb.retrieval_configuration( + 5, + { + "andAll": [ + {"equals": {"key": "a", "value": "1"}}, + {"orAll": [{"startsWith": {"key": "b", "value": "2"}}]}, + ] + }, + ) + + +class TestManagedSearch: + @pytest.mark.asyncio + async def test_score_is_relevance_and_is_not_converted(self): + """The inversion that raises nothing and just makes answers worse. + + ``Retrieve`` already returns higher-is-better, which is the protocol's + direction, so this adapter must pass the score through. Negating it here — + copying the legacy adapter's conversion — would reverse the ranking with + no error anywhere. + """ + runtime = FakeBedrockAgentRuntime( + [_result("doc-best", 0.9), _result("doc-mid", 0.4), _result("doc-worst", 0.1)] + ) + backend = mb.ManagedKbBackend(runtime_client=runtime, locator=_locator()) + + chunks = await backend.search(APP_KB_ID, "query") + + assert [c.relevance for c in chunks] == [0.9, 0.4, 0.1] + assert chunks[0].document_id == "doc-best" + assert chunks[0].relevance > chunks[-1].relevance, ( + "relevance is inverted: the best chunk now scores lowest" + ) + + @pytest.mark.asyncio + async def test_document_id_comes_from_the_custom_document_identifier(self): + """Requirement 9.4 — the join key the status filter needs.""" + runtime = FakeBedrockAgentRuntime([_result("doc-platform-id", 0.5)]) + backend = mb.ManagedKbBackend(runtime_client=runtime, locator=_locator()) + + chunks = await backend.search(APP_KB_ID, "query") + assert chunks[0].document_id == "doc-platform-id" + + @pytest.mark.asyncio + async def test_the_service_document_id_is_not_used_as_the_platform_id(self): + """``documentId`` is a GetDocumentContent handle, not our id. + + Using it would produce a chunk whose ``document_id`` looks plausible, + joins against no ``DOC#`` record, and is dropped by the fail-closed status + filter — results disappearing two layers from the cause. + """ + result = { + "content": {"text": "t"}, + "score": 0.5, + "documentId": "service-assigned-handle", + "metadata": {}, + } + runtime = FakeBedrockAgentRuntime([result]) + backend = mb.ManagedKbBackend(runtime_client=runtime, locator=_locator()) + + chunks = await backend.search(APP_KB_ID, "query") + assert chunks[0].document_id != "service-assigned-handle" + + @pytest.mark.asyncio + async def test_metadata_carries_text_and_document_id_for_consumers_above_the_seam(self): + runtime = FakeBedrockAgentRuntime([_result("doc-a", 0.5, text="the passage")]) + backend = mb.ManagedKbBackend(runtime_client=runtime, locator=_locator()) + + chunk = (await backend.search(APP_KB_ID, "query"))[0] + assert chunk.metadata["text"] == "the passage" + assert chunk.metadata["document_id"] == "doc-a" + + @pytest.mark.asyncio + async def test_a_missing_score_stays_none(self): + """Not 0.0: a fabricated score is indistinguishable from a real one, and + on this backend it would rank the chunk last while on legacy it ranks + first.""" + runtime = FakeBedrockAgentRuntime([{"content": {"text": "t"}, "metadata": {}}]) + backend = mb.ManagedKbBackend(runtime_client=runtime, locator=_locator()) + + assert (await backend.search(APP_KB_ID, "q"))[0].relevance is None + + @pytest.mark.asyncio + async def test_the_aws_kb_id_is_resolved_not_taken_from_the_caller(self): + runtime = FakeBedrockAgentRuntime([]) + backend = mb.ManagedKbBackend(runtime_client=runtime, locator=_locator("KB-RESOLVED")) + + await backend.search(APP_KB_ID, "query") + assert runtime.calls[0]["knowledgeBaseId"] == "KB-RESOLVED" + + @pytest.mark.asyncio + async def test_an_unprovisioned_knowledge_base_raises(self): + backend = mb.ManagedKbBackend( + runtime_client=FakeBedrockAgentRuntime([]), locator=lambda _ref: (None, None) + ) + with pytest.raises(mb.ManagedKbNotProvisioned): + await backend.search(APP_KB_ID, "query") + + @pytest.mark.asyncio + async def test_the_kb_record_supplies_the_ids_when_no_locator_is_given(self, table): + """The default path: resolve from the record, never from the caller.""" + r.create_provisioning( + ASSISTANT_ID, r.KbRecord(app_kb_id=APP_KB_ID, owner_user_id=OWNER) + ) + r.attach_aws_ids(ASSISTANT_ID, APP_KB_ID, AWS_KB_ID, AWS_DS_ID, "2026-01-01T00:00:00Z") + + runtime = FakeBedrockAgentRuntime([]) + backend = mb.ManagedKbBackend(runtime_client=runtime) + await backend.search(APP_KB_ID, "query") + + assert runtime.calls[0]["knowledgeBaseId"] == AWS_KB_ID + + +# =========================================================================== +# 8.4 — direct ingestion and deletion +# =========================================================================== + + +class TestBatching: + def test_the_limit_is_ten(self): + """Requirement 9.3. Server-enforced; the service model's document list + carries ``max: 10``. AWS's user guide says 25 and is wrong for managed + knowledge bases.""" + assert mb.MAX_DOCUMENTS_PER_CALL == 10 + + def test_batches_never_exceed_the_limit(self): + batches = mb.batched(list(range(25))) + assert [len(b) for b in batches] == [10, 10, 5] + assert all(len(b) <= 10 for b in batches) + + def test_a_batch_size_above_the_limit_is_refused(self): + """25 is the number the user guide gives. An 11-document call fails as a + whole, so the other ten documents are lost with the eleventh.""" + with pytest.raises(ValueError, match="exceeds the server-enforced maximum"): + mb.batched(list(range(25)), 25) + + def test_an_exact_multiple_produces_no_empty_batch(self): + assert [len(b) for b in mb.batched(list(range(20)))] == [10, 10] + + def test_no_items_means_no_batches(self): + assert mb.batched([]) == [] + + +class TestDocumentPayload: + def test_custom_document_identifier_is_the_platform_document_id(self): + """Requirement 9.4, verbatim — it is also the deletion handle and the + status-filter join key, so any transformation here needs undoing twice.""" + payload = mb.document_payload(DocumentSource("doc-42", "a.pdf", chunks=["x"])) + custom = payload["content"]["custom"] + assert custom["customDocumentIdentifier"] == {"id": "doc-42"} + + def test_content_data_source_type_is_custom(self): + payload = mb.document_payload(DocumentSource("doc-42", "a.pdf", chunks=["x"])) + assert payload["content"]["dataSourceType"] == "CUSTOM" + + def test_an_s3_backed_source_points_at_the_object(self): + payload = mb.document_payload( + DocumentSource("doc-42", "a.pdf", s3_key="assistants/ast/documents/doc-42/a.pdf"), + bucket="docs-bucket", + ) + custom = payload["content"]["custom"] + assert custom["sourceType"] == "S3_LOCATION" + assert custom["s3Location"]["uri"] == ( + "s3://docs-bucket/assistants/ast/documents/doc-42/a.pdf" + ) + + def test_a_chunked_source_is_sent_inline(self): + payload = mb.document_payload(DocumentSource("doc-42", "a.pdf", chunks=["a", "b"])) + custom = payload["content"]["custom"] + assert custom["sourceType"] == "IN_LINE" + assert custom["inlineContent"]["textContent"]["data"] == "a\n\nb" + + def test_an_empty_source_is_refused(self): + with pytest.raises(mb.ManagedKbError, match="nothing to ingest"): + mb.document_payload(DocumentSource("doc-42", "a.pdf")) + + def test_metadata_is_string_valued(self): + payload = mb.document_payload( + DocumentSource("doc-42", "a.pdf", chunks=["x"], metadata={"pages": 3}) + ) + attributes = {a["key"]: a["value"] for a in payload["metadata"]["inlineAttributes"]} + assert attributes["pages"] == {"type": "STRING", "stringValue": "3"} + assert attributes["document_id"]["stringValue"] == "doc-42" + + def test_no_chunk_index_appears_anywhere_in_the_payload(self): + """Requirement 9.6 — the ``{doc_id}#{chunk_index}`` scheme is retired on + this path, along with ``delete_vector_tail`` and the shrinkage stash.""" + payload = mb.document_payload(DocumentSource("doc-42", "a.pdf", chunks=["a", "b"])) + assert "#" not in repr(payload) + + +class TestInlineAttributeCap: + """Bedrock caps inlineAttributes at 50; caller metadata is unbounded. + + The failure this guards is disproportionate: metadata is per-document but the + API call is per-batch, so one over-decorated document fails the ingestion of + the nine innocent documents travelling with it. + """ + + def test_the_cap_is_the_literal_service_limit(self): + """Pinned to 50 as a literal, not to the constant, so moving the constant + cannot satisfy this. 50 is `{'min': 1, 'max': 50}` in the packaged service + model — a property of Bedrock, not a tuning knob.""" + assert mb.MAX_INLINE_ATTRIBUTES == 50 + + def test_metadata_is_truncated_to_the_cap(self): + payload = mb.document_payload( + DocumentSource( + "doc-42", "a.pdf", chunks=["x"], + metadata={f"key_{i:03d}": i for i in range(200)}, + ) + ) + attributes = payload["metadata"]["inlineAttributes"] + assert len(attributes) == mb.MAX_INLINE_ATTRIBUTES + + def test_reserved_keys_survive_truncation(self): + """``document_id`` is the status filter's join key, and it sorts after + plenty of plausible caller keys — so alphabetical truncation would drop the + one attribute the platform cannot do without, and every chunk of that + document would then be discarded as unverifiable. + """ + payload = mb.document_payload( + DocumentSource( + "doc-42", "a.pdf", chunks=["x"], + # All sort BEFORE "document_id", so a plain sorted() truncation + # would evict it. + metadata={f"aaa_{i:03d}": i for i in range(200)}, + ) + ) + keys = [a["key"] for a in payload["metadata"]["inlineAttributes"]] + assert "document_id" in keys, "truncation dropped the status-filter join key" + assert "filename" in keys + assert len(keys) == mb.MAX_INLINE_ATTRIBUTES + + def test_a_caller_cannot_override_the_reserved_keys(self): + """Otherwise a caller could point document_id at someone else's document.""" + payload = mb.document_payload( + DocumentSource( + "doc-42", "a.pdf", chunks=["x"], + metadata={"document_id": "doc-other"}, + ) + ) + attributes = {a["key"]: a["value"]["stringValue"] for a in payload["metadata"]["inlineAttributes"]} + assert attributes["document_id"] == "doc-42" + + def test_metadata_under_the_cap_is_untouched(self): + payload = mb.document_payload( + DocumentSource("doc-42", "a.pdf", chunks=["x"], metadata={"pages": 3}) + ) + keys = {a["key"] for a in payload["metadata"]["inlineAttributes"]} + assert keys == {"document_id", "filename", "pages"} + + +class TestIngestion: + @pytest.mark.asyncio + async def test_a_single_document_is_one_call(self): + agent = FakeBedrockAgent() + backend = mb.ManagedKbBackend(agent_client=agent, locator=_locator()) + + await backend.ingest(APP_KB_ID, DocumentSource("doc-a", "a.pdf", chunks=["x"])) + + assert len(agent.ingest_calls) == 1 + call = agent.ingest_calls[0] + assert call["knowledgeBaseId"] == AWS_KB_ID + assert call["dataSourceId"] == AWS_DS_ID + assert len(call["documents"]) == 1 + + @pytest.mark.asyncio + async def test_twenty_five_documents_are_split_into_batches_of_ten(self): + """Requirement 9.3. The interesting number is 25, because that is what the + user guide claims is allowed.""" + agent = FakeBedrockAgent() + backend = mb.ManagedKbBackend(agent_client=agent, locator=_locator()) + sources = [DocumentSource(f"doc-{i}", f"{i}.pdf", chunks=["x"]) for i in range(25)] + + await backend.ingest_documents(APP_KB_ID, sources) + + sizes = sorted(len(call["documents"]) for call in agent.ingest_calls) + assert sizes == [5, 10, 10] + assert all(len(call["documents"]) <= 10 for call in agent.ingest_calls) + + @pytest.mark.asyncio + async def test_every_document_is_sent_exactly_once(self): + agent = FakeBedrockAgent() + backend = mb.ManagedKbBackend(agent_client=agent, locator=_locator()) + sources = [DocumentSource(f"doc-{i}", f"{i}.pdf", chunks=["x"]) for i in range(23)] + + await backend.ingest_documents(APP_KB_ID, sources) + + sent = [ + document["content"]["custom"]["customDocumentIdentifier"]["id"] + for call in agent.ingest_calls + for document in call["documents"] + ] + assert sorted(sent) == sorted(f"doc-{i}" for i in range(23)) + assert len(sent) == len(set(sent)) + + @pytest.mark.asyncio + async def test_start_ingestion_job_is_never_called(self): + """Requirement 9.2 — 0.1 RPS account-wide, not adjustable: one document + every ten seconds for the entire account.""" + agent = FakeBedrockAgent() + backend = mb.ManagedKbBackend(agent_client=agent, locator=_locator()) + + await backend.ingest_documents( + APP_KB_ID, [DocumentSource(f"doc-{i}", "a.pdf", chunks=["x"]) for i in range(12)] + ) + + assert agent.start_ingestion_job_calls == [] + + def test_the_source_never_mentions_start_ingestion_job(self): + """A stronger guard than the call assertion: the code cannot call it.""" + source = Path(mb.__file__).read_text(encoding="utf-8") + assert "start_ingestion_job" not in source + + @pytest.mark.asyncio + async def test_no_client_token_is_sent_on_ingest(self): + """Idempotency comes from ``customDocumentIdentifier`` being 1:1, so a + re-ingest replaces. A content-blind token would look like extra safety and + instead silently swallow a legitimate re-upload.""" + agent = FakeBedrockAgent() + backend = mb.ManagedKbBackend(agent_client=agent, locator=_locator()) + + await backend.ingest(APP_KB_ID, DocumentSource("doc-a", "a.pdf", chunks=["x"])) + assert "clientToken" not in agent.ingest_calls[0] + + @pytest.mark.asyncio + async def test_concurrency_is_bounded_at_ten(self): + """Requirement 9.5. Ingest and delete share one account-wide budget of 10 + concurrent document operations.""" + agent = FakeBedrockAgent() + backend = mb.ManagedKbBackend(agent_client=agent, locator=_locator()) + # 150 documents ⇒ 15 batches, more than the bound permits at once. + sources = [DocumentSource(f"doc-{i}", "a.pdf", chunks=["x"]) for i in range(150)] + + await backend.ingest_documents(APP_KB_ID, sources) + + assert len(agent.ingest_calls) == 15 + # The literal 10, deliberately, not ``mb.MAX_CONCURRENT_DOCUMENT_OPERATIONS``: + # comparing observed concurrency against the constant that produced it is + # tautological, and would pass just as happily with the bound raised to 100. + assert agent.max_in_flight <= 10 + assert agent.max_in_flight > 1, "the batches ran serially; nothing was bounded" + + @pytest.mark.asyncio + async def test_ingesting_nothing_calls_nothing(self): + agent = FakeBedrockAgent() + backend = mb.ManagedKbBackend(agent_client=agent, locator=_locator()) + await backend.ingest_documents(APP_KB_ID, []) + assert agent.ingest_calls == [] + + @pytest.mark.asyncio + async def test_a_failed_batch_surfaces(self): + agent = FakeBedrockAgent() + + def _boom(**_kwargs): + raise RuntimeError("ingest rejected") + + agent.ingest_knowledge_base_documents = _boom + backend = mb.ManagedKbBackend(agent_client=agent, locator=_locator()) + + with pytest.raises(RuntimeError, match="ingest rejected"): + await backend.ingest(APP_KB_ID, DocumentSource("doc-a", "a.pdf", chunks=["x"])) + + @pytest.mark.asyncio + async def test_ingest_requires_a_data_source(self): + backend = mb.ManagedKbBackend( + agent_client=FakeBedrockAgent(), locator=lambda _ref: (AWS_KB_ID, None) + ) + with pytest.raises(mb.ManagedKbNotProvisioned, match="awsDataSourceId"): + await backend.ingest(APP_KB_ID, DocumentSource("doc-a", "a.pdf", chunks=["x"])) + + +class TestDeletion: + @pytest.mark.asyncio + async def test_delete_is_by_platform_document_id(self): + agent = FakeBedrockAgent() + backend = mb.ManagedKbBackend(agent_client=agent, locator=_locator()) + + await backend.delete_document(APP_KB_ID, "doc-42") + + identifiers = agent.delete_calls[0]["documentIdentifiers"] + assert identifiers == [{"dataSourceType": "CUSTOM", "custom": {"id": "doc-42"}}] + + @pytest.mark.asyncio + async def test_deletes_are_batched_at_ten(self): + agent = FakeBedrockAgent() + backend = mb.ManagedKbBackend(agent_client=agent, locator=_locator()) + + await backend.delete_documents(APP_KB_ID, [f"doc-{i}" for i in range(25)]) + + sizes = sorted(len(call["documentIdentifiers"]) for call in agent.delete_calls) + assert sizes == [5, 10, 10] + + @pytest.mark.asyncio + async def test_deletes_share_the_ingest_concurrency_budget(self): + agent = FakeBedrockAgent() + backend = mb.ManagedKbBackend(agent_client=agent, locator=_locator()) + + await backend.delete_documents(APP_KB_ID, [f"doc-{i}" for i in range(150)]) + + # Literal, for the same reason as the ingest case above. + assert agent.max_in_flight <= 10 + assert agent.max_in_flight > 1, "the batches ran serially; nothing was bounded" + + @pytest.mark.asyncio + async def test_deleting_nothing_calls_nothing(self): + agent = FakeBedrockAgent() + backend = mb.ManagedKbBackend(agent_client=agent, locator=_locator()) + await backend.delete_documents(APP_KB_ID, []) + await backend.delete_documents(APP_KB_ID, ["", None]) + assert agent.delete_calls == [] + + +# =========================================================================== +# Seam conformance and import weight +# =========================================================================== + + +class TestSeamConformance: + def test_the_managed_backend_satisfies_the_protocol(self): + assert isinstance(mb.ManagedKbBackend(), KnowledgeBaseBackend) + + @pytest.mark.asyncio + async def test_search_returns_protocol_chunks(self): + runtime = FakeBedrockAgentRuntime([_result("doc-a", 0.5)]) + backend = mb.ManagedKbBackend(runtime_client=runtime, locator=_locator()) + chunks = await backend.search(APP_KB_ID, "q") + assert all(isinstance(chunk, Chunk) for chunk in chunks) + + def test_constructing_the_backend_creates_no_aws_client(self): + """Lazy clients: importing and constructing must not touch credentials.""" + backend = mb.ManagedKbBackend() + assert backend._runtime_client is None + assert backend._agent_client is None + + +class TestImportWeight: + """The new modules obey the package's import boundary. + + Run in a fresh interpreter because by now the suite has imported boto3 + already, so an in-process check would pass regardless. + """ + + @staticmethod + def _loaded(module: str) -> List[str]: + backend_root = Path(__file__).resolve().parents[2] + program = ( + "import sys\n" + f"import {module}\n" + "forbidden = ('apis.shared.assistants', 'boto3')\n" + "print(','.join(sorted({n for n in sys.modules " + "if any(n == f or n.startswith(f + '.') for f in forbidden)})))\n" + ) + result = subprocess.run( + [sys.executable, "-c", program], + capture_output=True, + text=True, + cwd=str(backend_root), + env={"PYTHONPATH": str(backend_root / "src"), "PATH": "/usr/bin:/bin"}, + ) + assert result.returncode == 0, result.stderr + return [name for name in result.stdout.strip().split(",") if name] + + @pytest.mark.parametrize( + "module", + [ + "apis.shared.kb_backend.provisioning", + "apis.shared.kb_backend.managed_backend", + ], + ) + def test_import_pulls_in_neither_boto3_nor_assistants(self, module): + assert self._loaded(module) == [], ( + f"importing {module} loaded a forbidden module; keep boto3 and " + "anything from apis.shared.assistants function-local" + ) + + +class TestNoLiveAws: + """No test in this file constructs a Bedrock client. + + Both service clients are hand-rolled fakes and DynamoDB is moto. Checked by + walking this file's own AST rather than by substring, because a substring + search for ``boto3.client("bedrock`` finds the assertion that looks for it and + reports a failure that is only ever itself. + """ + + #: Anything that would hand back a real Bedrock client. + _FORBIDDEN_CALLS = frozenset( + {"bedrock_agent_client", "bedrock_agent_runtime_client"} + ) + + def test_this_file_never_creates_a_bedrock_client(self): + tree = ast.parse(Path(__file__).read_text(encoding="utf-8"), filename=__file__) + + offenders: List[str] = [] + for node in ast.walk(tree): + if not isinstance(node, ast.Call): + continue + func = node.func + if ( + isinstance(func, ast.Attribute) + and func.attr == "client" + and isinstance(func.value, ast.Name) + and func.value.id == "boto3" + and node.args + and isinstance(node.args[0], ast.Constant) + and str(node.args[0].value).startswith("bedrock") + ): + offenders.append(f"line {node.lineno}: boto3.client('bedrock...')") + elif isinstance(func, ast.Name) and func.id in self._FORBIDDEN_CALLS: + offenders.append(f"line {node.lineno}: {func.id}()") + elif isinstance(func, ast.Attribute) and func.attr in self._FORBIDDEN_CALLS: + offenders.append(f"line {node.lineno}: {func.attr}()") + + assert offenders == [], ( + "this test file would construct a real Bedrock client:\n" + + "\n".join(offenders) + ) + + def test_the_only_boto3_clients_are_dynamodb_under_moto(self): + """The DynamoDB access is real botocore, intercepted by moto.""" + tree = ast.parse(Path(__file__).read_text(encoding="utf-8"), filename=__file__) + services = { + str(node.args[0].value) + for node in ast.walk(tree) + if isinstance(node, ast.Call) + and isinstance(node.func, ast.Attribute) + and node.func.attr in ("client", "resource") + and isinstance(node.func.value, ast.Name) + and node.func.value.id == "boto3" + and node.args + and isinstance(node.args[0], ast.Constant) + } + assert services == {"dynamodb"}, f"unexpected AWS services used: {services}" + assert "mock_aws" in Path(__file__).read_text(encoding="utf-8") + + +def test_asyncio_gather_is_used_rather_than_ensure_future(): + """Requirement 10.8's spirit at this layer: no fire-and-forget orchestration. + + ``asyncio.ensure_future`` without an await loses the failure entirely — the + task is garbage collected and the exception is reported, at best, as an + unretrieved-exception warning nobody reads. + """ + source = Path(mb.__file__).read_text(encoding="utf-8") + assert "ensure_future" not in source + assert "asyncio.gather" in source + + +def test_the_module_constants_match_the_verified_api_limits(): + """One place to look when a doc page disagrees with reality.""" + assert p.CLIENT_TOKEN_MIN_LENGTH == 33 + assert p.CLIENT_TOKEN_MAX_LENGTH == 256 + assert mb.MAX_DOCUMENTS_PER_CALL == 10 + assert mb.MAX_CONCURRENT_DOCUMENT_OPERATIONS == 10 + assert p.DATA_DELETION_POLICY == "RETAIN" + assert p.IMAGE_EXTRACTION_STATUS == "ENABLED" + assert p.EMBEDDING_MODEL_ID == "amazon.titan-embed-text-v2:0" + assert p.EMBEDDING_DIMENSIONS == 1024 + assert mb.RERANKING_MODEL_TYPE == "MANAGED" diff --git a/backend/tests/shared/test_search_filtering.py b/backend/tests/shared/test_search_filtering.py index daa3a2951..a993788de 100644 --- a/backend/tests/shared/test_search_filtering.py +++ b/backend/tests/shared/test_search_filtering.py @@ -164,14 +164,25 @@ def test_filter_excludes_missing_records(mock_boto3_resource): # ----------------------------------------------------------------------- -# Requirement 3.4: DynamoDB error → graceful degradation (unfiltered) +# Requirement 5.1 (managed-kb-migration): DynamoDB error → fail closed +# Supersedes reliable-document-deletion Requirement 3.4, which said unfiltered. # ----------------------------------------------------------------------- +@patch("apis.shared.assistants.rag_service.emit_count") @patch("boto3.resource") @patch.dict("os.environ", ENV_PATCH) -def test_filter_graceful_degradation_on_dynamo_error(mock_boto3_resource): - """DynamoDB raises exception — return unfiltered results.""" +def test_filter_fails_closed_on_dynamo_error(mock_boto3_resource, mock_emit): + """DynamoDB raises — drop every chunk. + + INVERTED from the previous "return unfiltered" expectation by + managed-kb-migration Requirement 5.1, which supersedes + reliable-document-deletion Requirement 3.4. The old behaviour was deliberate; + what changed is evidence. 936 retrievals in a trailing 30-day window had chunks + removed by this filter, so a lookup failure would have served users content + they believe they deleted — worse than serving nothing, because the response + gives no hint the check was skipped. + """ mock_dynamo = MagicMock() mock_dynamo.Table.side_effect = Exception("DynamoDB unavailable") mock_boto3_resource.return_value = mock_dynamo @@ -185,10 +196,36 @@ def test_filter_graceful_degradation_on_dynamo_error(mock_boto3_resource): result = _filter_vectors_by_document_status(vectors, ASSISTANT_ID) - # Graceful degradation: all vectors returned unfiltered - assert len(result) == 2 - doc_ids = [v["metadata"]["document_id"] for v in result] - assert doc_ids == ["doc-a", "doc-b"] + assert result == [], "unconfirmable document status must not leak chunks" + # The degradation is reported, so an empty result here is distinguishable from + # the ordinary "corpus had no match" case. + mock_emit.assert_called_once() + + +# ----------------------------------------------------------------------- +# Requirement 5.2 (managed-kb-migration): missing table name → fail closed +# +# The second of the two former fail-open paths. It was previously untested; a +# test pinning the old behaviour was added first precisely so that inverting it +# here would be a deliberate, visible edit rather than a silent one. +# ----------------------------------------------------------------------- + + +@patch("apis.shared.assistants.rag_service.emit_count") +@patch("boto3.resource") +@patch.dict("os.environ", {}, clear=True) +def test_filter_fails_closed_when_table_name_unset(mock_boto3_resource, mock_emit): + """DYNAMODB_ASSISTANTS_TABLE_NAME unset — drop every chunk.""" + from apis.shared.assistants.rag_service import _filter_vectors_by_document_status + + vectors = [_make_vector("doc-a", 0), _make_vector("doc-b", 0)] + + result = _filter_vectors_by_document_status(vectors, ASSISTANT_ID) + + assert result == [], "no table means status is unconfirmable, so nothing may leak" + mock_emit.assert_called_once() + # No table to reach, so DynamoDB is never contacted. + mock_boto3_resource.assert_not_called() # ----------------------------------------------------------------------- diff --git a/backend/tests/supply_chain/test_kb_tag_contract.py b/backend/tests/supply_chain/test_kb_tag_contract.py new file mode 100644 index 000000000..12764da00 --- /dev/null +++ b/backend/tests/supply_chain/test_kb_tag_contract.py @@ -0,0 +1,232 @@ +""" +The managed knowledge base tag contract, asserted across three languages. + +Requirements 20.8, 20.11, 20.12, 14.1. + +Four components have to agree on the tags that identify this platform's knowledge +bases: the Python that writes them, the Python that filters on them, the CDK that +supplies their values, and the shell script that tears them down. Nothing in any +type system spans that set, and when nothing asserted it they drifted **three +ways** — different key names, different value sources, and a construct exporting +environment variables that no code read. + +The failure was silent in the worst way. The writer and the reconciler agreed with +each other, because both fell back to the same hardcoded defaults; so knowledge +bases were created, found, and reconciled normally. Only teardown disagreed, and +its symptom was matching nothing and reporting success — a leak of resources +billing at $5.00/GB-month, with no CloudFormation console to notice them in. + +So these tests read the TypeScript and the shell script as text and compare them +against the Python constants. Static, cross-language, and ugly — and the only +shape of test that can catch this class of defect. + +Feature: managed-kb-migration +Requirements: 20.8, 20.11, 20.12, 14.1 +""" + +import re +from pathlib import Path + +import pytest + +from apis.shared.kb_backend import tags as canonical + +REPO_ROOT = Path(__file__).resolve().parents[3] +CONSTRUCT = ( + REPO_ROOT / "infrastructure" / "lib" / "constructs" / "managed-kb" / "kb-migration-construct.ts" +) +TEARDOWN = REPO_ROOT / "scripts" / "teardown" / "managed-kb.sh" + + +def _ts_tag_keys() -> dict: + """Extract ``MANAGED_KB_TAG_KEYS`` from the construct. + + Parsed rather than duplicated, so this test cannot itself drift from the file + it is checking. + """ + body = CONSTRUCT.read_text(encoding="utf-8") + match = re.search( + r"export const MANAGED_KB_TAG_KEYS\s*=\s*\{(.*?)\}\s*as const", body, re.DOTALL + ) + if not match: + match = re.search(r"export const MANAGED_KB_TAG_KEYS\s*=\s*\{(.*?)^\};", body, re.DOTALL | re.MULTILINE) + assert match, "could not find MANAGED_KB_TAG_KEYS in the construct" + return dict(re.findall(r"(\w+):\s*'([^']+)'", match.group(1))) + + +class TestPythonIsInternallyConsistent: + """The writer and the filter must derive from one implementation, not mirror + each other. ``tombstones.project_tag_filter`` was documented as a *mirror* of + ``provisioning.build_tags`` and had drifted from it.""" + + def test_the_filter_is_a_subset_of_what_is_written(self, monkeypatch): + monkeypatch.setenv(canonical.ENV_TAG_VALUE_PREFIX, "bsu-agentcore") + monkeypatch.setenv(canonical.ENV_TAG_VALUE_ENVIRONMENT, "prod") + + written = canonical.build_tags("ast-1", "user-1") + expected = canonical.project_tag_filter() + + for key, value in expected.items(): + assert written[key] == value, f"filter expects {key}={value!r}, writer wrote {written.get(key)!r}" + + def test_provisioning_and_tombstones_resolve_identically(self, monkeypatch): + """Both delegate now; this asserts the delegation rather than the values.""" + from apis.shared.kb_backend import provisioning, tombstones + + monkeypatch.setenv(canonical.ENV_TAG_VALUE_PREFIX, "some-prefix") + monkeypatch.setenv(canonical.ENV_TAG_VALUE_ENVIRONMENT, "some-env") + + written = provisioning.build_tags("ast-1", "user-1") + expected = tombstones.project_tag_filter() + + assert tombstones.matches_project_tags(written, expected) is True + + def test_a_knowledge_base_from_another_environment_does_not_match(self, monkeypatch): + """Both scope keys are required. A knowledge base carrying our project + prefix but another environment's tag belongs to that environment.""" + from apis.shared.kb_backend import tombstones + + monkeypatch.setenv(canonical.ENV_TAG_VALUE_PREFIX, "shared-prefix") + monkeypatch.setenv(canonical.ENV_TAG_VALUE_ENVIRONMENT, "prod") + theirs = canonical.build_tags("ast-1", "user-1") + + monkeypatch.setenv(canonical.ENV_TAG_VALUE_ENVIRONMENT, "dev") + ours = tombstones.project_tag_filter() + + assert tombstones.matches_project_tags(theirs, ours) is False + + def test_untagged_never_matches(self): + from apis.shared.kb_backend import tombstones + + assert tombstones.matches_project_tags({}, {"ManagedKbPrefix": "p"}) is False + assert tombstones.matches_project_tags(None, {"ManagedKbPrefix": "p"}) is False + assert canonical.matches_project(None) is False + + +class TestValueResolution: + def test_the_construct_supplied_variable_wins(self, monkeypatch): + monkeypatch.setenv(canonical.ENV_TAG_VALUE_PREFIX, "from-construct") + monkeypatch.setenv("PROJECT_PREFIX", "from-fallback") + assert canonical.tag_prefix() == "from-construct" + + def test_the_fallback_is_used_when_the_primary_is_absent(self, monkeypatch): + """So a local run, and the App API which receives ``PROJECT_PREFIX`` but not + the tag variables, still resolve to the right scope.""" + monkeypatch.delenv(canonical.ENV_TAG_VALUE_PREFIX, raising=False) + monkeypatch.setenv("PROJECT_PREFIX", "from-fallback") + assert canonical.tag_prefix() == "from-fallback" + + def test_an_explicit_argument_beats_the_environment(self, monkeypatch): + monkeypatch.setenv(canonical.ENV_TAG_VALUE_PREFIX, "from-env") + assert canonical.tag_prefix("explicit") == "explicit" + + def test_the_writer_and_the_filter_share_one_fallback_chain(self, monkeypatch): + """An asymmetric fallback is how a writer and a reader disagree while both + look correct — which is exactly what happened.""" + monkeypatch.delenv(canonical.ENV_TAG_VALUE_PREFIX, raising=False) + monkeypatch.delenv(canonical.ENV_TAG_VALUE_ENVIRONMENT, raising=False) + monkeypatch.setenv("PROJECT_PREFIX", "p") + monkeypatch.setenv("ENVIRONMENT", "e") + + written = canonical.build_tags("ast-1", "user-1") + assert canonical.matches_project(written) is True + + def test_the_last_resort_default_is_warned_about(self, monkeypatch, caplog): + """Two deployments that both reach the default share a tag scope and will + each treat the other's knowledge bases as their own.""" + for name in ( + canonical.ENV_TAG_VALUE_PREFIX, + *canonical.FALLBACK_PREFIX_VARS, + ): + monkeypatch.delenv(name, raising=False) + + import logging + + with caplog.at_level(logging.WARNING): + assert canonical.tag_prefix() == canonical.DEFAULT_PREFIX + + assert any("falling back" in record.message for record in caplog.records) + + +class TestTheOwnerTagCarriesNoPii: + """Requirement 20.12. Unlike a database column, a tag cannot be scrubbed + retroactively from the audit trail it has already entered.""" + + @pytest.mark.parametrize("owner", ["a@b.test", "First.Last@example.edu"]) + def test_an_email_is_refused_rather_than_trimmed(self, owner): + with pytest.raises(ValueError, match="opaque"): + canonical.build_tags("ast-1", owner) + + def test_an_opaque_id_is_accepted(self): + written = canonical.build_tags("ast-1", "u-8f3c1e") + assert written[canonical.TAG_KEY_OWNER_USER_ID] == "u-8f3c1e" + + +class TestTheCdkConstructAgrees: + def test_the_construct_declares_the_same_key_names(self): + """The mismatch that started this. The construct declared + ``ManagedKbPrefix`` while the Python wrote ``prefix``, and nothing read the + construct's declaration, so neither side was wrong on its own.""" + ts_keys = _ts_tag_keys() + + assert ts_keys.get("PREFIX") == canonical.TAG_KEY_PREFIX + assert ts_keys.get("ENVIRONMENT") == canonical.TAG_KEY_ENVIRONMENT + assert ts_keys.get("APP_KB_ID") == canonical.TAG_KEY_APP_KB_ID + assert ts_keys.get("OWNER_USER_ID") == canonical.TAG_KEY_OWNER_USER_ID + + def test_the_construct_exports_the_variables_the_python_reads(self): + """Exporting a variable nothing reads is how the correct values sat unused + while the defaults were written into AWS.""" + body = CONSTRUCT.read_text(encoding="utf-8") + + assert f"{canonical.ENV_TAG_VALUE_PREFIX}:" in body, ( + f"the construct does not set {canonical.ENV_TAG_VALUE_PREFIX}, so the " + f"provisioning Lambda would fall back to a default" + ) + assert f"{canonical.ENV_TAG_VALUE_ENVIRONMENT}:" in body + + def test_the_construct_supplies_real_values_not_literals(self): + """`config.projectPrefix`, not a hardcoded string — the whole point is that + two deployments differ.""" + body = CONSTRUCT.read_text(encoding="utf-8") + line = next( + line for line in body.splitlines() if f"{canonical.ENV_TAG_VALUE_PREFIX}:" in line + ) + assert "config." in line, f"prefix tag value is not derived from config: {line.strip()}" + + +class TestTheTeardownScriptAgrees: + def test_the_script_matches_the_canonical_key_names(self): + """It previously matched ``prefix``/``env`` — keys nothing ever wrote.""" + body = TEARDOWN.read_text(encoding="utf-8") + + assert f'TAG_KEY_PREFIX="{canonical.TAG_KEY_PREFIX}"' in body + assert f'TAG_KEY_ENVIRONMENT="{canonical.TAG_KEY_ENVIRONMENT}"' in body + + def test_the_script_reads_the_same_variables_in_the_same_order(self): + """The variable *and* the order: a script that consulted the fallback first + would disagree with the Python on any host where both are set.""" + body = TEARDOWN.read_text(encoding="utf-8") + + prefix_line = next(line for line in body.splitlines() if line.startswith("PREFIX=")) + env_line = next(line for line in body.splitlines() if line.startswith("ENVIRONMENT=")) + + assert canonical.ENV_TAG_VALUE_PREFIX in prefix_line + assert canonical.ENV_TAG_VALUE_ENVIRONMENT in env_line + + # Primary before fallback. + assert prefix_line.index(canonical.ENV_TAG_VALUE_PREFIX) < prefix_line.index( + canonical.FALLBACK_PREFIX_VARS[0] + ) + + def test_the_script_no_longer_reads_the_cdk_only_variables_first(self): + """`CDK_PROJECT_PREFIX` is set at deploy time by `load-env.sh` and is *not* + set in a Lambda, so reading it first is how the teardown and the writer + ended up with different scopes.""" + body = TEARDOWN.read_text(encoding="utf-8") + prefix_line = next(line for line in body.splitlines() if line.startswith("PREFIX=")) + + assert not prefix_line.startswith('PREFIX="${CDK_PROJECT_PREFIX'), ( + "the teardown script reads CDK_PROJECT_PREFIX before the variable the " + "provisioning code writes from" + ) diff --git a/backend/tests/supply_chain/test_managed_kb_teardown.py b/backend/tests/supply_chain/test_managed_kb_teardown.py new file mode 100644 index 000000000..16adfffa3 --- /dev/null +++ b/backend/tests/supply_chain/test_managed_kb_teardown.py @@ -0,0 +1,545 @@ +""" +Teardown of runtime-created knowledge bases. + +Requirement 24.9. Managed knowledge bases are created by the provisioning saga at +runtime, so they are not CloudFormation children and ``delete-stack`` does not touch +them. Left behind they keep billing at $5.00/GB-month while being invisible in the +CloudFormation console — a leak with no console to notice it in. + +These tests run the **real script** with a stub ``aws`` on ``PATH``. Static analysis +of the file could confirm that a tag filter is written; it could not confirm that an +untagged knowledge base survives a run, which is the assertion that matters. The +stub is a small Python program that answers ``list-knowledge-bases``, +``list-tags-for-resource`` and ``delete-knowledge-base`` from a JSON fixture and +records every call, so the script's behaviour is observable without an AWS account. + +Two properties carry the risk: + +* **Only tagged resources are deleted.** Two environments share an account, so a + filter that matched by name prefix — or that treated unreadable tags as ours — + would delete another environment's corpus. +* **The service role outlives its knowledge bases.** Bedrock needs the role to + perform the delete, so the script must fail *before* any stack comes down when a + knowledge base is not confirmed absent. ``DELETE_UNSUCCESSFUL`` is a real terminal + state that retrying does not clear. + +Feature: managed-kb-migration +Requirements: 20.8, 13.4, 13.5, 24.9 +""" + +import json +import os +import subprocess +import textwrap +from pathlib import Path + +import pytest + +from apis.shared.kb_backend import tags as kb_tags + +REPO_ROOT = Path(__file__).resolve().parents[3] +TEARDOWN_DIR = REPO_ROOT / "scripts" / "teardown" +KB_SCRIPT = TEARDOWN_DIR / "managed-kb.sh" +DESTROY_SCRIPT = TEARDOWN_DIR / "destroy.sh" + +PREFIX = "testprefix" +ENVIRONMENT = "dev" +REGION = "us-west-2" +ACCOUNT = "123456789012" + +#: Every case here finishes in well under a second. Anything approaching this +#: limit is a runaway loop, and the harness turns it into a failed assertion +#: rather than a blocked suite. +SCRIPT_TIMEOUT_SECONDS = 30 + +#: The stub CLI. Reads its scripted world from ``KB_STUB_STATE`` and appends one +#: line per call to ``KB_STUB_CALLS``, so a test can assert on what was *not* +#: called as easily as on what was. +_STUB = textwrap.dedent( + ''' + #!/usr/bin/env python3 + import json, os, sys + + state_path = os.environ["KB_STUB_STATE"] + calls_path = os.environ["KB_STUB_CALLS"] + + with open(state_path) as fh: + state = json.load(fh) + + argv = sys.argv[1:] + with open(calls_path, "a") as fh: + fh.write(" ".join(argv) + "\\n") + + def arg(name, default=None): + return argv[argv.index(name) + 1] if name in argv else default + + command = argv[1] if len(argv) > 1 else "" + + if command == "list-knowledge-bases": + # Optional: start failing after N list calls, which is what an + # unreadable account looks like mid-teardown. + fail_after = os.environ.get("KB_STUB_FAIL_LIST_AFTER") + state["listCalls"] = state.get("listCalls", 0) + 1 + if fail_after and state["listCalls"] > int(fail_after): + with open(state_path, "w") as fh: + json.dump(state, fh) + sys.stderr.write("ThrottlingException\\n") + sys.exit(254) + + # Each call pops one "view" of the account, so a test can script a + # knowledge base disappearing after its delete. + views = state["views"] + view = views[0] if len(views) == 1 else views.pop(0) + with open(state_path, "w") as fh: + json.dump(state, fh) + print(json.dumps({"knowledgeBaseSummaries": view})) + sys.exit(0) + + if command == "list-tags-for-resource": + resource_arn = arg("--resource-arn", "") + kb_id = resource_arn.rsplit("/", 1)[-1] + if kb_id in state.get("tagErrors", []): + sys.stderr.write("AccessDeniedException\\n") + sys.exit(254) + print(json.dumps({"tags": state.get("tags", {}).get(kb_id, {})})) + sys.exit(0) + + if command == "delete-knowledge-base": + kb_id = arg("--knowledge-base-id", "") + if kb_id in state.get("deleteErrors", []): + sys.stderr.write("ValidationException\\n") + sys.exit(254) + print(json.dumps({"knowledgeBaseId": kb_id, "status": "DELETING"})) + sys.exit(0) + + sys.stderr.write("unexpected aws invocation: " + " ".join(argv) + "\\n") + sys.exit(2) + ''' +).strip() + + +def _kb(kb_id: str, status: str = "ACTIVE"): + return {"knowledgeBaseId": kb_id, "name": f"{PREFIX}-kb-{kb_id}", "status": status} + + +def _ours(): + """Built through the canonical helper rather than spelled out. + + A fixture that hardcodes tag keys keeps passing after the keys change under + it — which is exactly how the writer, the reconciler, the construct and this + script came to disagree three ways. + """ + return kb_tags.build_tags("ast-1", "u-1", PREFIX, ENVIRONMENT) + + +@pytest.fixture +def run_teardown(tmp_path): + """Run ``managed-kb.sh`` against a scripted account. Returns (result, calls).""" + + def _run(views, tags, tag_errors=None, delete_errors=None, extra_env=None, script=KB_SCRIPT): + bin_dir = tmp_path / "bin" + bin_dir.mkdir(exist_ok=True) + stub = bin_dir / "aws" + stub.write_text(_STUB) + stub.chmod(0o755) + + state_path = tmp_path / "state.json" + state_path.write_text( + json.dumps( + { + "views": views, + "tags": tags, + "tagErrors": tag_errors or [], + "deleteErrors": delete_errors or [], + } + ) + ) + calls_path = tmp_path / "calls.txt" + calls_path.write_text("") + + env = { + "PATH": f"{bin_dir}:{os.environ.get('PATH', '')}", + "HOME": str(tmp_path), + "KB_STUB_STATE": str(state_path), + "KB_STUB_CALLS": str(calls_path), + # Supplied directly so load-env.sh is not needed: the script only + # sources it when log_info is undefined, and these are the values it + # would have exported. + "CDK_AWS_REGION": REGION, + "CDK_AWS_ACCOUNT": ACCOUNT, + kb_tags.ENV_TAG_VALUE_PREFIX: PREFIX, + kb_tags.ENV_TAG_VALUE_ENVIRONMENT: ENVIRONMENT, + # Fast but never zero. An interval of 0 made the script's elapsed + # counter stop advancing, so its timeout was never reached and the + # loop span at full CPU issuing an `aws` call per iteration — this + # test ran for sixteen hours before it was killed. The script now + # clamps the interval to 1; the test no longer asks for 0 either, + # because a test should not rely on a clamp to terminate. + "KB_DELETE_POLL_TIMEOUT_SECONDS": "1", + "KB_DELETE_POLL_INTERVAL_SECONDS": "1", + } + env.update(extra_env or {}) + + # The script sources load-env.sh only if log_info is undefined, so define + # the log helpers up front and dot-source the script into that shell. + harness = textwrap.dedent( + f""" + log_info() {{ echo "[INFO] $1"; }} + log_warn() {{ echo "[WARN] $1"; }} + log_success() {{ echo "[SUCCESS] $1"; }} + GREEN=""; NC="" + source "{script}" + """ + ) + # A hard wall. Every case here should finish in well under a second, so a + # run that reaches this limit is a runaway loop and must fail the test + # rather than block the suite. `timeout` raises, which is the correct + # outcome: a hang is a defect, not a slow pass. + try: + result = subprocess.run( + ["bash", "-c", harness], + capture_output=True, + text=True, + env=env, + cwd=str(REPO_ROOT), + timeout=SCRIPT_TIMEOUT_SECONDS, + ) + except subprocess.TimeoutExpired as expired: + calls = [line for line in calls_path.read_text().splitlines() if line.strip()] + raise AssertionError( + f"managed-kb.sh did not terminate within {SCRIPT_TIMEOUT_SECONDS}s " + f"— this is the infinite-poll shape, not slowness. " + f"It made {len(calls)} AWS calls before being killed." + ) from expired + + calls = [line for line in calls_path.read_text().splitlines() if line.strip()] + return result, calls + + return _run + + +def _deleted_ids(calls): + return [ + line.split("--knowledge-base-id ")[1].split()[0] + for line in calls + if "delete-knowledge-base" in line + ] + + +# ── Only tagged resources ──────────────────────────────────────────────────── +class TestOnlyTaggedResourcesAreDeleted: + def test_ours_is_deleted_and_nothing_else_is(self, run_teardown): + """The whole property in one run: one of ours, one from another + environment, one from another project, and one with no tags at all.""" + views = [ + [_kb("KB-OURS"), _kb("KB-OTHERENV"), _kb("KB-OTHERPROJ"), _kb("KB-UNTAGGED")], + [_kb("KB-OTHERENV"), _kb("KB-OTHERPROJ"), _kb("KB-UNTAGGED")], + ] + tags = { + "KB-OURS": _ours(), + "KB-OTHERENV": kb_tags.build_tags("ast-2", "u-2", PREFIX, "prod"), + "KB-OTHERPROJ": kb_tags.build_tags("ast-3", "u-3", "someone-else", ENVIRONMENT), + "KB-UNTAGGED": {}, + } + + result, calls = run_teardown(views, tags) + + assert result.returncode == 0, result.stdout + result.stderr + assert _deleted_ids(calls) == ["KB-OURS"] + + def test_a_prefix_match_with_the_wrong_environment_survives(self, run_teardown): + """Two environments share an account. A knowledge base carrying our + project prefix but another environment's tag belongs to that + environment, and its name looks exactly like ours.""" + views = [[_kb("KB-PROD")], [_kb("KB-PROD")]] + tags = {"KB-PROD": kb_tags.build_tags("ast-p", "u-p", PREFIX, "prod")} + + result, calls = run_teardown(views, tags) + + assert result.returncode == 0 + assert _deleted_ids(calls) == [] + + def test_unreadable_tags_mean_not_ours(self, run_teardown): + """Unknown ownership is not ownership. Refusing to delete something we + cannot attribute is the only safe direction for a destructive pass.""" + views = [[_kb("KB-OPAQUE")], [_kb("KB-OPAQUE")]] + + result, calls = run_teardown(views, {}, tag_errors=["KB-OPAQUE"]) + + assert result.returncode == 0 + assert _deleted_ids(calls) == [] + + def test_an_empty_account_is_a_clean_no_op(self, run_teardown): + result, calls = run_teardown([[]], {}) + + assert result.returncode == 0 + assert _deleted_ids(calls) == [] + assert "No Managed Knowledge Bases to delete" in result.stdout + + def test_pagination_is_followed(self, run_teardown): + """A single-page walk would silently leave every knowledge base past the + first page billing.""" + views = [ + [_kb("KB-A"), _kb("KB-B")], + [_kb("KB-B")], + [], + ] + tags = {"KB-A": _ours(), "KB-B": _ours()} + + result, calls = run_teardown(views, tags) + + assert result.returncode == 0, result.stdout + result.stderr + assert sorted(_deleted_ids(calls)) == ["KB-A", "KB-B"] + + def test_the_scope_is_bounded(self, run_teardown): + """A tag filter that suddenly matched everything costs at most one + refusal rather than the account.""" + many = [_kb(f"KB-{i}") for i in range(5)] + tags = {f"KB-{i}": _ours() for i in range(5)} + + result, calls = run_teardown([many, many], tags, extra_env={"KB_TEARDOWN_MAX": "2"}) + + assert result.returncode == 1 + assert _deleted_ids(calls) == [] + assert "Refusing to delete" in result.stdout + + +# ── The role must outlive its knowledge bases ──────────────────────────────── +class TestTheRoleOutlivesItsKnowledgeBases: + def test_a_knowledge_base_that_never_disappears_fails_the_run(self, run_teardown): + """Requirements 13.4, 13.5. "Delete call accepted" is not "resource + gone": deletion is asynchronous and took 2-6 minutes when measured. A + non-zero exit is what stops the caller deleting the service role out from + under a knowledge base that still needs it.""" + views = [[_kb("KB-STUCK")]] # one view, reused: it never disappears + tags = {"KB-STUCK": _ours()} + + result, calls = run_teardown(views, tags) + + assert result.returncode == 1 + assert _deleted_ids(calls) == ["KB-STUCK"] + assert "not confirmed absent" in result.stdout + assert "must" in result.stdout and "outlive" in result.stdout + + def test_absence_is_confirmed_by_polling_the_list(self, run_teardown): + """Not by the delete call's own return. The script must ask the account.""" + views = [[_kb("KB-GONE")], []] + tags = {"KB-GONE": _ours()} + + result, calls = run_teardown(views, tags) + + assert result.returncode == 0 + assert "confirmed absent" in result.stdout + # At least two list calls: one to find it, one to confirm it went. + assert sum(1 for line in calls if "list-knowledge-bases" in line) >= 2 + + def test_delete_unsuccessful_is_surfaced_with_its_remedy(self, run_teardown): + """A real terminal state — the dev account has held one since + 2025-11-24 — that retrying does not clear. The message must name the + remedy rather than time out generically.""" + views = [[_kb("KB-BROKEN", status="DELETE_UNSUCCESSFUL")]] + tags = {"KB-BROKEN": _ours()} + + result, calls = run_teardown(views, tags) + + assert result.returncode == 1 + assert _deleted_ids(calls) == [], "retried a delete that cannot succeed" + assert "DELETE_UNSUCCESSFUL" in result.stdout + assert "dataDeletionPolicy" in result.stdout + + def test_a_failed_delete_call_does_not_report_success(self, run_teardown): + views = [[_kb("KB-REFUSED")], [_kb("KB-REFUSED")]] + tags = {"KB-REFUSED": _ours()} + + result, _ = run_teardown(views, tags, delete_errors=["KB-REFUSED"]) + + assert result.returncode == 1 + + def test_a_failed_discovery_listing_refuses_to_continue(self, run_teardown): + """The very first listing fails, so the script never learns what exists. + + An unreadable account is indistinguishable from an empty one. Treating it + as empty would report a clean teardown having deleted nothing, while the + knowledge bases kept billing — and `done < <(list_knowledge_bases)` hides + the lister's exit status entirely, which is how the first version of this + script behaved. + """ + result, calls = run_teardown( + [[_kb("KB-A")]], + {"KB-A": _ours()}, + extra_env={"KB_STUB_FAIL_LIST_AFTER": "0"}, + ) + + assert result.returncode == 1 + assert _deleted_ids(calls) == [] + assert "Refusing to continue" in result.stdout + assert "No Managed Knowledge Bases to delete" not in result.stdout, ( + "an unreadable account was reported as an empty one" + ) + + def test_an_unlistable_account_is_not_read_as_confirmed_absence(self, run_teardown): + """Fail safe on the *confirmation* path. The presence check runs out of + scripted views and the stub then errors, which is what an unreadable + account looks like mid-teardown. Answering "absent" there would report a + successful teardown and let the caller delete the service role out from + under a live knowledge base. + """ + # Two views: one to discover and delete, then the stub runs dry and fails. + views = [[_kb("KB-MAYBE")]] + tags = {"KB-MAYBE": _ours()} + + result, _ = run_teardown( + views, + tags, + extra_env={"KB_STUB_FAIL_LIST_AFTER": "2"}, + ) + + assert result.returncode == 1 + # The success line specifically, not the substring — "were not confirmed + # absent" contains it and the first version of this assertion matched that. + assert "KB-MAYBE confirmed absent" not in result.stdout + assert "not confirmed absent" in result.stdout + + +# ── Ordering inside destroy.sh ─────────────────────────────────────────────── +class TestTeardownOrdering: + """Asserted statically, because running ``destroy.sh`` would delete stacks. + + The property is textual position: knowledge bases must be handled before the + first stack delete, and the failure must abort rather than continue. + """ + + def test_knowledge_bases_are_torn_down_before_any_stack(self): + """Line-based, and comparing against the first *invocation*. + + Two traps, both of which this test originally fell into. Searching the raw + text for ``aws cloudformation delete-stack`` matches the header comment + that explains why the script uses that command instead of ``cdk destroy`` + — a match hundreds of lines above any code. And ``destroy_stack`` appears + first as a function *definition*, which executes nothing. + + So: skip comments, and find where ``destroy_stack`` is *called*. + """ + lines = DESTROY_SCRIPT.read_text().splitlines() + + def first_line_matching(predicate) -> int: + for index, line in enumerate(lines): + stripped = line.strip() + if not stripped or stripped.startswith("#"): + continue + if predicate(stripped): + return index + return -1 + + kb_phase = first_line_matching(lambda s: "managed-kb.sh" in s) + first_call = first_line_matching( + lambda s: (s.startswith("destroy_stack") or s.startswith("if destroy_stack")) + and not s.startswith("destroy_stack()") + ) + + assert kb_phase != -1, "destroy.sh never invokes managed-kb.sh" + assert first_call != -1, "no destroy_stack invocation found; has the script changed shape?" + assert kb_phase < first_call, ( + f"destroy.sh calls destroy_stack at line {first_call + 1} before tearing " + f"down knowledge bases at line {kb_phase + 1}; the Bedrock service role " + f"lives in PlatformStack and its knowledge bases cannot be deleted " + f"without it" + ) + + def test_a_knowledge_base_failure_aborts_the_teardown(self): + """Rather than being logged and stepped over. Leaving paid resources + behind is the one outcome worse than a teardown that stops and says why. + + Asserted on the *shape* of the invocation, not on the presence of the + strings ``managed-kb.sh`` and ``exit 1`` somewhere in the block — both + survive a mutation that changes the condition to ``if false`` and adds + ``|| true``, which is exactly the regression this is meant to catch. + """ + lines = [line.strip() for line in DESTROY_SCRIPT.read_text().splitlines()] + invocations = [line for line in lines if "managed-kb.sh" in line and not line.startswith("#")] + + assert invocations, "destroy.sh never invokes managed-kb.sh" + assert len(invocations) == 1, f"invoked more than once: {invocations}" + + invocation = invocations[0] + assert invocation.startswith("if ! bash"), ( + f"the knowledge base teardown is not invoked in a failing condition: " + f"{invocation!r}. A bare call, or one ending in `|| true`, means a " + f"teardown that could not delete a knowledge base proceeds to delete " + f"the service role it needs." + ) + assert "|| true" not in invocation + assert "exit 1" in DESTROY_SCRIPT.read_text()[DESTROY_SCRIPT.read_text().index(invocation):] + + def test_the_kb_script_is_executable_and_strict(self): + body = KB_SCRIPT.read_text() + assert body.startswith("#!/bin/bash") + assert "set -euo pipefail" in body + + +class TestTheWaitLoopCannotRunAway: + """Regression tests for a script that once ran for sixteen hours. + + The wait loop advanced a counter by the poll interval and stopped when the + counter reached the timeout. At an interval of ``0`` the counter never + advanced, so the timeout was never reached — an unbounded loop issuing an + ``aws`` call and a ``python3`` parse per iteration, at full CPU. It was a test + that set the interval to 0 to run quickly that found it, by not finishing. + """ + + def test_a_zero_poll_interval_still_terminates(self, run_teardown): + """The exact configuration that hung. It must now be clamped.""" + views = [[_kb("KB-STUCK")]] # never disappears, so the loop runs to its bound + tags = {"KB-STUCK": _ours()} + + result, _ = run_teardown( + views, + tags, + extra_env={ + "KB_DELETE_POLL_INTERVAL_SECONDS": "0", + "KB_DELETE_POLL_TIMEOUT_SECONDS": "1", + }, + ) + + assert result.returncode == 1 + assert "not confirmed deleted" in result.stdout + + def test_a_non_numeric_tunable_does_not_abort_the_teardown(self, run_teardown): + """``[ abc -lt 1 ]`` fails under ``set -e``. A typo in a tunable must not + take down a teardown that is otherwise fine. + + Asserted on **stderr** as well as the exit code, because the numeric-check + `|| fallback` clamps the value either way — so termination alone passes with + the type guard removed. What the guard actually buys is that bash never + evaluates ``[ ten -ge 1 ]``, and that comparison announces itself with + "integer expression expected". + """ + views = [[_kb("KB-GONE")], []] + tags = {"KB-GONE": _ours()} + + result, _ = run_teardown( + views, + tags, + extra_env={"KB_DELETE_POLL_INTERVAL_SECONDS": "ten"}, + ) + + assert result.returncode == 0 + assert "confirmed absent" in result.stdout + assert "integer expression expected" not in result.stderr, ( + "a non-numeric tunable reached a numeric comparison; the type guard is " + f"not catching it. stderr: {result.stderr}" + ) + + def test_the_loop_is_bounded_by_attempts_not_only_by_arithmetic(self): + """Belt and braces, asserted in the source: termination must not depend + solely on a counter that a later edit could stop advancing.""" + body = KB_SCRIPT.read_text() + assert "ATTEMPTS_LEFT" in body + assert 'while [ "${ATTEMPTS_LEFT}" -gt 0 ]' in body + + def test_the_polling_bound_is_at_least_the_measured_deletion_time(self): + """Deletion took 2-6 minutes when measured, so the default tolerance must + comfortably exceed it. Pinned as a literal: this is a property of AWS's + behaviour, not a knob.""" + body = KB_SCRIPT.read_text() + assert "KB_DELETE_POLL_TIMEOUT_SECONDS:-480}" in body diff --git a/docs/specs/bedrock-managed-kb-evaluation.md b/docs/specs/bedrock-managed-kb-evaluation.md index 1eea2b3ab..b83c97bfc 100644 --- a/docs/specs/bedrock-managed-kb-evaluation.md +++ b/docs/specs/bedrock-managed-kb-evaluation.md @@ -1,15 +1,26 @@ # Bedrock Managed Knowledge Base — evaluation and target topology -**Status:** Evaluation complete, target topology proposed. Benchmark and -implementation-readiness gates defined below; no product implementation. +**Status:** Evaluation complete. **§13 benchmark executed 2026-08-13/14 — the +§13.4 decision gate is CLEARED (4 of 5 conditions; see §13.5).** Recommendation is +to proceed to a product vertical slice, subject to the four requirements in §13.5. +No product implementation yet. **Question asked:** Can Bedrock Managed Knowledge Base replace our custom RAG pipeline? Given usage, pricing and quotas, should a user have *multiple* KBs per agent, or one KB filtered by agent id? And is it a fit for impromptu document uploads in a conversation? **Evidence:** AWS Price List API query + a live 3-KB probe in dev-ai -(490617140655, us-west-2), both 2026-08-11. Every number below is either quoted -from official AWS docs, extracted from the Price List API, or measured. Blog and -unverified claims are marked as such. +(490617140655, us-west-2), both 2026-08-11 — plus the **§13 benchmark harness** +(2026-08-13/14, same account): 9 questions × 3 backends with every variable held +constant, 8 capability probes, a context-cap sweep, and live quota probes. Every +number below is either quoted from official AWS docs, extracted from the Price List +API, or measured. Blog and unverified claims are marked as such. Where the original +probe and the benchmark disagree, §5.1 and §11 carry the revised figures. +**Production baseline (2026-08-14):** §3.2/§3.3 and §7.4 add a measurement pass +against **prod** (897729136999, us-west-2) — S3 bucket metrics, CloudWatch Logs +Insights over the live AgentCore runtime, and the `ROLLUP#MONTHLY` / +`rag-assistants` tables. These replace the earlier hypothetical corpus and +retrieval-volume figures, and are the only numbers here drawn from prod rather +than dev-ai. **Supersedes:** the KB-per-assistant decision and the `CUSTOM`/`WEB` data-source shapes in the earlier `bedrock-managed-knowledge-base` draft (that draft reasoned from *classic* KB quotas, which no longer bind). @@ -105,13 +116,96 @@ accounts are us-west-2. | Embed | Titan v2 per chunk | included | | Rerank | **none today** | included | -Retrieval is noise (~$17.50 at 5 RAG turns × 3,500 sessions; ~$52 with 3-way -fan-out) against $0.02–0.05 turn costs. **Storage is the whole story: ~33× more -expensive per GB.** 10 GB corpus = $50/mo; 50 GB = $250/mo. Under ~10 GB the ops -win dominates; above that it needs a deliberate decision. +Two line items the first draft of this table omitted, both from the same +pricing page: + +| | Managed KB | +|---|---| +| Agentic retrieval | **$4.00/1k `AgenticRetrieve` + $1.00/1k underlying `Retrieve`** ≈ $0.006/turn at 2 underlying calls | +| Gateway invocation | **not included** — standard AgentCore Gateway tool-invocation charges apply if the KB is reached through Gateway | + +The $5.00/GB meters **raw source bytes, not post-parse text.** AWS's own example +prices 50 GB as "approximately 100,000 documents including PDFs, presentations, +Word files, and images" — 500 KB/document, which is a file size, not an extracted-text +size. For a PDF-heavy corpus this is the *less* favourable of the two readings, so +the number here is already conservative. Our 20 MB per-file ingest cap +(`documents/ingestion/handler.py`, `MAX_FILE_SIZE_MB`) bounds the worst single +document at **$0.10/mo**. ⇒ **The thing to garbage-collect is gigabytes, not knowledge bases** (§7). +### 3.2 Measured production baseline (2026-08-14) + +§3.1's original 10 GB / 50 GB illustrations were hypotheticals. Replaced with +measurement. + +| Metric | Measured | Source | +|---|---|---| +| Corpus, all versions + icons | 525 MB / 3,110 objects | `boisestateai-v2-rag-documents-*` CloudWatch `BucketSizeBytes` | +| Corpus growth | ~1.75× in the 12 days to 08/13 | same metric, daily series | +| RAG retrievals, trailing 30 d | **3,886** | Logs Insights, runtime `h4MSyY7YSh` | +| Total chat requests, same window | ~15,150 | `ROLLUP#MONTHLY`, Jul + Aug-to-date | +| **RAG attach rate** | **~26% of turns** | derived | +| Cost per turn | $0.043–0.045 | `ROLLUP#MONTHLY` | +| Active users, peak month | 469 (2026-07) | `ROLLUP#MONTHLY` | + +**Cost today, like-for-like:** 0.53 GB × $5.00 = $2.65 storage + 3,886 × $0.001 = +$3.89 retrieval ≈ **$6.54/mo**, against ~$0.09/mo on the current stack. A ~$6.50 +delta. With agentic retrieval instead, ≈ $26/mo. + +So the original framing — "storage is the whole story, and above 10 GB it needs a +deliberate decision" — is **wrong at present scale in both halves.** Nothing here +is a decision. But it is wrong for a reason worth stating: we are at **~1.5% of +the 30,000-user target**, so every absolute number above is a 1.5% number. + +Scaling the measured per-active-user figures (1.13 MB and 8.3 retrievals/user-month): + +| Scale | Users | Storage | Retrieval (std) | Retrieval (agentic) | Total/mo | +|---|---|---|---|---|---| +| Today | ~470 | $2.65 | $3.89 | $23 | **$6.50 – $26** | +| 10× | ~4,700 | $27 | $39 | $233 | **$66 – $260** | +| Full adoption | 30,000 | **$169** | **$249** | **$1,494** | **$418 – $1,663** | + +Retrieval scales with *users*; storage scales with *corpus*. At full adoption +retrieval overtakes storage, and the standard-vs-agentic choice becomes a +four-figure monthly line item. Frame it in per-turn terms: standard `Retrieve` +adds **2.3%** to a $0.044 turn; agentic adds **~14%**. + +**These figures are an expected-value trajectory, not a ceiling — do not read them +as a substitute for §13.5's condition 1.** The $169 above is what 30,000 users cost +*if today's measured 1.13 MB/user behaviour holds*. The existing 1 GB-per-user +allowance **permits** 30,000 GB, i.e. **$150,000/month**. Both numbers are correct +and they answer different questions: this table forecasts likely spend, §13.5 +bounds the policy exposure. The per-owner byte cap is required precisely because +nothing but that cap separates the two. + +⇒ **Migrate on operational grounds now; adopt agentic retrieval as a separate, +later decision justified on retrieval quality** (see §6.4 for its quota wall). + +### 3.3 No seasonality baseline exists yet + +`ROLLUP#MONTHLY` begins 2026-03. Every month on record is adoption ramp, not +season: + +| Month | Requests | Active users | +|---|---|---| +| 2026-03 | 366 | 19 | +| 2026-04 | 1,855 | 135 | +| 2026-05 | 1,820 | 89 | +| 2026-06 | 2,455 | 134 | +| 2026-07 | 11,941 | 469 | +| 2026-08 (14 d) | 9,178 → ~20,300 pace | 350 | + +~55× in five months, with **no summer trough** — August is pacing 1.7× July. Any +"summer is quiet, multiply by N for fall" adjustment is unsupported by data. First +real seasonal signal arrives October 2026; re-baseline the attach rate and +per-user retrieval figures then. + +*(Aside, out of scope: 2026-04 and 2026-05 record zero cache read/write tokens +while 03 and 06+ record heavy use. Prompt caching appears to have been off for two +months. Cache savings currently carry ~46% of token cost, so this is worth a +separate look.)* + --- ## 4. Corrected API shapes @@ -196,8 +290,65 @@ Corrections this forces: - `Retrieve` returned `score: 1.0` on an exact hit ⇒ **relevance, higher = better**. Legacy code assumes cosine *distance*. -**Untested, and it matters:** whether a KB goes cold again after idleness. If it -does, the dormant tier in §7 pays the warm-up on wake. +### 5.1 Revised measurements from the §13 benchmark (2026-08-13/14) + +The §13 harness re-measured everything above across **7 knowledge base creations** +and three document classes. Where the two disagree, prefer these numbers — the +sample is larger and the documents are realistic rather than a few hundred bytes. + +| step | revised measurement | +|---|---| +| `CreateKnowledgeBase` → ACTIVE | **47–124 s** (n=7, median ≈73 s) | +| cold 1st ingest, 1.4 KiB markdown | **68.2–68.3 s** (n=3, spread <0.15 s) | +| **warm** ingest, small markdown, same KB | **2.5 s** → retrievable 3.2 s | +| INDEXED → actually retrievable | **0.75–1.03 s** | +| ingest, 50 KiB native PDF (warm KB) | **68–264 s** | +| ingest, 260 KiB scanned PDF (warm KB) | **37–58 s** | +| `Retrieve` steady state | **p50 662–695 ms, p95 762–800 ms** | +| current pipeline `Retrieve` for comparison | **p50 257 ms, p95 262 ms** | + +Refinements to the original conclusions: + +- **Creation is more variable than 84–97 s** — the observed range is 47 s to 124 s, + a 2.6× spread. Size provisioning timeouts for the tail, not the median. +- **The per-KB warm-up is real and remarkably constant**: 68.296 s, 68.232 s and + 68.334 s on three independent knowledge bases for the same small file. That is a + fixed cost of the *knowledge base*, not of the document. +- **But real documents do add parsing time on top**, so "subsequent ingests are + ~4–6 s" is too optimistic for anything substantial: the same 50 KiB PDF took + 68 s, 89 s, 99 s and 264 s across four runs. Managed ingestion has a **long + tail** and must be treated as background work with generous timeouts. +- **INDEXED → retrievable is ~1 s, not 3.4 s.** Still a distinct event that must be + measured separately, but smaller than first recorded. +- **Added time-to-first-token is +405 ms at p50 and +538 ms at p95** versus the + current pipeline. (One current-pipeline sample of 5.7 s was the first query of a + run paying a cold Bedrock embedding call; it is excluded from the percentiles and + reported separately, rather than being presented as steady state.) + +⚠️ **Two API shapes in §4 are wrong for retrieval and must be corrected:** + +1. **`vectorSearchConfiguration` is rejected by a MANAGED knowledge base.** + ``` + ValidationException: Incompatible configuration: vectorSearchConfiguration is + not supported for managed knowledge bases. Use managedSearchConfiguration instead. + ``` + `retrievalConfiguration` has two mutually exclusive branches. Managed uses + `managedSearchConfiguration`, whose members are `numberOfResults`, + `rerankingModelType` (`CUSTOM`/`MANAGED`/`NONE`), `rerankingConfiguration`, and + `filter`. +2. **`clientToken` has a 33-character minimum** (max 256, pattern + `[a-zA-Z0-9](-*[a-zA-Z0-9]){0,256}`). A natural `{id}-{variant}-kb` token is + 31 characters and fails client-side validation. Build tokens, don't interpolate + them. + +Also: creating a knowledge base with `embeddingModelType: CUSTOM` can fail with +*"Unable to verify the specified embedding model"* purely from **IAM eventual +consistency** — the model was confirmed `ACTIVE` and directly invokable at the +time. Treat that message as retryable, or lazy per-KB provisioning (§7.1) will +fail intermittently while pointing at the wrong cause. + +**Answered:** whether a KB goes cold again after idleness — see §11 question 1. + --- @@ -268,6 +419,183 @@ offload of native blocks, rehydrated on demand, never on the introducing turn. the agent's knowledge base" action. Non-blocking, ~5–9 s into a warm KB, and the user has asked for persistence. +**This decision does not prevent an attachment and a knowledge base from being +used together — they already compose in a single turn.** Worth stating explicitly, +because "attachments stay out of Managed KB" reads as though the combination is +excluded, and it is not. The two paths are independent and both reach the model: + +- the attachment arrives as an inline `document` block, **whole and unchunked** + (`agents/main_agent/multimodal/prompt_builder.py`); +- knowledge-base material arrives as retrieved chunks prepended by + `augment_prompt_with_context`; +- `inference_api/chat/routes.py:2234` merges them (`final_message = + augmented_message`, then attachment guidance), and the comment at :2215 names + both mechanisms operating on the same prompt. + +So "here is my essay, compare it against the exemplars in your knowledge base" +works today with no change. **Ingesting the attachment into the KB would be +strictly worse for that shape of task,** for the reason in objection 3 above — +chunking destroys the whole-document structure being compared — and for a second +reason that is not merely a quality concern: a chat attachment ingested into an +agent's shared KB becomes **retrievable by every other user of that agent.** For a +class-assignment agent, one student's essay leaking into another's retrieval is an +incident, not a papercut. Keeping attachments session-scoped prevents it by +default. + +⚠️ **What *does* limit this shape of task is the 2,000-character context cap, and +§13.6's result explicitly does not cover it.** The attachment arrives in full while +the corpus arrives as ~2,000 characters of fragments — measured at 1,987 of 2,000 +characters and **2 of 5 chunks** on every managed question. §13.6's finding that the +cap costs nothing holds only for single-fact lookups; compare-and-contrast is +precisely the multi-chunk synthesis case it flags as untested. **Use this scenario +as the question set for that experiment** — it is scorable in a way "summarise" is +not: does the answer cite specific exemplars, or generalise vaguely? Sizing from +§13.6: 8,000 characters is where all five chunks fit, at ~966 extra input tokens +per turn. + +A related gap: retrieval returns five *fragments*, never "exemplar #3 in full", so +"compare my structure against a strong essay's structure" may have nothing +structural to compare against. Two candidate mechanisms, both unproven here: +`bedrock:GetDocumentContent` (§11.1, §14.0) to fetch a whole KB document after +using retrieval only to *identify* it — API shape, size limits and cost +unverified; or agentic retrieval, which per §13.6 "does its own retrieval and is +**not** subject to this cap at all", making this class of request a natural trigger +for the §6.5 escalation rather than a configuration flag. + +--- + +### 6.4 Quota headroom — verified against measured scale + +Quotas re-read from the [managed KB quota page](https://docs.aws.amazon.com/bedrock/latest/userguide/kb-managed-quotas.html) +on 2026-08-14. Projections use §3.2's measured per-user figures and a 5× +peak-to-average factor over ~9,600 business-hour minutes/month. + +| Quota | Default | Adjustable | Today | At 30,000 users | Verdict | +|---|---|---|---|---|---| +| Managed KBs / account / Region | 10,000 | Yes | *unmeasured* | scales with KB-creating users | **measure** | +| Data sources / KB | 200 | **No** | ~4 | ~4 | safe — see below | +| Concurrent ingestion jobs / KB | 50 | **No** | 1 | few | safe | +| Raw data storage / KB | 10 TB | **No** | <1 GB | <100 GB | safe | +| Query input chars / `Retrieve` | 10,000 | **No** | unbounded | unbounded | **fix required** | +| `Retrieve` RPM / KB | 600 (25 RPS burst) | Yes | 0.09 | ~26 avg on a hot shared KB | safe; pre-request for hot KBs | +| `AgenticRetrieveStream` RPM / **account** | **60** | Yes | 0.09 | ~130 peak | **hard wall ~30× current** | + +**The one blocker: query length.** `search_assistant_knowledgebase` +(`apis/shared/embeddings/bedrock_embeddings.py`) passes `input_data.message` +straight through, with an inline comment stating *"short string, no token +validation needed."* Titan v2 tolerates ~32,000 characters, so nothing fails +today — the §3.2 zero-result accounting closes exactly (1,743 + 287 = 2,030), +which proves no query is currently failing to embed. Managed KB's cap is +**10,000 characters and non-adjustable**, roughly 3× tighter. A pasted essay +would hard-fail the `Retrieve` call. Truncate or summarise the query at the §10.1 +seam before promoting any traffic. + +**The one ceiling to plan around: agentic RPM is per *account*, not per KB.** 60 +RPM is a single contention point shared by every agent, and it is strikingly low +next to the per-KB `Retrieve` allowance of 600 — read that as AWS signalling that +agentic retrieval is a heavy operation. Combined with its 6× price (§3.2), agentic +retrieval should be **selective and opt-in, never the default path.** Headroom is +roughly 30× current volume; request an increase before full adoption, not after. + +**Why 200 data sources/KB is safe despite being non-adjustable:** a data source is +an *ingestion channel* (an S3 prefix, a crawl root, a Drive folder), not a +document. One S3 data source carries unbounded documents. This only becomes a wall +if we ever model document-per-data-source — which §6.1 already rules out, and +which must stay ruled out because the limit cannot be raised. + +**Bulk upload needs no queueing of our own.** An ingestion job syncs a whole data +source, so 100 simultaneous uploads is one job, not 100. The per-KB limit of 50 +concurrent jobs is per KB, so parallel syncs across many KBs don't contend. +*Unverified:* whether an account-level ingestion-concurrency limit exists — the +quota page lists none. Probe this during the §10.3 shadow phase with a +many-KB backfill before assuming a large migration can run wide. + +**KB count is the unmeasured risk.** At 30,000 users, KB count depends entirely on +what fraction create one. 5% → 1,500 (fine). 30% → 9,000 (at the default cap). +It is adjustable, but capacity requests take lead time. Establish the current +number and the per-user creation rate: + +```bash +aws dynamodb scan --table-name boisestateai-v2-rag-assistants \ + --filter-expression "begins_with(SK, :d)" \ + --expression-attribute-values '{":d":{"S":"DOC#"}}' \ + --projection-expression "PK" --output json \ + | python3 -c 'import json,sys; d=json.load(sys.stdin); pks={i["PK"]["S"] for i in d["Items"]}; print(f"{len(pks)} KBs, {len(d[\"Items\"])} documents")' +``` + +Track that ratio against `activeUsers` monthly; it is the leading indicator for +both the KB cap and the storage curve. + +--- + +### 6.5 Proposed: agentic retrieval as a user-triggered escalation + +Rather than a per-agent config flag, expose agentic retrieval as an explicit +**per-answer escalation**. The cheap `Retrieve` path runs by default; when an +answer reads as thin or incomplete, the user re-drives that turn with a deeper +search via a single control on the message. + +**This dissolves §6.4's quota wall.** At full adoption the standard path projects +to ~25.9 RPM average / ~130 RPM peak. If escalation is opt-in, only the escalated +fraction consumes the 60 RPM account budget: + +| Escalation rate | Agentic peak RPM at 30,000 users | Against the 60 RPM cap | +|---|---|---| +| 10% | ~13 | comfortable | +| 30% | ~39 | comfortable | +| 100% (default-on) | ~130 | **exceeds** | + +The constraint stops being "30× current volume" and becomes "keep escalation under +roughly a third of turns" — which a deliberate user action will trivially satisfy. + +**It also reprices the feature.** An escalation regenerates the whole turn, so it +costs ~$0.044 (LLM) + ~$0.006 (agentic retrieval) ≈ **$0.05**, of which the +retrieval meter is only ~12%. The $4/1k headline stops being the thing to reason +about; the escalation is really "pay for one more turn," which is a far easier +budget conversation and already flows through the existing quota tiers. + +**The best side effect is the data.** Every escalation is a human-labelled +*"cheap retrieval was insufficient here"* example, complete with the query and the +KB. That is the retrieval-quality dataset we do not currently collect — today the +signal is discarded. Log the pair regardless of whether the escalation succeeds. + +**Preconditions — do not ship this before them:** + +1. **Understand the 45% no-vector rate first.** 1,743 of 3,886 retrievals found no + vectors at all (§3.2). Agentic retrieval over a corpus that lacks the content + returns nothing either — the user pays 6×, waits longer, and receives the same + answer. The control would read as broken through no fault of its own. *(The + separate 7% emptied by the doc-status filter is **correct** behaviour over + deleted and failed documents — see §7.4 — and needs no fix for this feature.)* +2. **Gate the control on retrieval having found something.** Where the KB + genuinely had no vectors, the honest affordance is *"this agent has no material + on that topic"*, not *"search harder"*. Escalation is for **thin** results, not + **absent** ones. +3. **One escalation per turn.** Disable the control after use so it cannot be + spammed at ~$0.05 a click, and confirm the escalated turn meters through the + normal quota path. +4. **Measure agentic latency before exposing it.** §5's ~672 ms is for standard + `Retrieve`. Multi-hop planning plus N retrievals plus regeneration is unmeasured + and will need a progress affordance — acceptable because the user opted in, + unacceptable if it looks hung. + +**Decision to lock now, so it is not relitigated: escalation stays +human-triggered.** Auto-escalating on low retrieval scores is the obvious next +proposal and it is a trap — it reintroduces default-on agentic (the 130 RPM row +above) and, given the 45%-empty rate, would fire constantly on exactly the queries +it cannot help. + +**Set expectations on where it can help.** Agentic retrieval plans multi-hop +queries, so it wins when an answer must combine facts across documents. It does +not fix a corpus that lacks the content, a model that ignored a chunk it was +given, or whole-document tasks such as summarise/reformat (§6.3). Track +escalation → *satisfaction*, not escalation rate alone; a high escalation rate +that does not improve answers means the retrieval problem is upstream. + +**Naming:** avoid "search harder" in the UI — it implies the first attempt was +lazy and indicts the default path. Prefer framing it as a different strategy: +"Search more deeply", "Research this further". + --- ## 7. Lifecycle — bounding growth and reclaiming cost @@ -343,6 +671,50 @@ Emit EMF alongside the PromptCache metrics: `KbCount`, `KbStorageGB`, `KbIdleGB` `KbReclaimedGBPerDay`, `KbOrphansFound`. A sustained non-zero orphan count is the only signal that the delete saga is leaking. +**Document-level orphans — measured 2026-08-14.** Status distribution across all +1,692 `DOC#` records: + +| `status` | Count | Share | +|---|---|---| +| `complete` | 1,492 | 88.2% | +| `deleting` | 101 | 6.0% | +| `failed` | 95 | 5.6% | +| `uploading` | 4 | 0.2% | + +200 documents (11.8%) are non-`complete`, and vectors for many of them remain in +the S3 Vectors index. `_filter_vectors_by_document_status` masks them at query +time — 936 retrievals in the trailing 30 days had chunks dropped this way, 287 of +them reduced to zero results. **The filter is behaving correctly.** The defect is +upstream, in two places: + +1. **101 stuck `deleting` records mean deleted content is still indexed.** Only a + query-time filter separates a user-deleted document from retrieval, and that + filter **fails open**: `rag_service._filter_vectors_by_document_status` catches + DynamoDB errors and sets `valid_doc_ids = doc_ids`, returning results + *unfiltered*. A DynamoDB blip therefore serves content from documents users + deleted. Fail closed — drop chunks whose status cannot be confirmed. Scope note: + the *inner* per-document handler already fails closed (a single failed `get_item` + leaves that `doc_id` out of `valid_doc_ids`, dropping its chunks); only the + *outer* table-level handler fails open, so the window is table unavailability + rather than ordinary throttling. Low probability, but the consequence is a + privacy incident rather than a bad answer, and the fix is one line. Separately, + `documents/services/cleanup_service.py` is evidently not finishing these + deletes; the same tombstone-saga pattern proposed above for KBs applies at the + document level. + + *This is the canonical description of the fail-open defect. §11.1 and §14.4 + reference it rather than restate it — keep it that way.* +2. **95 `failed` records are invisible to their owners.** Users believe those + uploads worked. §10.3 ingests only `complete` docs, so migration will silently + drop all 95 — correct for the index, wrong for the user. Surface them and offer + retry before or during migration rather than omitting them quietly. Same for + any `uploading` record that is stuck rather than in flight. + +Migration incidentally resolves the orphaned-vector *exposure* by not carrying +non-`complete` documents across — but only once §10.3's `reclaim` deletes the +legacy vectors. Until then the fail-open path is live, so fix that independently +of the migration schedule. + ### 7.5 Exemptions, day one Marketplace-published KBs exempt while listed; `taken_down` needs an explicit @@ -563,18 +935,160 @@ served traffic on managed without a rollback. ## 11. Open questions -1. **Does a KB go cold again after idleness?** Untestable in one session; decides - whether §7.2's dormant tier is cheap or expensive to reverse. -2. **Empty-KB billing rounding.** Price List says no floor structurally; the - dev-ai control KB (`kb-probe-empty-2`, zero data sources) confirms or refutes - on the next CE refresh. -3. **Do unsupported filter operators fail open?** Blog claim only, and - security-relevant. Test before any metadata-based boundary. -4. **Native Google Drive connector vs our AgentCore-Identity adapter** — could - retire the vault principal-binding blocker, but moves token custody. -5. **Managed parser vs Docling** on our actual corpus (tables, scanned PDFs). No - quality comparison has been run; §9's storage delta is only justified if the - managed parser plus reranking measurably beats what we have. +Status after the §13 benchmark (2026-08-14). Raw evidence for every answer is in +the harness findings log; the harness itself is disposable and not committed. + +1. **Does a KB go cold again after idleness?** ⏳ **In progress.** A knowledge base + was deliberately left alive with a recorded warm baseline (2nd-ingest time and + retrieval p50). Re-checking it after an extended idle period compares the same + two numbers. Deliberately not a scheduler, per §13.3. If a cold penalty appears, + §7.2's dormant tier is expensive to reverse and owners must be warned before + eviction; if not, it is cheap. +2. **Empty-KB billing rounding.** ✅ **Answered: there is no floor, confirmed + empirically.** Cost Explorer for 2026-08-01 → 2026-08-15, filtered to + `Amazon Bedrock AgentCore` and grouped by usage type: + + | usagetype | quantity | cost | + |---|---|---| + | `USW2-Knowledge-Base:Consumption-based:Storage` | **0.000000406 GB-Mo** | **$0.00000203** | + | `USW2-Knowledge-Base:Consumption-based:Retrieval` | 32 queries | $0.032 | + + Storage billed a **fractional** GB-month with no minimum rounding unit, across + an account holding three probe knowledge bases (two with zero data sources) + since 2026-08-11. §3's structural argument — that AWS models hourly floors in + this service code for Runtime and deliberately did not for Knowledge Base — now + has matching billing evidence. **An idle or empty KB is effectively free; the + cost driver is gigabytes, exactly as §7 assumes.** + + The retrieval line also confirms the rate: 32 queries at $0.032 is $0.001 each. + No `AgenticRetrieval` usage type had appeared yet at the time of reading, so + that rate remains Price-List-only rather than invoice-confirmed. + + ⚠️ Separately unresolved: the `RawDataSize` CloudWatch metric — which §7.3 wants + for storage accounting — returned **0 datapoints** for a knowledge base with one + successfully indexed document over a 60-minute lookback. Possible causes, none + confirmed: the corpus was too small (0.0003 GB), the metric may only publish + after a *sync job* rather than direct ingestion, or it lags by more than an hour. + The service role already carried the required `cloudwatch:PutMetricData` grant + scoped to `AWS/Bedrock/KnowledgeBases`, so a missing permission is **not** the + explanation. Until confirmed for direct ingestion, compute per-owner bytes from + S3 `HEAD` sizes as §14.6 recommends. +3. **Do unsupported filter operators fail open?** ✅ **Answered: they fail CLOSED.** + Against a live managed knowledge base, an unfiltered query returned 5 chunks; + `equals`, `startsWith` and `stringContains` on a key that cannot exist each + returned **0**. All three operators were *accepted*, not rejected. The blog claim + of silent-ignore, and therefore of a tenant filter failing open, is **not + reproduced**. Note this also corrects §6.2 — see below. +4. **Native Google Drive connector vs our AgentCore-Identity adapter** — ❌ **not + investigated.** Out of scope for the benchmark; still open. +5. **Managed parser vs Docling on our actual corpus.** ✅ **Answered decisively, and + this is the finding that clears the §13.4 gate.** On a 9-question set with every + variable held constant, the current pipeline answered **4/9** and managed + answered **9/9**. By document class: plain text 3/3 both (no regression); native + layout PDF 1/3 current versus 3/3 managed; scanned image PDF **0/3** current + versus 3/3 managed. + + Two specifics worth carrying into the design: + - The current pipeline **discards most of a machine-readable PDF**. For a + two-page PDF with two-column prose, a five-row table and a chart, Docling + produced **one chunk** containing only the title and first sentence. The table + was never extracted, even though its text *is* present in the PDF text layer + (verified independently with PyMuPDF). This is not an OCR gap; it is a parsing + gap on documents we can already read. + - The current pipeline **cannot ingest a scanned PDF at all**: + `ValueError: Docling produced zero chunks`, surfaced to the user as the generic + *"Processing failed — please try again or contact support"*. 6 s of Lambda time, + 1.4 GB of 3 GB memory used — a capability gap, not a resource limit. + +### 11.1 Corrections to earlier sections, from measurement + +- **§6.2 is wrong that Managed KB lacks `startsWith`/`stringContains`.** Both are + present in `managedSearchConfiguration.filter`, alongside `equals`, `notEquals`, + `in`, `notIn`, `greaterThan(OrEquals)`, `lessThan(OrEquals)`, `listContains`, + `andAll` and `orAll` — and they fail closed (question 3). +- **§6.2's cross-KB fan-out economics are wrong.** It assumes N parallel `Retrieve` + calls plus our own reranker at ~$0.002/turn because "managed reranking is free + *within* a KB, not across them". `AgenticRetrieveStream` accepts a **`retrievers` + list** — each entry a `knowledgeBaseId` plus a natural-language `description` — + and applies managed reranking across all of them in one call. The Cohere line + item disappears. What replaces it is a quota, not a price: see below. +- ⚠️ **`AgenticRetrieveStream` is limited to 60 requests per minute per ACCOUNT** + (adjustable). That is roughly one request per second platform-wide. Agentic + retrieval is the mechanism behind query decomposition and multi-hop reasoning, so + **this quota must be raised before anything depends on it**, and it cannot be the + default retrieval path until then. This limit appears nowhere else in this + document. +- **Hybrid search cannot be toggled.** There is no `overrideSearchType` for managed + knowledge bases, and the configuration branch that carries it is rejected + outright. Hybrid is simply how managed retrieval works — so it can be compared + against today's dense-only pipeline, but not A/B tested against itself. +- **`IngestKnowledgeBaseDocuments` is capped at 10 documents per call, server-side.** + §10.3 was right to treat 10 as the safe limit. Confirmed by sending 11 with SDK + validation disabled: *"The number of documents (11) exceeds the maximum allowed + (10) for MANAGED knowledge base type."* AWS's user-guide claim of 25 does not + apply to managed knowledge bases. +- **One service role can serve many knowledge bases.** AWS's note that "a policy + cannot be shared between multiple roles" does not prevent role reuse — a second + knowledge base created against the first one's role reached ACTIVE normally. + 10,000 knowledge bases do not require 10,000 roles. +- **Custom embeddings work and cost nothing extra in time.** Pinning + `amazon.titan-embed-text-v2:0` (`embeddingModelType: CUSTOM`) produced identical + cold-ingest time and identical answer quality to the built-in embedding — 9/9 + either way. AWS constrains custom embeddings to **float32 with 1024 dimensions**, + which is exactly Titan v2's shape. A migration can therefore keep today's + embedding model for continuity, which matters because the choice is immutable + after creation. +- **Image extraction is opt-in.** Multimodal parsing requires + `mediaExtractionConfiguration.imageExtractionConfiguration.imageExtractionStatus + = ENABLED` on the data source; audio and video have sibling toggles. Left at its + default, chart and image content is never described and never indexed — a silent + loss of the capability being paid for. The mechanism is worth knowing: the parser + runs a vision model and indexes a **generated textual description**, e.g. + ` Bar Chart / Column Graph Figure 1. + … `. +- **Managed reranking measurably does the work.** With reranking, the five returned + scores separate sharply (0.89, 0.38, 0.25, 0.21, 0.19); with + `rerankingModelType: NONE` they are nearly flat (1.00, 0.84, 0.78, 0.77, 0.77) and + ordering changed on two of three queries. The flat case matters for the context + cap: when scores are undiscriminating, truncating to the top two chunks is close + to arbitrary. **The reranker is what makes a small context cap defensible.** +- **ACL-aware retrieval exists and fails closed**, which is *better* than today's + document-status filter — ours fails open on a table-level DynamoDB error (§7.4). + AWS is explicit that ACL awareness "is not + authorization" and does not authenticate users, so app-side authorization is still + required. Identity is **email only**, with no alias resolution; mismatches fail + silently. +- **Resource policies are MANAGED-only** and give genuine IAM-enforced sharing for + `bedrock:Retrieve` and `bedrock:GetDocumentContent` — the infrastructure-level + isolation §6.2 said metadata filtering could not provide. ⚠️ They attach to the + **AWS knowledge base ARN**, so any dormancy/rehydration cycle that produces a new + AWS id silently drops sharing and must re-apply it. +- **`DELETE_UNSUCCESSFUL` is a real terminal state, and the orphan risk in §7.4 is + already live.** The dev account contains a knowledge base + (`derrick-rag-test-delete-me`) stuck in `DELETE_UNSUCCESSFUL` since 2025-11-24, + with the failure naming its own remedy: set the data source's + `dataDeletionPolicy` to `RETAIN` and retry. Set that policy deliberately at + `CreateDataSource` time, and never treat "delete call accepted" as "resource + gone" — deletion is asynchronous and knowledge bases sat in `DELETING` for + minutes. + +### 11.2 Unrelated production bugs found while benchmarking + +Both are independent of this decision and worth fixing on their own. + +1. **The pipeline cannot ingest `.txt` at all.** The deployed Docling build has no + plain-text input format: *"File format not allowed: tmp….txt"*. The repo claims + support in three places (`documents/ingestion/handler.py`'s extension map, and + `docling_processor.py`'s `SUPPORTED_MIME_TYPES` and + `DOCLING_SUPPORTED_EXTENSIONS`) and the frontend's + `file-upload.service.ts` lists `.txt` as allowed. A user can upload one, wait + 56 s, and get *"Processing failed — please try again or contact support"*. + Cheapest fix: convert `text/plain` to markdown before handing it to Docling + (markdown *is* supported), or reject the format at upload with a clear message. +2. **Scanned PDFs fail with the same opaque message** (zero chunks, see question 5). + Whatever happens with Managed KB, the user-facing error should distinguish + "this format isn't supported" from "something broke". + ## 12. Probe resources @@ -583,6 +1097,49 @@ Live in dev-ai until torn down: KBs `kb-probe-empty-1`/`VZKNLS9T1F`, `kb-probe-loaded`/`DAK4HL3JU7`; IAM role `kb-billing-probe-role`. Keep until the §11 question 2 CE read, then delete all four. +**Update 2026-08-14: the §11 question 2 Cost Explorer read is done** (see §11), and +**all four probe resources have been deleted** — knowledge bases `VZKNLS9T1F`, +`0EKHSBWBOA` and `DAK4HL3JU7`, then IAM role `kb-billing-probe-role`. Total +Knowledge-Base storage charge they accrued for the month was **$0.00000203**. + +Two operational notes from that teardown, both relevant to §14.5's teardown ordering +and §7.2's reclaim timing: + +- **Deletion took 2–6 minutes** per knowledge base and was verified by polling + `ListKnowledgeBases` until the names disappeared. "Delete call accepted" is not + "resource gone", and any teardown that assumes otherwise will race. +- **The service role was deleted only after all three knowledge bases were + confirmed absent.** Deleting it while a knowledge base is still `DELETING` is a + plausible route into exactly the `DELETE_UNSUCCESSFUL` state described in §12.2. + A role also cannot be deleted until its inline policies are removed. + + +### 12.1 Additional resources from the §13 benchmark + +- **One knowledge base is deliberately still alive** for the §11 question 1 idle + test, with a recorded warm baseline (2.533 s ingest, retrieval p50 711 ms). It is + a paying resource; the harness records its id locally and has an explicit teardown + command. Delete it once the idle check has been taken. +- Everything else the benchmark created — knowledge bases, data sources, service + roles, staged S3 objects and `DOC#` rows — was removed and verified absent by + polling `ListKnowledgeBases` until the names disappeared, rather than trusting + that the delete call was accepted. + +### 12.2 Pre-existing debris found in dev, unrelated to this work + +⚠️ A knowledge base named **`derrick-rag-test-delete-me`** (`ZZ13KI12J1`, type +`VECTOR`) has been stuck in **`DELETE_UNSUCCESSFUL`** since 2025-11-24, with a last +update attempt on 2026-08-05. Its failure message names the remedy: *"consider +updating the dataDeletionPolicy of the data source to RETAIN and retry your +request."* + +This is a live instance of exactly the orphan class §7.4 describes — a resource +someone tried to delete, whose deletion failed, which no reconciler would ever +notice. It is a classic `VECTOR` knowledge base so it does not bill at the managed +$5.00/GB-month rate, but it implies a vector store still exists somewhere. Worth +cleaning up independently of this evaluation. + + --- ## 13. Required pre-build benchmark — current vs managed, 1:1 @@ -665,10 +1222,129 @@ Proceed to a product vertical slice only if the comparison shows: If Managed KB does not clear this gate, keep S3 Vectors and test the current 2,000-character cap independently before taking on a migration. +### 13.5 Gate outcome — CLEARED (2026-08-14) + +| Gate condition | Result | Evidence | +|---|---|---| +| no regression on plain text | ✅ **met** | 3/3 both backends | +| measurable benefit on layout/OCR documents | ✅ **met, large** | layout PDF 1/3 → 3/3; scanned PDF 0/3 → 3/3 | +| acceptable first-document delay as background work | ✅ **met** | 68 s cold ingest; 47–124 s KB creation; never interactive | +| subsequent-ingest improvement over the current warm path | ⚠️ **mixed** | comparable on small text (**2.5 s** managed vs 2.1–12.8 s current); much slower on PDFs (37–264 s vs 8–13 s) | +| acceptable added retrieval latency at p95 | ✅ **met** | +538 ms at p95 (257 ms → 762–800 ms) | + +**Four conditions met, one mixed.** On a comparable small text file the warm +managed ingest measured **2.533 s** — as fast as the current pipeline, and matching +§5's original "~4–6 s warm" observation. The order-of-magnitude gap only appears on +PDFs, where managed spends 37–264 s and the current pipeline spends 8–13 s. That +comparison flatters the current pipeline for the wrong reason: it is fast on the +layout PDF because it extracted a single title-only chunk, and fast on the scanned +PDF because it gave up. Managed is slower there because it is actually doing the +parsing, OCR and image analysis that produce the 1/3 → 3/3 and 0/3 → 3/3 gains. +Both paths are background work that no user waits on. + +**Recommendation: proceed to a product vertical slice.** The overall answer rate +goes from 4/9 to 9/9 with every other variable held constant, and two whole +document classes move from unusable to working. That is the "quality gain large +enough to justify the storage premium" the gate asks for. + +Four requirements on proceeding, each grounded in a measurement above: + +1. **A per-owner byte cap must land before, not after.** Storage is 35× more + expensive per GB. The existing 1 GB-per-user precedent would be $150,000/month + at 30,000 users. This is the only finding here that can cause real financial + damage. +2. **Do not depend on agentic retrieval until the 60-per-minute account quota is + raised.** Query decomposition works well — it answered a genuine multi-hop + question correctly with citations — but at one request per second platform-wide + it cannot be a default path. +3. **Hold the 2,000-character context cap at its current value.** §9 says hold it + constant during the swap; the cap experiment (below) shows there is no reason to + change it in either direction yet. +4. **Clamp the query string before it reaches `Retrieve`.** Managed KB caps query + input at **10,000 characters and the limit is not adjustable** (§6.4). + `search_assistant_knowledgebase` currently forwards the raw user message with + an inline comment asserting no validation is needed; Titan v2's ~32,000-character + tolerance is why nothing fails today. This is the only finding that produces a + hard API failure rather than a cost or quality effect, and it is a few lines at + the §10.1 seam. + +### 13.6 Context cap experiment — result + +Run separately from the parity comparison, as §9 requires. Retrieval happened once +per question and only the cap varied; token counts are the model's own reported +`usage.inputTokens`. + +| Cap | current: correct | current: chunks to model | managed: correct | managed: chunks to model | managed: input tokens | +|---|---|---|---|---|---| +| **2000** (today) | 4/9 | 2 of 2 | 9/9 | **2 of 5** | 550 | +| 4000 | 4/9 | 2 of 2 | 9/9 | 3.8 of 5 | 1047 | +| 8000 | 4/9 | 2 of 2 | 9/9 | **5 of 5** | 1516 | +| 12000 / 20000 | 4/9 | 2 of 2 | 9/9 | 5 of 5 | 1516 | + +**No answer changed correctness at any cap, on either backend.** + +⚠️ **This corrects §9's suggestion that the cap "may be the cheapest quality win +available".** It is not, for two different reasons: + +- **On the current pipeline the cap is not the constraint — the parser is.** Quality + is flat at 4/9 from 2,000 to 20,000 characters, and chunks reaching the model stay + at **2 of 2** throughout: the cap never truncates anything, because retrieval only + ever produced one or two chunks. +- **On managed the cap binds on every single question** — each used 1,987 of 2,000 + characters and sent exactly 2 of 5 chunks — **but quality is already saturated at + 9/9**. Raising it to 8,000 admits all five chunks and costs **+966 input tokens + per turn** for no measured gain. + +**Limitation that must travel with this result:** the corpus asks single-fact +lookup questions, where one good chunk suffices by construction — precisely the case +where extra chunks cannot help. This shows the cap is not costing *these* answers; +it does **not** show the cap is harmless in general. Questions needing multi-chunk +synthesis (summarise, compare across sections, list-everything, multi-document) +could still benefit, and should be tested with a question set built for that. Note +also that agentic retrieval does its own retrieval and is **not** subject to this +cap at all. + +Sizing note if it is ever revisited: **8,000 characters** is the point where all +five chunks fit; beyond that nothing changes. + + --- ## 14. Implementation-readiness gates +### 14.0 What AWS already closes (2026-08-14) + +Reviewed against `kb-managed-acl`, `kb-managed-cross-account`, +`kb-managed-observability`, `kb-managed-quotas`, `kb-managed-prereqs` and +`kb-managed-permissions`. Several gates below are smaller than written. + +| Gate | Status | Why | +|---|---|---| +| 14.1 durable ingestion control plane | still open | AWS provides nothing for the S3-event consumer. Eased: per-document ingestion logs can be delivered to CloudWatch Logs, S3 or Firehose | +| 14.2 stable KB identity and data model | still open, **plus a new risk** | Resource policies attach to the AWS KB ARN, so replacing an id during dormancy/rehydration silently drops sharing | +| 14.3 authorization and publication | **partially closed** | ACL-aware retrieval exists and fails closed; resource policies give IAM-enforced `Retrieve`/`GetDocumentContent` sharing. But AWS states ACL awareness "is not authorization" and does not authenticate users, so app-side authz remains ours | +| 14.4 provisioning and deletion sagas | still open, eased | `clientToken` on create operations; native `deletionProtectionConfiguration` (status + threshold) on the connector | +| 14.5 IAM and encryption | **closed** | AWS documents the exact confused-deputy trust policy this gate asks for: `aws:SourceAccount` plus `ArnLike AWS:SourceArn` on `knowledge-base/*`, `iam:PassRole` conditioned on `iam:PassedToService`, S3 conditioned on `aws:ResourceAccount`, and KMS via `serverSideEncryptionConfiguration.kmsKeyArn`. Teardown of runtime-created resources remains ours | +| 14.6 cost and quota controls | **partially closed, plus a hard new ceiling** | Quotas confirmed (10,000 KBs adjustable; 200 data sources, 50 concurrent ingestion jobs, 10 TB storage, 10,000 query characters all **not** adjustable; `Retrieve` 600/min per KB). New: `AgenticRetrieveStream` **60/min per account** | +| 14.7 deployment choreography | unchanged | entirely ours | +| 14.8 test matrix | unchanged, **plus three additions** | resource-policy re-application after rehydration; CloudWatch metric-permission presence; ACL fail-closed behaviour | +| §7.3 metrics | **largely closed** | `AWS/Bedrock/KnowledgeBases` publishes `Invocations`, `ClientErrors`, `ServerErrors`, `Throttles`, `TotalIterationCount` and `RawDataSize` (GB per `KnowledgeBaseId`) at no charge. `Invocations` per KB is a cheaper idleness signal than the throttled conditional write §7.3 proposes | + +⚠️ Two things to encode rather than discover later: + +- **Metric publishing is best effort and permission-gated.** It needs + `cloudwatch:PutMetricData` scoped to the `AWS/Bedrock/KnowledgeBases` namespace on + **both** the KB service role (for `Retrieve`) and the *calling identity* (for other + operations, via a forward access session). Omit it and metrics silently vanish + while requests keep succeeding. Worth a CDK IAM assertion. +- **`RawDataSize` has not yet been observed for a directly-ingested document** — + see §11 question 2. Do not build quota enforcement on it until confirmed. + +Also: **managed embedding and managed reranking require no Bedrock model access at +all.** Model access is only needed for `embeddingModelType: CUSTOM` or +`rerankingModelType: CUSTOM`. + + The topology decision is approved for evaluation, not implementation. The following details must be written into this spec or a linked design before PR-1. They are blocking because each one otherwise creates a leak, lockout, or @@ -735,9 +1411,10 @@ the process dies. Use durable tombstones for whole-KB, data-source, and individual-document deletes. Do not erase the last DDB record or let TTL remove it until AWS confirms -deletion. Keep the document-status filter during migration and make lookup -failure **fail closed**; the current legacy filter's unfiltered fallback is not -safe for deleting or access-controlled content. +deletion. Keep the document-status filter during migration and make lookup failure +**fail closed** — §7.4 carries the measured defect and the exact code path. It is +not safe for deleting or access-controlled content, and because it is live today it +should be fixed ahead of migration rather than as part of it. ### 14.5 IAM, encryption, audit, and teardown diff --git a/docs/specs/managed-kb-cost-attribution.md b/docs/specs/managed-kb-cost-attribution.md new file mode 100644 index 000000000..fca1ed562 --- /dev/null +++ b/docs/specs/managed-kb-cost-attribution.md @@ -0,0 +1,133 @@ +# Managed Knowledge Base cost attribution + +How to find what Managed Knowledge Base actually costs, and the two ways of asking +that quietly return the wrong number. + +Requirement 22.7 of `.kiro/specs/managed-kb-migration`. Figures are from +`docs/specs/bedrock-managed-kb-evaluation.md` §2 and §11.3, measured against +dev-ai — not estimates. + +--- + +## The two wrong queries + +**Keying on `AmazonBedrock` returns nothing at all.** Managed KB bills under the +service code **`AmazonBedrockAgentCore`**. Searching the Bedrock service code for +knowledge base SKUs returns an empty result, so a cost query filtered on +`AmazonBedrock` reports `$0.00` — and reports it *successfully*, which is worse +than an error. Nobody investigates a zero they asked for. + +**Keying on the service code alone blends it into Runtime memory.** AgentCore's +bill is dominated by the Runtime memory line — the 2026-08 AICC report put it at +**73%** of the AgentCore total. A query grouped by service code therefore shows a +large, slowly-growing AgentCore number in which a knowledge base storage curve of +any plausible size is invisible. + +**So: filter on `usagetype`.** + +--- + +## The three SKUs + +Exactly three per region, all `Consumption-based`, all `beginRange=0 → +endRange=Inf` — there is no tier-0 minimum block. + +| usagetype (us-west-2) | rate | +|---|---| +| `USW2-Knowledge-Base:Consumption-based:Storage` | $5.00 / GB-month | +| `USW2-Knowledge-Base:Consumption-based:Retrieval` | $0.001 / query | +| `USW2-Knowledge-Base:Consumption-based:AgenticRetrieval` | $0.004 / query, **stacks on** Retrieval | + +`AgenticRetrieval` should not appear on our bill: agentic retrieval is not enabled +on any path (Requirement 3.5), and its 60 RPM *account-wide* quota is why. A +non-zero value on that line means something turned it on. + +Storage at $5.00/GB-month is roughly **35×** the current S3 Vectors cost of about +$0.15/GB-month. That ratio, not the absolute number, is what the per-owner byte cap +exists to bound. + +## There is no idle floor + +Of 6,467 AgentCore usagetypes, 6,124 contain `Hours` — Runtime carries +`Instance-based::Management-Hours`. **Zero of the 21 Knowledge-Base +usagetypes do.** AWS models hourly floors in this exact service code and +deliberately did not for knowledge bases. Three empty probe knowledge bases left +running for a month billed **$0.00000203** in total. + +This is why lazy provisioning is safe and why an idle knowledge base is a storage +problem rather than a fixed cost. + +--- + +## The query + +```bash +aws ce get-cost-and-usage \ + --region us-east-1 \ + --time-period Start=2026-08-01,End=2026-09-01 \ + --granularity DAILY \ + --metrics UnblendedCost UsageQuantity \ + --filter '{"And":[ + {"Dimensions":{"Key":"SERVICE","Values":["Amazon Bedrock AgentCore"]}}, + {"Dimensions":{"Key":"USAGE_TYPE_GROUP","Values":[]}} + ]}' \ + --group-by Type=DIMENSION,Key=USAGE_TYPE +``` + +Notes that cost time if you skip them: + +- **Cost Explorer is `us-east-1` only.** The call fails elsewhere regardless of + where the knowledge bases live. +- **Group by `USAGE_TYPE`, then read the `Knowledge-Base:` rows.** Grouping by + `SERVICE` is the blended-into-Runtime-memory mistake above. +- **The usagetype carries a region prefix** (`USW2-`, `USE1-`). A filter written + against one region silently returns nothing for another, and KB SKUs exist in + seven regions only: USW2, USE1, EU, EUC1, EUW2, APN1, APS2. +- **Cost Explorer lags by up to 24 hours.** For "did we just spend something + alarming", read the CloudWatch metrics below instead. + +--- + +## What to watch instead of the bill + +Cost Explorer answers "what did we spend". These answer "what are we about to +spend", which is the question worth alarming on. Both sets live in +`{projectPrefix}/ManagedKb`. + +| Metric | Reads as | +|---|---| +| `KbStorageGB` | The storage curve. Multiply by $5.00 for a monthly run rate | +| `KbCount` | Progress toward the adjustable 10,000-knowledge-base account cap | +| `KbIdleGB` | Bytes nothing has needed for `KB_IDLE_THRESHOLD_DAYS`. **Baseline only in this phase** — nothing reclaims yet, and the follow-up spec's eviction threshold has to be chosen from this distribution, which cannot be backfilled | +| `KbByteCapRejected` | Whether the 100 MB default cap is actually workable before it hardens into policy | + +`KbReclaimedGBPerDay` is deliberately **not** emitted. Nothing reclaims in this +phase, so it would be structurally zero — and a permanently-zero metric on a +dashboard teaches people to stop reading the dashboard. + +### Reading Bedrock's own metrics + +Per-knowledge-base `Invocations` lives in Bedrock's `AWS/Bedrock/KnowledgeBases` +namespace and is a cheaper idleness signal than anything we can compute. That +namespace is a **read source only**: CloudWatch reserves every namespace beginning +with `AWS` and rejects writes to them, so the platform's own metrics go to +`{projectPrefix}/ManagedKb` (Requirement 20.10) and reading Bedrock's requires +`cloudwatch:GetMetricData` / `GetMetricStatistics` (Requirement 20.13). + +Conflating those two directions once produced a `PutMetricData` grant scoped to +`AWS/Bedrock/KnowledgeBases`, which would have deployed cleanly and published +nothing, forever. + +--- + +## Tags are not enforcement + +Knowledge bases are tagged at creation with `prefix`, `env`, `appKbId` and +`ownerUserId` (Requirement 20.11), and those tags are what make a tag-filtered +`ListKnowledgeBases` — and therefore the reconciler and teardown — possible at all. + +They are **not** the byte cap. Cost-allocation tags take up to 24 hours to appear +and are not queryable synchronously, so nothing that has to refuse an upload can +depend on them (Requirement 12.8). `ownerUserId` is also deliberately an opaque +identifier and never an email address: anyone with `bedrock:ListKnowledgeBases` can +read a tag. diff --git a/frontend/ai.client/src/app/knowledge-base/kb-upgrade.service.spec.ts b/frontend/ai.client/src/app/knowledge-base/kb-upgrade.service.spec.ts new file mode 100644 index 000000000..dc37a7ea1 --- /dev/null +++ b/frontend/ai.client/src/app/knowledge-base/kb-upgrade.service.spec.ts @@ -0,0 +1,152 @@ +import { describe, it, expect, beforeEach, afterEach } from 'vitest'; +import { TestBed } from '@angular/core/testing'; +import { provideHttpClient } from '@angular/common/http'; +import { HttpTestingController, provideHttpClientTesting } from '@angular/common/http/testing'; +import { KbUpgradeService, UpgradeStatus } from './kb-upgrade.service'; +import { ConfigService } from '../services/config.service'; + +const ENTITY = 'ast-1'; +const BASE = 'http://api.test/assistants/ast-1/knowledge-base/upgrade'; + +/** + * The upgrade surface, whose defining property is that it fails soft. + * + * This card is decoration on a page whose real job — uploading and listing + * documents — works regardless. Every assertion about failure here exists so a + * broken or absent upgrade endpoint cannot take that page down with it. + */ +describe('KbUpgradeService', () => { + let service: KbUpgradeService; + let http: HttpTestingController; + + beforeEach(() => { + TestBed.resetTestingModule(); + TestBed.configureTestingModule({ + providers: [ + provideHttpClient(), + provideHttpClientTesting(), + { + provide: ConfigService, + useValue: { appApiUrl: () => 'http://api.test' }, + }, + ], + }); + service = TestBed.inject(KbUpgradeService); + http = TestBed.inject(HttpTestingController); + }); + + afterEach(() => { + TestBed.resetTestingModule(); + }); + + describe('getStatus', () => { + it('reads the status from the knowledge-base upgrade endpoint', async () => { + const pending = service.getStatus(ENTITY); + const request = http.expectOne(BASE); + expect(request.request.method).toBe('GET'); + request.flush({ + phase: 'available', + canUpgrade: true, + progress: { completed: 0, total: 3, skipped: 1 }, + reason: null, + noticePending: false, + documentsNotCarried: [], + } satisfies UpgradeStatus); + + const status = await pending; + expect(status.phase).toBe('available'); + expect(status.progress?.total).toBe(3); + }); + + it('resolves to "nothing to show" when the request fails', async () => { + const pending = service.getStatus(ENTITY); + http.expectOne(BASE).flush('boom', { status: 500, statusText: 'Server Error' }); + + const status = await pending; + expect(status.phase).toBe('none'); + expect(status.canUpgrade).toBe(false); + expect(status.documentsNotCarried).toEqual([]); + }); + + it('never leaves documentsNotCarried undefined on a partial payload', async () => { + // A field the server omits must not become `undefined` on the way to a + // template that calls `.length` on it. + const pending = service.getStatus(ENTITY); + http.expectOne(BASE).flush({ phase: 'available', canUpgrade: true }); + + const status = await pending; + expect(status.documentsNotCarried).toEqual([]); + expect(status.noticePending).toBe(false); + }); + }); + + describe('start and retry', () => { + it('posts to the upgrade endpoint to start', async () => { + const pending = service.start(ENTITY); + const request = http.expectOne(BASE); + expect(request.request.method).toBe('POST'); + request.flush({ phase: 'in_progress', started: true, message: 'Upgrade started.' }); + + expect((await pending).started).toBe(true); + }); + + it('posts to the retry endpoint to restart', async () => { + const pending = service.retry(ENTITY); + const request = http.expectOne(`${BASE}/retry`); + expect(request.request.method).toBe('POST'); + request.flush({ phase: 'in_progress', started: true, message: 'Upgrade restarted.' }); + + expect((await pending).started).toBe(true); + }); + + it("rejects with the server's own message so the caller can show it", async () => { + const pending = service.start(ENTITY); + http + .expectOne(BASE) + .flush( + { detail: 'Upgrades are not being accepted at the moment.' }, + { status: 409, statusText: 'Conflict' }, + ); + + await expect(pending).rejects.toThrow('Upgrades are not being accepted at the moment.'); + }); + + it('falls back to actionable copy that never mentions a status code', async () => { + const pending = service.start(ENTITY); + http.expectOne(BASE).flush(null, { status: 500, statusText: 'Server Error' }); + + await expect(pending).rejects.toThrow(/Nothing has changed/); + await pending.catch((err: Error) => { + expect(err.message).not.toMatch(/\b500\b/); + }); + }); + }); + + describe('dismissNotice', () => { + it('posts to the notice endpoint', async () => { + const pending = service.dismissNotice(ENTITY); + const request = http.expectOne(`${BASE}/notice`); + expect(request.request.method).toBe('POST'); + request.flush(null, { status: 204, statusText: 'No Content' }); + await expect(pending).resolves.toBeUndefined(); + }); + + it('swallows failure, because the notice is already hidden locally', async () => { + // An error toast reading "we could not forget something" is pure noise; + // the worst real consequence is the notice returning on the next load. + const pending = service.dismissNotice(ENTITY); + http.expectOne(`${BASE}/notice`).flush(null, { status: 500, statusText: 'Server Error' }); + await expect(pending).resolves.toBeUndefined(); + }); + }); + + it('never says "vector" in any message it can produce', async () => { + // Requirement 23.6. The service authors only the fallback strings; the rest + // come from the server, which has its own sweep over the same rule. + const pending = service.start(ENTITY); + http.expectOne(BASE).flush(null, { status: 500, statusText: 'Server Error' }); + await pending.catch((err: Error) => { + expect(err.message.toLowerCase()).not.toContain('vector'); + }); + }); +}); diff --git a/frontend/ai.client/src/app/knowledge-base/kb-upgrade.service.ts b/frontend/ai.client/src/app/knowledge-base/kb-upgrade.service.ts new file mode 100644 index 000000000..9edd46088 --- /dev/null +++ b/frontend/ai.client/src/app/knowledge-base/kb-upgrade.service.ts @@ -0,0 +1,156 @@ +import { Injectable, inject, computed } from '@angular/core'; +import { HttpClient } from '@angular/common/http'; +import { firstValueFrom } from 'rxjs'; +import { ConfigService } from '../services/config.service'; + +/** + * The derived, UI-facing phase of a knowledge base upgrade. + * + * Deliberately not the backend record's internal migration state: `shadow`, + * `verify` and `promote` all arrive here as `in_progress`, so the client cannot + * grow a dependency on step names that belong to the worker. + * + * `none` means render nothing at all — no badge, no banner, no prompt. + */ +export type UpgradePhase = 'none' | 'available' | 'in_progress' | 'succeeded' | 'failed'; + +/** + * Why a document will not be carried across. + * + * `unsupported_format` and `processing_failure` are separate because the user's + * next action differs: convert and re-upload, versus retry. Telling someone to + * retry a file the platform cannot read teaches them the retry button is broken. + */ +export type DocumentIssueKind = + | 'unsupported_format' + | 'processing_failure' + | 'still_processing' + | 'being_removed'; + +export interface UpgradeProgress { + completed: number; + total: number; + skipped: number; +} + +export interface DocumentNotCarried { + documentId: string; + filename: string; + /** The stored processing status, verbatim. Not for display. */ + status: string; + kind: DocumentIssueKind; + /** Plain-language explanation from the server, safe to render directly. */ + message: string; + retryable: boolean; +} + +export interface UpgradeStatus { + phase: UpgradePhase; + canUpgrade: boolean; + progress: UpgradeProgress | null; + reason: string | null; + noticePending: boolean; + documentsNotCarried: DocumentNotCarried[]; +} + +export interface UpgradeResult { + phase: UpgradePhase; + /** False when the call found an upgrade already running. Still a success. */ + started: boolean; + message: string; +} + +/** What the client falls back to when the status call fails. */ +const NOTHING_TO_SHOW: UpgradeStatus = { + phase: 'none', + canUpgrade: false, + progress: null, + reason: null, + noticePending: false, + documentsNotCarried: [], +}; + +/** + * The knowledge base upgrade surface. + * + * Every method fails soft. This is an optional card on a page whose primary job + * — uploading and listing documents — works regardless, so a failing upgrade + * endpoint must never be able to take the section down with it. + */ +@Injectable({ providedIn: 'root' }) +export class KbUpgradeService { + private readonly http = inject(HttpClient); + private readonly config = inject(ConfigService); + private readonly baseUrl = computed(() => `${this.config.appApiUrl()}/assistants`); + + private url(entityId: string, suffix = ''): string { + return `${this.baseUrl()}/${entityId}/knowledge-base/upgrade${suffix}`; + } + + /** + * Read what the card should render. + * + * Resolves to `phase: 'none'` rather than rejecting, so a caller can assign + * the result straight to a signal. A viewer gets an honest phase but never + * `canUpgrade`; the server decides that, not this client. + */ + async getStatus(entityId: string): Promise { + try { + const status = await firstValueFrom( + this.http.get(this.url(entityId)), + ); + return { ...NOTHING_TO_SHOW, ...status }; + } catch { + return NOTHING_TO_SHOW; + } + } + + /** Opt in. Rejects with a user-safe message so the caller can toast it. */ + async start(entityId: string): Promise { + return this.post(entityId, ''); + } + + /** Restart after a failure, on a fresh generation server-side. */ + async retry(entityId: string): Promise { + return this.post(entityId, '/retry'); + } + + /** + * Dismiss the one-time success notice. + * + * Swallows failure: the notice is already hidden locally by the time this is + * called, and an error toast for "we could not forget something" is noise. + * The worst case is the notice returning on the next page load. + */ + async dismissNotice(entityId: string): Promise { + try { + await firstValueFrom(this.http.post(this.url(entityId, '/notice'), {})); + } catch { + // Intentionally ignored — see above. + } + } + + private async post(entityId: string, suffix: string): Promise { + try { + return await firstValueFrom( + this.http.post(this.url(entityId, suffix), {}), + ); + } catch (err: unknown) { + throw new Error(this.messageFrom(err)); + } + } + + /** + * Prefer the server's `detail`, which is written for a user. + * + * The generic fallback never mentions a status code: "409" tells the reader + * nothing they can act on. + */ + private messageFrom(err: unknown): string { + const detail = (err as { error?: { detail?: unknown } })?.error?.detail; + if (typeof detail === 'string' && detail.trim()) { + return detail; + } + return 'The upgrade could not be started. Nothing has changed — please try again.'; + } +} diff --git a/frontend/ai.client/src/app/knowledge-base/knowledge-base-section.component.html b/frontend/ai.client/src/app/knowledge-base/knowledge-base-section.component.html index 684bbacc9..7be14603a 100644 --- a/frontend/ai.client/src/app/knowledge-base/knowledge-base-section.component.html +++ b/frontend/ai.client/src/app/knowledge-base/knowledge-base-section.component.html @@ -11,6 +11,227 @@

Knowledge base } + + + @if (showUpgradeOffer()) { + +
+
+
+
+ } + + @if (showUpgradeProgress()) { + +
+
+
+
+ } + + @if (showUpgradeNotice()) { + +
+
+ } + + @if (showUpgradeFailure()) { + + + } + + @if (showStrandedDocuments()) { + +
+ + + @if (strandedExpanded()) { +
    + @for (doc of strandedDocuments(); track doc.documentId) { +
  • +

    + + {{ doc.filename }} + + + {{ strandedHeading(doc.kind) }} + +

    + +

    {{ doc.message }}

    + @if (doc.retryable) { +

    + To fix it, upload the file again using + Add files above. +

    + } +
  • + } +
+ } +
+ } +