From b8968a4edcf1c8f65e8e88d38d942d08100736ce Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 17:15:51 +0000 Subject: [PATCH 01/90] Add a pure kernel for derived row labels. Domain, sensitivity, and project requirement are computed in one place. No product path calls it yet, so reads and writes stay as they are in v0.20.0. --- CHANGELOG.md | 1 + .../src/alicebot_api/vnext_agent_control.py | 9 + .../src/alicebot_api/vnext_derived_labels.py | 1237 +++++++++++++++++ .../src/alicebot_api/vnext_project_scope.py | 83 +- tests/unit/test_derived_labels_kernel.py | 819 +++++++++++ tests/unit/test_derived_labels_mutations.py | 161 +++ 6 files changed, 2308 insertions(+), 2 deletions(-) create mode 100644 apps/api/src/alicebot_api/vnext_derived_labels.py create mode 100644 tests/unit/test_derived_labels_kernel.py create mode 100644 tests/unit/test_derived_labels_mutations.py diff --git a/CHANGELOG.md b/CHANGELOG.md index 95c3eeacc..510fd2dde 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,7 @@ ## Unreleased +- Unreleased (on main, not in v0.20.0): the label of a derived row (its domain, its sensitivity, and the projects it requires) is now decided by one pure function. v0.20.0 kept the label a derived row was given when it was made. Nothing in the running product calls the function yet, so a read and a write still behave as they do in v0.20.0. A locked key is refused when a caller passes a project floor that is not inside the key's binding. No migration is required. - Unreleased (on main, not in v0.20.0): on SQLite, `sources delete`, `sources prune --superseded` and `import-markdown --supersede` now scrub an open loop that names the source only in its metadata: the id, or `source:`, as the text under `source_id`, `source_ids`, `source_ref`, `source_refs`, `source_references` or `selected_source_ids` at any depth. That is the rule the open-loop lookup of a source already used. They blanked only loops whose `source_id` column held the id, so a loop with an empty column kept its title, description and metadata, and they reached a new export. The delete preview now counts the same loops as the receipt. The rule reads the id only as stored or as `source:`: a loop that names the source in any spelling other than those two keeps its text, although the memory rule reads several other spellings. No migration is required. - Unreleased (on main, not in v0.20.0): the SQLite derived-label repair now updates and records each row under the id it is stored with (SQLite keeps capitals, missing hyphens, braces and `urn:uuid:` as written) and matches rows by the normalised id only to read the graph. If two stored spellings of one id exist, each is compared with the label its own recorded inputs give it, so neither is left at its old label while the other is relabelled, and neither is lowered. An update that changes no row stops the repair with `DerivedDomainRepairError` before any event or the completion stamp is written, on open and on restore; migration `20261004_0095` refuses the same way. Before, a derived memory stored under such an id kept its label while its relabel event and the stamp were written, and the pair survived export and import. A PostgreSQL uuid column was not affected. A vault that the earlier repair already stamped keeps such a row at its old label until a restore repairs it again. No migration is required. - Unreleased (on main, not in v0.20.0): v0.20.0 could store a consolidation report with a sensitivity below the memories it prints. The report took its sensitivity from the near-duplicate clusters only, so a run whose proposals were roll-up cards had no cluster to take it from and was stored as `unknown`. A key whose ceiling is below those memories (a `trusted_local_agent` key for confidential ones, a `read_only_agent` key for private ones) could then read the card topic and the member ids through `GET /v0/vnext/artifacts/{id}`. The report now takes its domain and its sensitivity over every row it names: the cluster members, the members of every proposed roll-up group, the members of the groups that a skip line names by key (`topic:...`, `entity:...`, `semantic:cluster-`), the pending, accepted, expired or held roll-up cards it names by id, and the sources that the `source_refs` of its cluster members name, which the report still prints. The open-loop review is labelled over the sources whose ids it prints as well as over its loops. The run digest of both reports covers those sources, so a source that was reclassified makes a new report instead of returning the earlier one. The first run after the upgrade over loops that link a source, or over cluster members that cite one, makes one new report, and a run that names no source keeps the digest it had. A report whose inputs are all unrestricted is now stored with the label of those inputs, `internal` where it was `unknown`, and every permission profile reads the two the same way. Reports stored by v0.20.0 keep their labels, because the stored-row repair does not change sensitivity. The other report producers (daily brief, weekly synthesis, connection and contradiction reports, project updates, staleness reports) were checked and already label over every row they print. No migration is required. diff --git a/apps/api/src/alicebot_api/vnext_agent_control.py b/apps/api/src/alicebot_api/vnext_agent_control.py index e33df175c..434f8cba8 100644 --- a/apps/api/src/alicebot_api/vnext_agent_control.py +++ b/apps/api/src/alicebot_api/vnext_agent_control.py @@ -16,6 +16,7 @@ from alicebot_api.vnext_project_scope import ( normalize_project_identifier, normalize_project_scope, + project_floor_within, project_scope_identity, resolve_project_scope, ) @@ -360,6 +361,7 @@ def evaluate_agent_policy( domains: tuple[str, ...] = (), sensitivity_allowed: tuple[str, ...] = DEFAULT_AGENT_SENSITIVITY, project_scope: tuple[str, ...] = (), + project_floor: tuple[str, ...] = (), workflow_type: str | None = None, write_policy: str | None = None, require_explicit_project_scope: bool = False, @@ -454,6 +456,13 @@ def evaluate_agent_policy( reasons.append("project_scope_binding_violation") decision = "blocked" effective_project_scope = () + elif project_floor and not project_floor_within(project_floor, identity.project_scope): + # The floor is part of the same binding test as the scope. A locked + # key reads a derived row only when every project in the floor is + # inside the binding. An empty floor does not add a refusal. + reasons.append("project_floor_binding_violation") + decision = "blocked" + effective_project_scope = () else: effective_project_scope = project_scope or identity.project_scope diff --git a/apps/api/src/alicebot_api/vnext_derived_labels.py b/apps/api/src/alicebot_api/vnext_derived_labels.py new file mode 100644 index 000000000..ed0558aa7 --- /dev/null +++ b/apps/api/src/alicebot_api/vnext_derived_labels.py @@ -0,0 +1,1237 @@ +"""Pure labels for a derived row: domain, sensitivity, and project requirement. + +Nothing in the product calls this module until a later change wires it in. +The same functions serve generation, a relabel, a read, and the stored-row +repair, so a rule lives here once. + +A derived row is found by a marker the server wrote, never by reading text. +``value.kind`` alone does not make a row derived. +""" + +from __future__ import annotations + +import json +from collections import deque +from collections.abc import Iterable, Mapping, Sequence +from dataclasses import dataclass, replace +from uuid import UUID + +from alicebot_api.vnext_agent_control import RESTRICTED_DOMAINS +from alicebot_api.vnext_derived_domain import derived_domain +from alicebot_api.vnext_derived_domain_backfill import DerivedDomainRepairError, _input_groups +from alicebot_api.vnext_project_scope import ( + is_global_scope, + normalize_project_scope, + project_floor_shape, + project_identifier_identity, + project_scope_identity, + resolve_project_scope, + source_project_scope, +) +from alicebot_api.vnext_source_fence import cited_source_ids + +# Public 1, internal 2, unknown 2, private 3, confidential 4, +# highly_sensitive 5, sacred 6, regulated 6. Raise only when the rank is +# strictly higher, so unknown and internal never swap. +SENSITIVITY_RANK: dict[str, int] = { + "public": 1, + "internal": 2, + "unknown": 2, + "private": 3, + "confidential": 4, + "highly_sensitive": 5, + "sacred": 6, + "regulated": 6, +} + +HOP_BOUND = 32 +NODE_BOUND = 5_000 +PROPAGATION_BOUND = 100_000 + +DERIVED_WORKFLOWS = frozenset( + { + "daily_brief", + "weekly_synthesis", + "connection_finder", + "contradiction_finder", + "memory_consolidation", + "project_auto_update", + "staleness_sweep", + "open_loop_review", + } +) +DERIVED_ARTIFACT_TYPES = frozenset( + { + "daily_brief", + "weekly_synthesis", + "connection_report", + "contradiction_report", + "memory_consolidation", + "open_loop_report", + } +) +_LEGACY_SUMMARY_KINDS = frozenset({"daily_brief", "weekly_synthesis"}) +_LEGACY_COUNT_KINDS = frozenset( + {"connection_finder", "contradiction_finder", "connection_report", "contradiction_report"} +) + +# Top-level marker and record keys. A write that omits one of these keeps the +# stored value. Label keys (project_scope, project_floor) are a separate set. +MARKER_KEYS = frozenset( + { + "consolidation", + "candidate_kind", + "discovered_by", + "workflow", + "source_artifact_id", + "source_id", + "input_summary", + "derived_from", + "source_ids", + "memory_ids", + "open_loop_ids", + "artifact_ids", + "belief_ids", + "source_refs", + "stale_marked_memory_ids", + "connector_name", + "input_counts", + "candidate_memory_ids", + "artifact_id", + "cluster_member_ids", + "cluster_membership", + "member_ids", + "member_snapshots", + } +) + +# Id keys the v2 planner already reads. The v3 set includes every one of them. +V2_ID_KEYS = frozenset( + { + "source_ids", + "memory_ids", + "open_loop_ids", + "artifact_ids", + "member_ids", + "cluster_member_ids", + "cluster_membership", + "stale_marked_memory_ids", + "belief_ids", + "artifact_id", + "source_artifact_id", + } +) + +_ID_KIND = { + "source_ids": "source", + "source_id": "source", + "source_artifact_id": "artifact", + "artifact_id": "artifact", + "artifact_ids": "artifact", + "memory_ids": "memory", + "memory_id": "memory", + "member_ids": "memory", + "cluster_member_ids": "memory", + "cluster_membership": "memory", + "stale_marked_memory_ids": "memory", + "open_loop_ids": "open_loop", + "belief_ids": "belief", +} +_DERIVED_FROM_KIND = { + "sources": "source", + "memories": "memory", + "open_loops": "open_loop", + "artifacts": "artifact", + "beliefs": "belief", +} +_REF_KINDS = {"source": "source", "memory": "memory", "open_loop": "open_loop", "artifact": "artifact"} +_SKIP_RECURSE = frozenset( + { + "quote", + "content", + "title", + "description", + "canonical_text", + "summary", + "content_markdown", + "prompt", + "text", + "markdown", + "body", + } +) +_KIND_ALIASES = { + "source": "source", + "sources": "source", + "memory": "memory", + "memories": "memory", + "open_loop": "open_loop", + "open_loops": "open_loop", + "artifact": "artifact", + "generated_artifacts": "artifact", + "project": "project", + "projects": "project", + "belief": "belief", + "beliefs": "belief", +} + +_EVENT_FIELDS = ("domain", "sensitivity", "project_scope", "project_floor") + + +class LabelPropagationTooLarge(ValueError): + """A relabel would touch more derived rows than the propagation bound allows.""" + + +def canon_kind(kind: object) -> str: + """Return the short kind name (``memory``, not ``memories``).""" + + text = str(kind or "") + if text not in _KIND_ALIASES: + raise ValueError(f"unknown derived label kind: {text}") + return _KIND_ALIASES[text] + + +def identifier(value: object) -> str: + """Normalize any spelling ``uuid.UUID`` accepts. Anything else is kept as text.""" + + try: + return str(UUID(str(value))) + except (ValueError, AttributeError, TypeError): + return str(value) + + +def _object(value: object) -> Mapping[str, object]: + if isinstance(value, str): + try: + value = json.loads(value) + except (ValueError, TypeError): + return {} + return value if isinstance(value, Mapping) else {} + + +def _metadata(row: Mapping[str, object]) -> Mapping[str, object]: + return _object(row.get("metadata_json")) + + +def _nonempty_str(value: object) -> bool: + return isinstance(value, str) and bool(value.strip()) + + +def _scrubbed(row: Mapping[str, object]) -> bool: + return _metadata(row).get("scrubbed") is True + + +def ordered_identifiers(values: Iterable[object]) -> tuple[str, ...]: + """Stored spellings, first one kept, ordered by project identity.""" + + chosen: dict[str, str] = {} + for item in normalize_project_scope(list(values)): + identity = project_identifier_identity(item) + if identity and identity not in chosen: + chosen[identity] = item + return tuple(chosen[key] for key in sorted(chosen)) + + +def is_derived(kind: object, row: Mapping[str, object]) -> bool: + """True when ``row`` carries a server-written derived marker. + + A redacted row has no dependencies. A marker that lives only in ``value`` + does not count, so a forged ``value.kind`` does not make the row derived. + """ + + if _metadata(row).get("redacted") is True: + return False + name = canon_kind(kind) + meta = _metadata(row) + if name in {"source", "belief"}: + return False + if name == "project": + return "derived_from" in meta + if name == "open_loop": + discovered = meta.get("discovered_by") + if not _nonempty_str(discovered): + return False + return _nonempty_str(row.get("source_id")) or _nonempty_str(meta.get("source_id")) + if name == "artifact": + if "derived_from" in meta: + return True + if meta.get("workflow") in DERIVED_WORKFLOWS: + return True + if str(row.get("artifact_type") or "") in DERIVED_ARTIFACT_TYPES: + return True + return meta.get("connector_name") == "agent_output" + if isinstance(meta.get("consolidation"), Mapping): + return True + if meta.get("candidate_kind") in {"memory_consolidation", "memory_rollup"}: + return True + if meta.get("discovered_by") == "vnext_weekly_synthesis": + return True + if meta.get("workflow") == "project_auto_update": + return True + if _nonempty_str(meta.get("source_artifact_id")): + return True + if _nonempty_str(meta.get("source_id")): + return True + return "derived_from" in meta + + +def row_class(kind: object, row: Mapping[str, object]) -> str: + """``report``, ``aggregate``, ``copy``, ``project_state`` or ``original``.""" + + if not is_derived(kind, row): + return "original" + name = canon_kind(kind) + if name == "project": + return "project_state" + if name == "artifact": + return "report" + if name == "open_loop": + return "copy" + meta = _metadata(row) + if ( + _nonempty_str(meta.get("source_artifact_id")) + or meta.get("discovered_by") == "vnext_weekly_synthesis" + or meta.get("workflow") == "project_auto_update" + or isinstance(meta.get("consolidation"), Mapping) + or meta.get("candidate_kind") in {"memory_consolidation", "memory_rollup"} + ): + return "aggregate" + return "copy" + + +def infer_kind(row: Mapping[str, object]) -> str: + """Best-effort kind for a row that did not name one.""" + + if "artifact_type" in row or ( + "content_markdown" in row and "memory_key" not in row and "canonical_text" not in row + ): + return "artifact" + if "current_state" in row: + return "project" + if "content_hash" in row and "canonical_text" not in row and "memory_key" not in row: + return "source" + if "memory_key" in row or "canonical_text" in row: + return "memory" + return "open_loop" + + +def _floor_of(row: Mapping[str, object]) -> tuple[str, tuple[str, ...]]: + """``project_floor`` from the row or from parsed metadata.""" + + if "project_floor" in row: + return project_floor_shape(row) + return project_floor_shape({"metadata_json": _metadata(row)}) + + +def group_scope(row: Mapping[str, object], *, kind: str | None = None) -> tuple[str, ...]: + """Identity of the scope united with the floor. + + An original row has no floor, so this is the identity of its scope. + """ + + name = canon_kind(kind) if kind is not None else infer_kind(row) + scope = stored_scope(name, row) + if not is_derived(name, row): + return project_scope_identity(scope) + shape, floor = _floor_of(row) + if shape != "list": + floor = () + return project_scope_identity([*scope, *floor]) + + +def stored_scope(kind: object, row: Mapping[str, object]) -> tuple[str, ...]: + """The scope a row has stored, before the effective-scope rule.""" + + name = canon_kind(kind) + if name == "source": + if _scrubbed(row): + return () + return source_project_scope(row) + if name == "project": + return () + return resolve_project_scope(row).values + + +def carries_scope(kind: object, row: Mapping[str, object]) -> bool: + """False for a scrubbed source and for a project row. Both carry no scope.""" + + name = canon_kind(kind) + if name == "project": + return False + if name == "source" and _scrubbed(row): + return False + return True + + +def _strings(value: object) -> list[str]: + found: list[str] = [] + if isinstance(value, str): + found.append(identifier(value)) + elif isinstance(value, list): + for item in value: + found.extend(_strings(item)) + return found + + +def _as_string_list(value: object) -> tuple[str, list[str]] | None: + """``(problem, ids)`` or None when ``value`` is not a list of strings. + + A repeated id is kept, so the counts check can see it and skip. + """ + + if value is None: + return None + if not isinstance(value, list): + return ("malformed", []) + if any(not isinstance(item, str) for item in value): + return ("malformed", []) + return ("", _strings(value)) + + +def _source_ids_from(value: object) -> tuple[str, set[str]]: + """Named source ids, including every spelling the saved-quote reader names.""" + + if value is None: + return "", set() + named = {identifier(item) for item in cited_source_ids(value).named} + problem = "" + if isinstance(value, list): + for item in value: + if isinstance(item, str): + token = _source_token(item) + if token: + named.add(token) + elif isinstance(item, Mapping): + nested_problem, nested = _source_ids_from(item) + problem = problem or nested_problem + named |= nested + else: + return "malformed", set() + elif isinstance(value, str): + token = _source_token(value) + if token: + named.add(token) + elif value.strip() and not named: + # A sentence is not a source id. cited_source_ids already kept the + # explicit ones. A whole token that is not a uuid still counts. + if _plain_token(value): + named.add(identifier(value)) + elif isinstance(value, Mapping): + for key, child in value.items(): + key_text = key.lower() if isinstance(key, str) else "" + if key_text in _SKIP_RECURSE: + continue + child_problem, child_ids = _source_ids_from(child) + problem = problem or child_problem + named |= child_ids + else: + return "malformed", set() + return problem, named + + +def _plain_token(value: str) -> bool: + text = value.strip() + if not text or any(character.isspace() for character in text): + return False + return ":" not in text and "/" not in text + + +def _source_token(value: str) -> str | None: + text = value.strip() + lowered = text.lower() + for prefix in ("urn:uuid:", "source:", "uuid:"): + if lowered.startswith(prefix): + text = text[len(prefix) :].strip() + lowered = text.lower() + if not text or any(character.isspace() for character in text): + return None + if ":" in text or "/" in text: + return None + return identifier(text) + + +def _typed_refs(value: object) -> set[tuple[str, str]]: + found: set[tuple[str, str]] = set() + nodes: list[object] = [value] + while nodes: + node = nodes.pop() + if isinstance(node, str): + kind, separator, row_id = node.partition(":") + if separator and kind in _REF_KINDS and row_id and _plain_token(row_id): + found.add((_REF_KINDS[kind], identifier(row_id))) + elif isinstance(node, list): + nodes.extend(node) + elif isinstance(node, Mapping): + nodes.extend(node.values()) + return found + + +def _add_ids(found: set[tuple[str, str]], kind: str, values: Iterable[str]) -> None: + for item in values: + if item: + found.add((kind, item)) + + +def _collect_metadata_ids(value: object, found: set[tuple[str, str]]) -> str: + """Walk server-written metadata. Return ``malformed`` or ``""``.""" + + problem = "" + nodes: list[object] = [value] + while nodes: + node = nodes.pop() + if isinstance(node, list): + nodes.extend(node) + continue + if not isinstance(node, Mapping): + continue + for key, child in node.items(): + if not isinstance(key, str) or key in _SKIP_RECURSE: + continue + if key == "derived_from": + continue + if key == "source_refs" or key in {"source_id", "source_ids"}: + child_problem, source_ids = _source_ids_from(child) + problem = problem or child_problem + _add_ids(found, "source", source_ids) + found.update(ref for ref in _typed_refs(child) if ref[0] != "source" or ref[1]) + continue + if key == "member_snapshots" and isinstance(child, list): + for item in child: + if isinstance(item, Mapping) and _nonempty_str(item.get("id")): + found.add(("memory", identifier(item.get("id")))) + elif isinstance(item, str): + found.add(("memory", identifier(item))) + continue + if key in _ID_KIND: + parsed = _as_string_list(child if isinstance(child, list) else [child] if isinstance(child, str) else child) + if parsed is None: + continue + child_problem, ids = parsed + if child_problem: + problem = problem or child_problem + else: + _add_ids(found, _ID_KIND[key], ids) + continue + if isinstance(child, (Mapping, list)): + nodes.append(child) + return problem + + +def _derived_from_deps(record: object) -> tuple[str, set[tuple[str, str]]]: + """Read the canonical record. A bad shape is ``malformed`` or ``counts_disagree``.""" + + if not isinstance(record, Mapping): + return "malformed", set() + found: set[tuple[str, str]] = set() + lists: dict[str, list[str]] = {} + repeated = False + for key, kind in _DERIVED_FROM_KIND.items(): + if key not in record: + lists[key] = [] + continue + raw = record.get(key) + if not isinstance(raw, list) or any(not isinstance(item, str) for item in raw): + return "malformed", set() + if len(raw) != len(set(raw)): + repeated = True + ids = _strings(raw) + lists[key] = ids + _add_ids(found, kind, ids) + counts = record.get("counts") + if counts is None: + if any(lists.values()): + return "counts_disagree", found + return "", found + if not isinstance(counts, Mapping): + return "malformed", found + if repeated: + return "", found + for key, ids in lists.items(): + if key not in counts: + if ids: + return "counts_disagree", found + continue + expected = counts.get(key) + if not isinstance(expected, int) or isinstance(expected, bool): + return "malformed", found + if expected != len(ids): + return "counts_disagree", found + return "", found + + +def _legacy_count_problem(kind: str, row: Mapping[str, object], meta: Mapping[str, object]) -> str: + """Disagreeing legacy counts on the four producers that write them.""" + + workflow = str(meta.get("workflow") or "") + artifact_type = str(row.get("artifact_type") or "") + summary = meta.get("input_summary") + if workflow in _LEGACY_SUMMARY_KINDS or artifact_type in _LEGACY_SUMMARY_KINDS: + if isinstance(summary, Mapping): + problem = _counts_against_lists( + summary.get("counts"), + { + "sources": summary.get("source_ids"), + "memories": summary.get("memory_ids"), + "open_loops": summary.get("open_loop_ids"), + "artifacts": summary.get("artifact_ids"), + }, + ) + if problem: + return "legacy_counts" + if workflow in _LEGACY_COUNT_KINDS or artifact_type in _LEGACY_COUNT_KINDS: + problem = _counts_against_lists( + meta.get("input_counts"), + { + "sources": meta.get("source_ids"), + "memories": meta.get("memory_ids"), + "beliefs": meta.get("belief_ids"), + "open_loops": meta.get("open_loop_ids"), + "artifacts": meta.get("artifact_ids"), + }, + ) + if problem: + return "legacy_counts" + return "" + + +def _counts_against_lists(counts: object, lists: Mapping[str, object]) -> str: + if not isinstance(counts, Mapping): + return "" + present_lists: dict[str, list[object]] = {} + for key, raw in lists.items(): + if raw is None: + continue + if not isinstance(raw, list): + return "malformed" + present_lists[key] = list(raw) + if any(len(values) != len(set(map(str, values))) for values in present_lists.values()): + return "" + for key, values in present_lists.items(): + if key not in counts: + continue + expected = counts.get(key) + if isinstance(expected, int) and not isinstance(expected, bool) and expected != len(values): + return "counts_disagree" + return "" + + +def _record_present(meta: Mapping[str, object], row: Mapping[str, object]) -> bool: + if "derived_from" in meta or "input_summary" in meta or "source_refs" in meta or "input_counts" in meta: + return True + for key in ( + "source_ids", + "memory_ids", + "open_loop_ids", + "artifact_ids", + "belief_ids", + "stale_marked_memory_ids", + "source_id", + "source_artifact_id", + "artifact_id", + ): + if key in meta: + return True + consolidation = meta.get("consolidation") + if isinstance(consolidation, Mapping) and consolidation: + return True + if _nonempty_str(row.get("source_id")): + return True + return False + + +def dependencies_of(kind: object, row: Mapping[str, object]) -> frozenset[tuple[str, str]]: + """Recorded inputs as ``(kind, normalized id)``. + + Belief ids are returned as ``belief`` and resolved by :func:`settle_labels`. + A row that is not derived has no dependencies. The value fields are read + only when a metadata marker is already present. + """ + + deps, _problem = dependency_record(kind, row) + return frozenset(deps) + + +def dependency_record(kind: object, row: Mapping[str, object]) -> tuple[frozenset[tuple[str, str]], str]: + """Dependencies and a structural problem, or ``""`` when the record is sound. + + An empty record is sound. A missing record on a derived row is ``no_record``. + """ + + if not is_derived(kind, row): + return frozenset(), "" + meta = _metadata(row) + found: set[tuple[str, str]] = set() + problem = _collect_metadata_ids(meta, found) + if "derived_from" in meta: + derived_problem, derived_deps = _derived_from_deps(meta.get("derived_from")) + problem = problem or derived_problem + found |= derived_deps + if is_derived(kind, row): + found |= _value_dependencies(row) + if canon_kind(kind) == "open_loop" and _nonempty_str(row.get("source_id")): + found.add(("source", identifier(row.get("source_id")))) + floor_shape, _floor = _floor_of(row) + if floor_shape == "malformed": + problem = problem or "malformed_floor" + if not problem: + problem = _legacy_count_problem(canon_kind(kind), row, meta) + if not problem and not _record_present(meta, row) and not found: + # A weekly candidate may still be completed from its parent artifact + # inside settle_labels. The structural pass says no_record until then. + problem = "no_record" + if problem: + return frozenset(found), problem + return frozenset(found), "" + + +def _value_dependencies(row: Mapping[str, object]) -> set[tuple[str, str]]: + value = _object(row.get("value")) + found: set[tuple[str, str]] = set() + if _nonempty_str(value.get("source_id")): + found.add(("source", identifier(value.get("source_id")))) + if _nonempty_str(value.get("artifact_id")): + found.add(("artifact", identifier(value.get("artifact_id")))) + parsed = _as_string_list(value.get("cluster_member_ids")) if "cluster_member_ids" in value else None + if parsed and not parsed[0]: + _add_ids(found, "memory", parsed[1]) + rollup = value.get("rollup") + if isinstance(rollup, Mapping): + members = _as_string_list(rollup.get("member_ids")) if "member_ids" in rollup else None + if members and not members[0]: + _add_ids(found, "memory", members[1]) + return found + + +def _raised_sensitivity(current: str, inputs: Iterable[str]) -> str: + """Highest rank. A lower rank never replaces the current label.""" + + current_rank = SENSITIVITY_RANK.get(current, SENSITIVITY_RANK["unknown"]) + chosen = current + chosen_rank = current_rank + for value in inputs: + rank = SENSITIVITY_RANK.get(str(value), SENSITIVITY_RANK["unknown"]) + if rank > chosen_rank: + chosen = str(value) + chosen_rank = rank + return chosen + + +def union_floor(stored: Sequence[str], parts: Iterable[Sequence[str]]) -> tuple[str, ...]: + """Stored floor united with every part. The stored projects stay.""" + + combined = [*stored, *[item for part in parts for item in part]] + return ordered_identifiers(combined) + + +def intersect_scope(stored: Sequence[str], parent: Sequence[str]) -> tuple[str, ...]: + parent_ids = set(project_scope_identity(parent)) + return ordered_identifiers(item for item in stored if project_identifier_identity(item) in parent_ids) + + +def generation_domain(payload_domain: str, dep_domains: Iterable[str]) -> str: + """Insert rule: a restricted payload domain is kept. Otherwise the derived domain.""" + + if payload_domain in RESTRICTED_DOMAINS: + return payload_domain + return derived_domain(({"domain": item} for item in dep_domains), fallback=payload_domain) + + +def labels_raised_payload(*, cause: str, previous: Mapping[str, object], new: Mapping[str, object]) -> dict[str, object]: + """One audit payload. Labels only: no text, titles or input ids.""" + + def side(source: Mapping[str, object]) -> dict[str, object]: + scope = source.get("project_scope", ()) + floor = source.get("project_floor", ()) + return { + "domain": str(source.get("domain") or "unknown"), + "sensitivity": str(source.get("sensitivity") or "unknown"), + "project_scope": list(scope) if isinstance(scope, (list, tuple)) else [], + "project_floor": list(floor) if isinstance(floor, (list, tuple)) else [], + } + + return {"cause": cause, "previous": side(previous), "new": side(new)} + + +@dataclass(frozen=True, slots=True) +class SettledLabel: + """Effective label of one stored row.""" + + kind: str + user_id: str + stored_id: str + normalized_id: str + domain: str + sensitivity: str + project_scope: tuple[str, ...] + project_floor: tuple[str, ...] + stored_domain: str + stored_sensitivity: str + stored_scope: tuple[str, ...] + stored_floor: tuple[str, ...] + unverified: bool + reason: str | None + carries_scope: bool + derived: bool + row_class: str + + @property + def key(self) -> tuple[str, str, str]: + return (self.kind, self.user_id, self.normalized_id) + + +@dataclass(frozen=True, slots=True) +class SettleResult: + """Settled labels, one entry per stored row, in input order.""" + + rows: tuple[SettledLabel, ...] + + def by_stored(self, kind: str, stored_id: str, *, user_id: str | None = None) -> SettledLabel: + name = canon_kind(kind) + for row in self.rows: + if row.kind == name and row.stored_id == str(stored_id) and (user_id is None or row.user_id == user_id): + return row + raise KeyError((name, stored_id)) + + def derived_rows(self) -> tuple[SettledLabel, ...]: + return tuple(row for row in self.rows if row.derived) + + +def _node_label(kind: str, row: Mapping[str, object]) -> SettledLabel: + meta_floor_shape, floor = _floor_of(row) + if meta_floor_shape != "list": + floor = () + scope = stored_scope(kind, row) + domain = str(row.get("domain") or "unknown") + sensitivity = str(row.get("sensitivity") or "unknown") + return SettledLabel( + kind=kind, + user_id=str(row.get("user_id") or ""), + stored_id=str(row.get("id") or ""), + normalized_id=identifier(row.get("id")), + domain=domain, + sensitivity=sensitivity, + project_scope=scope, + project_floor=floor, + stored_domain=domain, + stored_sensitivity=sensitivity, + stored_scope=scope, + stored_floor=floor, + unverified=False, + reason=None, + carries_scope=carries_scope(kind, row), + derived=is_derived(kind, row), + row_class=row_class(kind, row), + ) + + +def _copy_scope(stored: tuple[str, ...], parents: Sequence[SettledLabel]) -> tuple[str, ...]: + live = [item for item in parents if item.carries_scope] + if not parents: + return stored + if not live: + # Every parent is a scrubbed source: keep the stored scope. + return stored + scope = stored + for item in live: + if len(project_scope_identity(item.project_scope)) == 0: + return () + scope = intersect_scope(scope, item.project_scope) + return scope + + +def _effective_scope(current: SettledLabel, deps: Sequence[SettledLabel], floor: tuple[str, ...]) -> tuple[str, ...]: + if current.row_class == "project_state": + return () + if current.row_class == "original" or not current.derived: + return current.stored_scope + informative = [item for item in deps if item.carries_scope] + if not deps: + return current.stored_scope + any_empty = any(len(project_scope_identity(item.project_scope)) == 0 for item in informative) + if current.row_class == "report": + if any_empty: + return () # a global input empties the report scope + return current.stored_scope + if current.row_class == "aggregate": + stored = current.stored_scope + single = len(project_scope_identity(stored)) == 1 + floor_inside = set(project_scope_identity(floor)).issubset(set(project_scope_identity(stored))) + if single and not any_empty and floor_inside: + return stored + return () # aggregate rule otherwise leaves the scope empty + parents = [item for item in deps if item.kind == "source"] + return _copy_scope(current.stored_scope, parents) + + +def _apply_dependencies( + current: SettledLabel, + deps: Sequence[SettledLabel], + *, + domain_fallback: str | None = None, + sensitivity_fallback: str | None = None, + scope_fallback: tuple[str, ...] | None = None, + floor_fallback: tuple[str, ...] | None = None, +) -> SettledLabel: + """One row from the effective labels of its dependencies.""" + + fallback_domain = current.domain if domain_fallback is None else domain_fallback + fallback_sensitivity = current.sensitivity if sensitivity_fallback is None else sensitivity_fallback + base = current + if scope_fallback is not None or floor_fallback is not None: + base = replace( + current, + stored_scope=current.stored_scope if scope_fallback is None else scope_fallback, + stored_floor=current.stored_floor if floor_fallback is None else floor_fallback, + domain=fallback_domain, + sensitivity=fallback_sensitivity, + ) + domain = derived_domain(({"domain": item.domain} for item in deps), fallback=fallback_domain) + if domain not in RESTRICTED_DOMAINS and fallback_domain in RESTRICTED_DOMAINS: + domain = fallback_domain + sensitivity = _raised_sensitivity(fallback_sensitivity, (item.sensitivity for item in deps)) + informative = [item for item in deps if item.carries_scope] + floor = union_floor( + base.stored_floor, + [*(item.project_scope for item in informative), *(item.project_floor for item in informative)], + ) + if base.row_class == "project_state": + floor = () + scope = _effective_scope(base, deps, floor) + return replace(base, domain=str(domain), sensitivity=sensitivity, project_scope=scope, project_floor=floor) + + +def _changed(before: SettledLabel, after: SettledLabel) -> bool: + return ( + before.domain != after.domain + or before.sensitivity != after.sensitivity + or project_scope_identity(before.project_scope) != project_scope_identity(after.project_scope) + or project_scope_identity(before.project_floor) != project_scope_identity(after.project_floor) + ) + + +def _weekly_parent_deps( + labels: Mapping[tuple[str, str, str], SettledLabel], + own: Mapping[tuple[str, str, str], set[tuple[str, str, str]]], + nodes: Sequence[tuple[SettledLabel, Mapping[str, object]]], +) -> None: + """Old weekly candidates take the input lists of the artifact that names them.""" + + artifacts = [ + (label, _metadata(row)) + for label, row in nodes + if label.kind == "artifact" and isinstance(_metadata(row).get("input_summary"), Mapping) + ] + for label, meta in artifacts: + for candidate in _strings(meta.get("candidate_memory_ids")): + candidate_key = ("memory", label.user_id, candidate) + candidate_label = labels.get(candidate_key) + if candidate_label is None or candidate_label.row_class != "aggregate": + continue + if _metadata_discovered(nodes, candidate_key) != "vnext_weekly_synthesis": + continue + own.setdefault(candidate_key, set()).update(own.get(label.key, set())) + + +def _metadata_discovered( + nodes: Sequence[tuple[SettledLabel, Mapping[str, object]]], + key: tuple[str, str, str], +) -> object: + for label, row in nodes: + if label.key == key: + return _metadata(row).get("discovered_by") + return None + + +def settle_labels( + nodes: Sequence[Mapping[str, object]], + *, + unavailable_kinds: Iterable[str] = (), + on_cycle: str = "raise", + max_hops: int | None = None, + max_nodes: int | None = None, +) -> SettleResult: + """Settle every derived row in ``nodes``. + + ``nodes`` are mappings with ``kind`` (or a table name) plus the row fields. + Originals are fixed inputs. A cycle that does not settle raises + ``DerivedDomainRepairError`` unless ``on_cycle`` is ``unverified``. + ``max_hops`` and ``max_nodes`` bound a read. The repair leaves them unset. + """ + + if on_cycle not in {"raise", "unverified"}: + raise ValueError("on_cycle must be raise or unverified") + unavailable = {canon_kind(kind) for kind in unavailable_kinds} + prepared: list[tuple[SettledLabel, Mapping[str, object]]] = [] + for node in nodes: + kind = canon_kind(node.get("kind") if node.get("kind") is not None else infer_kind(node)) + row = node + if "row" in node and isinstance(node.get("row"), Mapping): + row = node["row"] # type: ignore[assignment] + prepared.append((_node_label(kind, row), row)) + + labels: dict[tuple[str, str, str], SettledLabel] = {} + order: list[SettledLabel] = [] + for label, _row in prepared: + order.append(label) + labels[label.key] = label + + own: dict[tuple[str, str, str], set[tuple[str, str, str]]] = {} + problems: dict[tuple[str, str, str], str] = {} + for label, row in prepared: + if not label.derived: + continue + deps, problem = dependency_record(label.kind, row) + if problem == "no_record" and label.row_class == "aggregate": + # Filled from the parent artifact below when one names this row. + problem = "" + problems[label.key] = "no_record" + elif problem: + problems[label.key] = problem + keyed = {(kind, label.user_id, row_id) for kind, row_id in deps} + own[label.key] = keyed + _weekly_parent_deps(labels, own, prepared) + for key, reason in list(problems.items()): + if reason == "no_record" and own.get(key): + del problems[key] + elif reason == "no_record" and not own.get(key): + pass + elif reason == "" and not own.get(key): + pass + + def resolve_belief(ref: tuple[str, str, str]) -> tuple[str, str, str]: + if ref[0] != "belief": + return ref + belief = labels.get(ref) + if belief is None: + return ref + for label, row in prepared: + if label.key == ref: + memory_id = row.get("memory_id") + if _nonempty_str(memory_id) or memory_id not in (None, ""): + return ("memory", ref[1], identifier(memory_id)) + return ref + + resolved: dict[tuple[str, str, str], set[tuple[str, str, str]]] = {} + for key, deps in own.items(): + resolved[key] = {resolve_belief(ref) for ref in deps} + + for key, deps in resolved.items(): + if key in problems and problems[key] not in {"", "no_record"}: + continue + for ref in deps: + if ref[0] in unavailable: + problems[key] = "missing_table" + break + if ref not in labels: + problems[key] = "missing_dependency" + break + + if max_hops is not None or max_nodes is not None: + _mark_bounds(labels, resolved, problems, max_hops=max_hops, max_nodes=max_nodes) + + _spread_unverified(resolved, problems) + + inputs = {key: set(resolved.get(key, set())) for key, label in labels.items() if label.derived and key not in problems} + for key in problems: + inputs.pop(key, None) + + cycle_keys: set[tuple[str, str, str]] = set() + if inputs: + try: + _iterate(labels, inputs, on_cycle=on_cycle, cycle_keys=cycle_keys, problems=problems) + except DerivedDomainRepairError: + if on_cycle != "raise": + raise + raise + + for key in cycle_keys: + problems[key] = "cycle_unsettled" + _spread_unverified(resolved, problems) + + published: list[SettledLabel] = [] + for label, row in prepared: + current = labels[label.key] + if not label.derived: + published.append(label) + continue + if label.key in problems: + published.append( + replace( + label, + unverified=True, + reason=problems[label.key], + carries_scope=False, + ) + ) + continue + settled = labels[label.key] + same_stored_row = ( + settled.stored_id == label.stored_id + and settled.stored_domain == label.stored_domain + and settled.stored_sensitivity == label.stored_sensitivity + and settled.stored_scope == label.stored_scope + and settled.stored_floor == label.stored_floor + ) + if same_stored_row: + published.append(replace(settled, unverified=False, reason=None)) + continue + deps = [labels[ref] for ref in sorted(resolved.get(label.key, set())) if ref in labels] + recomputed = _apply_dependencies( + settled, + deps, + domain_fallback=label.stored_domain, + sensitivity_fallback=label.stored_sensitivity, + scope_fallback=label.stored_scope, + floor_fallback=label.stored_floor, + ) + published.append( + replace( + recomputed, + stored_id=label.stored_id, + stored_domain=label.stored_domain, + stored_sensitivity=label.stored_sensitivity, + stored_scope=label.stored_scope, + stored_floor=label.stored_floor, + unverified=False, + reason=None, + ) + ) + return SettleResult(rows=tuple(published)) + + +def _spread_unverified( + resolved: Mapping[tuple[str, str, str], set[tuple[str, str, str]]], + problems: dict[tuple[str, str, str], str], +) -> None: + """A row that reads an unverified row is unverified. The reason is contagious.""" + + changed = True + while changed: + changed = False + for key, deps in resolved.items(): + if key in problems: + continue + if any(ref in problems for ref in deps): + problems[key] = "dependency_unverified" + changed = True + + +def _mark_bounds( + labels: Mapping[tuple[str, str, str], SettledLabel], + resolved: Mapping[tuple[str, str, str], set[tuple[str, str, str]]], + problems: dict[tuple[str, str, str], str], + *, + max_hops: int | None, + max_nodes: int | None, +) -> None: + """Mark a derived row unverified when the walk from that row passes a bound.""" + + for origin, label in labels.items(): + if not label.derived or origin in problems: + continue + pending: deque[tuple[tuple[str, str, str], int]] = deque([(origin, 0)]) + seen: set[tuple[str, str, str]] = set() + while pending: + key, depth = pending.popleft() + if key in seen: + continue + if max_nodes is not None and len(seen) >= max_nodes: + problems.setdefault(origin, "bound_exceeded") + break + if max_hops is not None and depth > max_hops: + problems.setdefault(origin, "bound_exceeded") + break + seen.add(key) + for ref in resolved.get(key, set()): + pending.append((ref, depth + 1)) + + +def _iterate( + labels: dict[tuple[str, str, str], SettledLabel], + inputs: dict[tuple[str, str, str], set[tuple[str, str, str]]], + *, + on_cycle: str, + cycle_keys: set[tuple[str, str, str]], + problems: dict[tuple[str, str, str], str], +) -> None: + dependants: dict[tuple[str, str, str], set[tuple[str, str, str]]] = {} + for key, refs in inputs.items(): + for ref in refs: + dependants.setdefault(ref, set()).add(key) + for component in _input_groups(inputs): + members = set(component) + pending: deque[tuple[str, str, str]] = deque(component) + queued = set(component) + remaining_changes = len(component) * (len(RESTRICTED_DOMAINS) + 1) + changes: dict[tuple[str, str, str], int] = {} + while pending: + key = pending.popleft() + queued.remove(key) + current = labels[key] + deps = [labels[ref] for ref in sorted(inputs[key]) if ref in labels] + updated = _apply_dependencies( + current, + deps, + domain_fallback=current.stored_domain, + sensitivity_fallback=current.stored_sensitivity, + scope_fallback=current.stored_scope, + floor_fallback=current.stored_floor, + ) + if not _changed(current, updated): + continue + remaining_changes -= 1 + changes[key] = changes.get(key, 0) + 1 + if remaining_changes < 0: + if on_cycle == "unverified": + for member in members: + cycle_keys.add(member) + problems[member] = "cycle_unsettled" + return + shown = ", ".join(f"{item[0]} {item[2]}" for item in sorted(changes)[:5]) + raise DerivedDomainRepairError( + "derived label repair did not settle: derived rows record each other as inputs in a cycle, " + f"so their labels kept changing (rows: {shown})." + ) + labels[key] = updated + for dependant in sorted(dependants.get(key, set()) & members): + if dependant not in queued: + pending.append(dependant) + queued.add(dependant) + + +def scope_is_global(scope: object) -> bool: + """True when a scope holds no Alice project id.""" + + return is_global_scope(scope) + + +__all__ = [ + "DERIVED_ARTIFACT_TYPES", + "DERIVED_WORKFLOWS", + "HOP_BOUND", + "LabelPropagationTooLarge", + "MARKER_KEYS", + "NODE_BOUND", + "PROPAGATION_BOUND", + "SENSITIVITY_RANK", + "SettleResult", + "SettledLabel", + "V2_ID_KEYS", + "canon_kind", + "carries_scope", + "dependencies_of", + "dependency_record", + "generation_domain", + "group_scope", + "identifier", + "infer_kind", + "intersect_scope", + "is_derived", + "labels_raised_payload", + "ordered_identifiers", + "row_class", + "scope_is_global", + "settle_labels", + "stored_scope", + "union_floor", +] diff --git a/apps/api/src/alicebot_api/vnext_project_scope.py b/apps/api/src/alicebot_api/vnext_project_scope.py index 16132ba42..4e6faca2c 100644 --- a/apps/api/src/alicebot_api/vnext_project_scope.py +++ b/apps/api/src/alicebot_api/vnext_project_scope.py @@ -361,12 +361,86 @@ def refuse_global_marker(scope: object, *, where: str) -> None: raise ValueError(f"{where} takes explicit project names only and does not accept the global marker") -def project_scopes_overlap(resource_scope: object, requested_scope: object) -> bool: +def project_floor(resource: Mapping[str, object] | None) -> tuple[str, ...]: + """Return the stored ``project_floor`` list, or ``()`` when the key is absent. + + A present value that is not a list of strings is not a floor. Callers that + must refuse that shape ask :func:`project_floor_shape` first. This helper + returns ``()`` for it so a selection door does not treat a bad value as names. + """ + + shape, values = project_floor_shape(resource) + if shape != "list": + return () + return values + + +def project_floor_shape(resource: Mapping[str, object] | None) -> tuple[str, tuple[str, ...]]: + """Classify ``project_floor`` as ``absent``, ``list`` or ``malformed``. + + ``list`` carries the stored spellings, ordered by identity, with no duplicates. + """ + + if resource is None: + return "absent", () + containers: list[Mapping[str, object]] = [resource] + metadata = resource.get("metadata_json") + if isinstance(metadata, Mapping): + containers.append(metadata) + seen = False + raw: object = None + for container in containers: + if "project_floor" in container: + seen = True + raw = container.get("project_floor") + break + if not seen: + return "absent", () + if not isinstance(raw, Sequence) or isinstance(raw, (str, bytes, bytearray)): + return "malformed", () + if any(not isinstance(item, str) for item in raw): + return "malformed", () + return "list", _ordered_scope(raw) + + +def project_floor_within(floor: object, binding: object) -> bool: + """True when every project in ``floor`` is inside ``binding``, by identity. + + An empty floor is inside every binding. A free-form name is inside only when + the binding names that same identity. + """ + + return set(project_scope_identity(floor)).issubset(set(project_scope_identity(binding))) + + +def _ordered_scope(value: object) -> tuple[str, ...]: + """Stored spellings, first one kept, ordered by identity.""" + + chosen: dict[str, str] = {} + for item in normalize_project_scope(value): + identity = project_identifier_identity(item) + if identity and identity not in chosen: + chosen[identity] = item + return tuple(chosen[identity] for identity in sorted(chosen)) + + +def project_scopes_overlap( + resource_scope: object, + requested_scope: object, + *, + floor: object = (), +) -> bool: """Does the resource's scope meet the requested tuple? The tuple may hold the reserved marker. The marker asks for a resource whose scope holds no Alice project id, and is never compared with a stored value. Every other entry is an identifier compared by identity. + + ``floor`` is the row's project floor. It is consulted only on the global + branch: a row whose scope holds no Alice project id is in a view that asks + for global rows only when every Alice project id in the floor is in the + view. Free-form names in the floor do not hide the row. An empty floor keeps + the previous answer. """ requested = set(project_scope_identity(requested_scope)) @@ -376,7 +450,9 @@ def project_scopes_overlap(resource_scope: object, requested_scope: object) -> b if GLOBAL_PROJECT_MARKER in requested: requested.discard(GLOBAL_PROJECT_MARKER) if not any(is_alice_project_id(item) for item in resource): - return True + floor_ids = {item for item in project_scope_identity(floor) if is_alice_project_id(item)} + if floor_ids.issubset(requested): + return True return bool(requested.intersection(resource)) @@ -391,6 +467,9 @@ def project_scopes_overlap(resource_scope: object, requested_scope: object) -> b "normalize_project_identifier", "normalize_project_scope", "ProjectScopeResolution", + "project_floor", + "project_floor_shape", + "project_floor_within", "project_identifier_identity", "project_scope_identity", "project_scopes_overlap", diff --git a/tests/unit/test_derived_labels_kernel.py b/tests/unit/test_derived_labels_kernel.py new file mode 100644 index 000000000..3d1528b65 --- /dev/null +++ b/tests/unit/test_derived_labels_kernel.py @@ -0,0 +1,819 @@ +"""The pure derived-label kernel. No store and no product caller.""" + +from __future__ import annotations + +import ast +from pathlib import Path +from uuid import UUID + +import pytest + +from alicebot_api import vnext_brain as brain +from alicebot_api import vnext_consolidation as consolidation +from alicebot_api.vnext_agent_control import RESTRICTED_DOMAINS, AgentIdentity, evaluate_agent_policy +from alicebot_api.vnext_derived_domain_backfill import DerivedDomainRepairError, _ID_KEYS +from alicebot_api.vnext_derived_labels import ( + SENSITIVITY_RANK, + V2_ID_KEYS, + dependencies_of, + generation_domain, + group_scope, + is_derived, + labels_raised_payload, + row_class, + settle_labels, +) +from alicebot_api.vnext_project_scope import GLOBAL_PROJECT_MARKER, project_floor_within, project_scopes_overlap + +ALPHA = "prj_" + "a" * 16 +BETA = "prj_" + "b" * 16 +USER = "user-1" +SOURCE_UUID = "550e8400-e29b-41d4-a716-446655440000" + + +def _row(kind: str, row_id: str, **fields: object) -> dict[str, object]: + row: dict[str, object] = { + "kind": kind, + "id": row_id, + "user_id": USER, + "domain": "unknown", + "sensitivity": "unknown", + "metadata_json": {}, + } + row.update(fields) + return row + + +def _meta(row: dict[str, object], **fields: object) -> dict[str, object]: + metadata = dict(row["metadata_json"]) if isinstance(row["metadata_json"], dict) else {} + metadata.update(fields) + row["metadata_json"] = metadata + return row + + +def _brief(row_id: str, *, sources: list[str] | None = None, memories: list[str] | None = None, + loops: list[str] | None = None, artifacts: list[str] | None = None, scope: list[str] | None = None, + domain: str = "unknown", sensitivity: str = "public") -> dict[str, object]: + sources = sources or [] + memories = memories or [] + loops = loops or [] + artifacts = artifacts or [] + summary = { + "source_ids": sources, + "memory_ids": memories, + "open_loop_ids": loops, + "artifact_ids": artifacts, + "counts": { + "sources": len(sources), + "memories": len(memories), + "open_loops": len(loops), + "artifacts": len(artifacts), + }, + } + return _row( + "artifact", + row_id, + artifact_type="daily_brief", + domain=domain, + sensitivity=sensitivity, + metadata_json={ + "workflow": "daily_brief", + "input_summary": summary, + "project_scope": list(scope or []), + "source_refs": [f"source:{item}" for item in sources], + }, + ) + + +def _source(row_id: str, *, domain: str = "unknown", sensitivity: str = "public", scope: list[str] | None = None, + scrubbed: bool = False) -> dict[str, object]: + metadata: dict[str, object] = {} + if scope is not None: + metadata["project_scope"] = list(scope) + if scrubbed: + metadata["scrubbed"] = True + return _row("source", row_id, domain=domain, sensitivity=sensitivity, metadata_json=metadata) + + +def _memory(row_id: str, **fields: object) -> dict[str, object]: + row = _row("memory", row_id, memory_key=row_id, canonical_text=row_id) + row.update(fields) + return row + + +def test_a_chain_settles_in_one_pass(monkeypatch: pytest.MonkeyPatch) -> None: + """A chain of any length is labelled from the leaf inward, each row once.""" + + calls: list[str] = [] + real = __import__("alicebot_api.vnext_derived_labels", fromlist=["_apply_dependencies"])._apply_dependencies + + def wrapped(*args: object, **kwargs: object) -> object: + current = args[0] + calls.append(str(getattr(current, "stored_id"))) + return real(*args, **kwargs) + + monkeypatch.setattr("alicebot_api.vnext_derived_labels._apply_dependencies", wrapped) + for size in (2, 5, 40): + calls.clear() + source = _source("s", domain="health", sensitivity="confidential", scope=[ALPHA]) + rows = [source] + leaf = _brief("n0", sources=["s"], scope=[ALPHA]) + rows.append(leaf) + previous = "n0" + for index in range(1, size): + rows.append(_brief(f"n{index}", artifacts=[previous], scope=[ALPHA])) + previous = f"n{index}" + settled = settle_labels(rows) + derived_ids = [row.stored_id for row in settled.derived_rows()] + assert calls == derived_ids + for row in settled.derived_rows(): + assert row.domain == "health" + assert row.sensitivity == "confidential" + assert row.unverified is False + + +def test_a_diamond_settles_once(monkeypatch: pytest.MonkeyPatch) -> None: + calls: list[str] = [] + module = __import__("alicebot_api.vnext_derived_labels", fromlist=["_apply_dependencies"]) + real = module._apply_dependencies + + def wrapped(*args: object, **kwargs: object) -> object: + calls.append(str(getattr(args[0], "stored_id"))) + return real(*args, **kwargs) + + monkeypatch.setattr(module, "_apply_dependencies", wrapped) + rows = [ + _source("s", domain="legal", sensitivity="private", scope=[ALPHA]), + _brief("a", sources=["s"], scope=[ALPHA]), + _brief("b", sources=["s"], scope=[ALPHA]), + _brief("c", artifacts=["a", "b"], scope=[ALPHA]), + ] + settled = settle_labels(rows) + assert calls.count("c") == 1 + assert settled.by_stored("artifact", "c").domain == "legal" + + +def test_a_cycle_that_does_not_settle_is_refused() -> None: + ring = [] + names = ["r0", "r1", "r2"] + for index, name in enumerate(names): + ring.append( + _brief( + name, + artifacts=[names[(index + 1) % 3]], + domain="health" if index % 2 == 0 else "legal", + ) + ) + with pytest.raises(DerivedDomainRepairError, match="did not settle"): + settle_labels(ring) + + +def test_unverified_is_contagious_through_every_level() -> None: + rows = [ + _brief("b", artifacts=["missing-artifact"]), + _brief("a", artifacts=["b"]), + _brief("top", artifacts=["a"]), + ] + settled = settle_labels(rows) + assert settled.by_stored("artifact", "b").reason == "missing_dependency" + assert settled.by_stored("artifact", "a").reason == "dependency_unverified" + assert settled.by_stored("artifact", "top").reason == "dependency_unverified" + + +def test_every_unverified_case_is_unverified() -> None: + belief_id = "b1" + memory_id = "m1" + cases = { + "missing id": ( + [_brief("r", memories=["does-not-exist"])], + "r", + "missing_dependency", + ), + "kind with no table": ( + [_brief("r", artifacts=["ghost"])], + "r", + "missing_table", + ), + "derived marker with no record": ( + [_row("artifact", "r", artifact_type="daily_brief", metadata_json={"workflow": "daily_brief"})], + "r", + "no_record", + ), + "counts disagree": ( + [ + _row( + "artifact", + "r", + artifact_type="daily_brief", + metadata_json={ + "workflow": "daily_brief", + "derived_from": { + "v": 1, + "sources": ["s"], + "memories": [], + "open_loops": [], + "artifacts": [], + "beliefs": [], + "counts": {"sources": 2, "memories": 0, "open_loops": 0, "artifacts": 0, "beliefs": 0}, + }, + }, + ) + ], + "r", + "counts_disagree", + ), + "malformed list": ( + [ + _row( + "artifact", + "r", + artifact_type="daily_brief", + metadata_json={"workflow": "daily_brief", "derived_from": {"sources": [1]}}, + ) + ], + "r", + "malformed", + ), + "malformed floor": ( + [_meta(_brief("r"), **{"project_floor": "alpha"})], + "r", + "malformed_floor", + ), + "legacy counts": ( + [ + _row( + "artifact", + "r", + artifact_type="daily_brief", + metadata_json={ + "workflow": "daily_brief", + "input_summary": { + "source_ids": ["s"], + "memory_ids": [], + "open_loop_ids": [], + "artifact_ids": [], + "counts": {"sources": 3, "memories": 0, "open_loops": 0, "artifacts": 0}, + }, + }, + ) + ], + "r", + "legacy_counts", + ), + "bound exceeded": ( + [ + _brief("c", artifacts=["b"]), + _brief("b", artifacts=["a"]), + _brief("a", sources=["s"]), + _source("s", domain="health"), + ], + "c", + "bound_exceeded", + ), + "dependency unverified": ( + [_brief("child", memories=["gone"]), _brief("parent", artifacts=["child"])], + "parent", + "dependency_unverified", + ), + } + for name, (rows, row_id, reason) in cases.items(): + kwargs = {"unavailable_kinds": ["artifact"]} if name == "kind with no table" else {} + if name == "bound exceeded": + kwargs = {"max_hops": 1} + settled = settle_labels(rows, **kwargs) + got = settled.by_stored("artifact", row_id) + assert got.unverified is True, name + assert got.reason == reason, name + + empty = _brief("empty") + assert settle_labels([empty]).by_stored("artifact", "empty").unverified is False + + redacted = _memory("red", metadata_json={"redacted": True, "source_id": "s", "workflow": "daily_brief"}) + assert is_derived("memory", redacted) is False + assert settle_labels([redacted]).by_stored("memory", "red").unverified is False + + project = _row("project", "p", metadata_json={}, current_state="notes") + assert is_derived("project", project) is False + assert row_class("project", project) == "original" + + scrubbed = _source("s", domain="health", sensitivity="confidential", scope=[ALPHA], scrubbed=True) + copy = _memory( + "copy", + domain="unknown", + sensitivity="public", + metadata_json={"source_id": "s", "project_scope": [ALPHA]}, + ) + kept = settle_labels([scrubbed, copy]).by_stored("memory", "copy") + assert kept.unverified is False + assert kept.project_scope == (ALPHA,) + assert kept.project_floor == () + assert kept.domain == "health" + assert kept.sensitivity == "confidential" + + repeated = _row( + "artifact", + "rep", + artifact_type="connection_report", + metadata_json={ + "workflow": "connection_finder", + "source_ids": ["s", "s"], + "memory_ids": [], + "input_counts": {"sources": 1, "memories": 0}, + }, + ) + assert settle_labels([_source("s"), repeated]).by_stored("artifact", "rep").reason != "legacy_counts" + + belief = _row("belief", belief_id, memory_id=memory_id) + backing = _memory(memory_id, domain="health", sensitivity="private") + report = _row( + "artifact", + "belief-report", + artifact_type="contradiction_report", + metadata_json={ + "workflow": "contradiction_finder", + "source_ids": [], + "memory_ids": [], + "belief_ids": [belief_id], + "input_counts": {"sources": 0, "memories": 0, "beliefs": 1}, + }, + ) + followed = settle_labels([belief, backing, report]).by_stored("artifact", "belief-report") + assert followed.unverified is False + assert followed.domain == "health" + assert followed.sensitivity == "private" + + +def test_sensitivity_is_raise_only_and_unknown_never_swaps_with_internal() -> None: + source = _source("s", domain="unknown", sensitivity="confidential") + raised = _memory("m", sensitivity="public", metadata_json={"source_id": "s"}) + assert settle_labels([source, raised]).by_stored("memory", "m").sensitivity == "confidential" + + high = _source("s2", domain="unknown", sensitivity="public") + kept = _memory("m2", sensitivity="confidential", metadata_json={"source_id": "s2"}) + assert settle_labels([high, kept]).by_stored("memory", "m2").sensitivity == "confidential" + + internal = _source("s3", sensitivity="internal") + unknown = _memory("m3", sensitivity="unknown", metadata_json={"source_id": "s3"}) + assert settle_labels([internal, unknown]).by_stored("memory", "m3").sensitivity == "unknown" + + unknown_source = _source("s4", sensitivity="unknown") + internal_copy = _memory("m4", sensitivity="internal", metadata_json={"source_id": "s4"}) + assert settle_labels([unknown_source, internal_copy]).by_stored("memory", "m4").sensitivity == "internal" + + +def test_domain_follows_derived_domain_and_never_becomes_unrestricted() -> None: + rows = [ + _source("h1", domain="health"), + _source("h2", domain="health"), + _source("l", domain="legal"), + _brief("r", sources=["h1", "h2", "l"]), + ] + assert settle_labels(rows).by_stored("artifact", "r").domain == "health" + tie = [ + _source("h", domain="health"), + _source("l2", domain="legal"), + _brief("tie", sources=["h", "l2"]), + ] + assert settle_labels(tie).by_stored("artifact", "tie").domain == "health" + kept = [ + _source("open", domain="project"), + _brief("kept", sources=["open"], domain="health"), + ] + assert settle_labels(kept).by_stored("artifact", "kept").domain == "health" + moved = [ + _source("fin", domain="financial"), + _source("fin2", domain="financial"), + _brief("moved", sources=["fin", "fin2"], domain="health"), + ] + assert settle_labels(moved).by_stored("artifact", "moved").domain == "financial" + assert generation_domain("health", ["financial"]) == "health" + assert generation_domain("unknown", ["financial", "financial", "legal"]) == "financial" + + +def test_floor_is_the_union_and_never_shrinks() -> None: + rows = [ + _source("a", scope=[ALPHA]), + _source("b", scope=[BETA]), + _meta(_brief("r", sources=["a", "b"], scope=[ALPHA]), **{"project_floor": [ALPHA]}), + ] + floor = settle_labels(rows).by_stored("artifact", "r").project_floor + assert set(floor) == {ALPHA, BETA} + + shrink = [ + _source("only", scope=[ALPHA]), + _meta(_brief("kept", sources=["only"], scope=[ALPHA]), **{"project_floor": [ALPHA, BETA]}), + ] + kept = settle_labels(shrink).by_stored("artifact", "kept").project_floor + assert set(kept) == {ALPHA, BETA} + + +def test_scope_rules_per_row_class() -> None: + global_memory = _memory("g", metadata_json={}) + alpha_source = _source("a", scope=[ALPHA], domain="health") + report = _brief("report", sources=["a"], memories=["g"], scope=[ALPHA]) + stored = settle_labels([global_memory, alpha_source, report]).by_stored("artifact", "report") + assert stored.project_scope == () + assert stored.project_floor == (ALPHA,) + + both = _brief("both", sources=["a"], scope=[ALPHA, BETA]) + beta = _source("beta", scope=[BETA]) + # a second source so the report is not global + multi = settle_labels([alpha_source, beta, _brief("multi", sources=["a", "beta"], scope=[ALPHA, BETA])]).by_stored( + "artifact", "multi" + ) + assert multi.project_scope == (ALPHA, BETA) + + card = _memory( + "card", + metadata_json={ + "candidate_kind": "memory_rollup", + "project_scope": [ALPHA, BETA], + "consolidation": {"cluster_member_ids": ["m1", "m2"]}, + }, + ) + members = [ + _memory("m1", metadata_json={"project_scope": [ALPHA, BETA]}), + _memory("m2", metadata_json={"project_scope": [ALPHA, BETA]}), + ] + aggregate = settle_labels([*members, card]).by_stored("memory", "card") + assert aggregate.row_class == "aggregate" + assert aggregate.project_scope == () + assert set(aggregate.project_floor) == {ALPHA, BETA} + + single = _memory( + "one", + metadata_json={ + "candidate_kind": "memory_rollup", + "project_scope": [ALPHA], + "project_floor": [ALPHA], + "consolidation": {"cluster_member_ids": ["only"]}, + }, + ) + only = _memory("only", metadata_json={"project_scope": [ALPHA]}) + kept = settle_labels([only, single]).by_stored("memory", "one") + assert kept.project_scope == (ALPHA,) + + copy = _memory("copy", metadata_json={"source_id": "a", "project_scope": [ALPHA, BETA]}) + narrowed = settle_labels([alpha_source, copy]).by_stored("memory", "copy") + assert narrowed.project_scope == (ALPHA,) + + global_source = _source("glob") + emptied = _memory("emptied", metadata_json={"source_id": "glob", "project_scope": [ALPHA]}) + assert settle_labels([global_source, emptied]).by_stored("memory", "emptied").project_scope == () + + promoted = _memory( + "promoted", + metadata_json={"source_artifact_id": "report", "project_scope": [ALPHA]}, + ) + promoted_row = settle_labels([global_memory, alpha_source, report, promoted]).by_stored("memory", "promoted") + assert promoted_row.project_scope == () + assert ALPHA in promoted_row.project_floor + + assert both["id"] == "both" + + +def test_a_forged_marker_in_value_makes_no_row_derived() -> None: + forged = _memory("forged", value={"kind": "promoted_artifact", "artifact_id": "secret", "source_id": "s"}) + assert is_derived("memory", forged) is False + assert dependencies_of("memory", forged) == frozenset() + settled = settle_labels([forged, _source("s", domain="health", sensitivity="sacred")]) + row = settled.by_stored("memory", "forged") + assert row.derived is False + assert row.domain == "unknown" + assert row.sensitivity == "unknown" + + +def test_a_dependency_is_found_in_every_recorded_field_of_every_derived_kind() -> None: + source = "11111111-1111-1111-1111-111111111111" + memory = "22222222-2222-2222-2222-222222222222" + loop = "33333333-3333-3333-3333-333333333333" + artifact = "44444444-4444-4444-4444-444444444444" + samples = [ + _brief("daily", sources=[source], memories=[memory], loops=[loop], artifacts=[artifact]), + _row( + "artifact", + "connection", + artifact_type="connection_report", + metadata_json={ + "workflow": "connection_finder", + "source_ids": [source], + "memory_ids": [memory], + "source_refs": [f"source:{source}"], + "input_counts": {"sources": 1, "memories": 1}, + }, + ), + _row( + "artifact", + "contradiction", + artifact_type="contradiction_report", + metadata_json={ + "workflow": "contradiction_finder", + "source_ids": [source], + "memory_ids": [memory], + "belief_ids": ["belief-1"], + "source_refs": [f"memory:{memory}"], + "input_counts": {"sources": 1, "memories": 1, "beliefs": 1}, + }, + ), + _row( + "artifact", + "consolidation", + artifact_type="memory_consolidation", + metadata_json={ + "workflow": "memory_consolidation", + "consolidation": {"cluster_member_ids": [memory], "member_snapshots": [{"id": memory}]}, + "source_refs": [f"source:{source}"], + }, + ), + _row( + "artifact", + "project", + artifact_type="project_update", + metadata_json={"workflow": "project_auto_update", "source_ids": [source], "memory_ids": [memory]}, + ), + _row( + "artifact", + "stale", + metadata_json={"workflow": "staleness_sweep", "stale_marked_memory_ids": [memory]}, + ), + _row( + "artifact", + "loops", + metadata_json={"workflow": "open_loop_review", "open_loop_ids": [loop], "source_refs": [f"source:{source}"]}, + ), + _row( + "artifact", + "agent", + metadata_json={"connector_name": "agent_output", "source_id": source, "source_refs": [f"source:{source}"]}, + ), + _memory( + "weekly", + metadata_json={"discovered_by": "vnext_weekly_synthesis", "input_summary": {"memory_ids": [memory], "source_ids": [], "open_loop_ids": [], "artifact_ids": [], "counts": {"memories": 1, "sources": 0, "open_loops": 0, "artifacts": 0}}}, + ), + _memory( + "rollup", + metadata_json={"candidate_kind": "memory_rollup", "consolidation": {"cluster_member_ids": [memory]}}, + value={"rollup": {"member_ids": [memory]}}, + ), + _memory("promoted", metadata_json={"source_artifact_id": artifact}, value={"artifact_id": artifact}), + _memory("extracted", metadata_json={"source_id": source, "extraction_rule": "sentence"}), + _row("open_loop", "found", metadata_json={"discovered_by": "vnext_daily_brief", "source_id": source}, source_id=source), + _row( + "project", + "state", + current_state="line", + metadata_json={"derived_from": {"memories": [memory], "artifacts": [artifact], "sources": [], "open_loops": [], "beliefs": [], "counts": {"memories": 1, "artifacts": 1, "sources": 0, "open_loops": 0, "beliefs": 0}}}, + ), + ] + expected = { + "daily": {("source", source), ("memory", memory), ("open_loop", loop), ("artifact", artifact)}, + "connection": {("source", source), ("memory", memory)}, + "contradiction": {("source", source), ("memory", memory), ("belief", "belief-1")}, + "consolidation": {("memory", memory), ("source", source)}, + "project": {("source", source), ("memory", memory)}, + "stale": {("memory", memory)}, + "loops": {("open_loop", loop), ("source", source)}, + "agent": {("source", source)}, + "weekly": {("memory", memory)}, + "rollup": {("memory", memory)}, + "promoted": {("artifact", artifact)}, + "extracted": {("source", source)}, + "found": {("source", source)}, + "state": {("memory", memory), ("artifact", artifact)}, + } + for sample in samples: + assert is_derived(sample["kind"], sample) is True + assert set(dependencies_of(sample["kind"], sample)) == expected[str(sample["id"])] + + +def test_a_ref_is_read_in_every_spelling_and_a_belief_resolves_through_its_memory() -> None: + canonical = str(UUID(SOURCE_UUID)) + spellings = [ + SOURCE_UUID.upper(), + SOURCE_UUID.replace("-", ""), + "{" + SOURCE_UUID + "}", + "urn:uuid:" + SOURCE_UUID, + "source:" + SOURCE_UUID.upper(), + ] + for spelling in spellings: + row = _memory("m", metadata_json={"source_id": spelling}) + assert dependencies_of("memory", row) == frozenset({("source", canonical)}) + refs = _brief("r", sources=[]) + _meta(refs, **{"source_refs": ["source:" + SOURCE_UUID.upper(), "memory:" + SOURCE_UUID]}) + found = dependencies_of("artifact", refs) + assert ("source", canonical) in found + assert ("memory", canonical) in found + + +def test_dependency_keys_cover_every_producer() -> None: + """Every id key a producer writes as a dict key is classified. + + A new key fails this test until it is either read as a dependency or named + in ``_NOT_A_DEPENDENCY``. The v2 keys are a subset of the keys v3 reads. + """ + + assert set(_ID_KEYS) <= V2_ID_KEYS | {"source_refs"} + root = Path(__file__).resolve().parents[2] / "apps" / "api" / "src" / "alicebot_api" + producers = [ + "vnext_brain.py", + "vnext_connections.py", + "vnext_contradictions.py", + "vnext_consolidation.py", + "vnext_rollups.py", + "vnext_projects.py", + "vnext_scheduler.py", + "vnext_queue.py", + "vnext_connectors.py", + "vnext_capture.py", + ] + found: set[str] = set() + for name in producers: + tree = ast.parse((root / name).read_text(encoding="utf-8")) + for node in ast.walk(tree): + if not isinstance(node, ast.Dict): + continue + for key in node.keys: + if isinstance(key, ast.Constant) and isinstance(key.value, str): + text = key.value + if text.endswith("_id") or text.endswith("_ids") or text in {"source_refs", "source_ref"}: + found.add(text) + unknown = sorted(found - _DEPENDENCY_KEYS - _NOT_A_DEPENDENCY) + assert unknown == [] + + +_DEPENDENCY_KEYS = frozenset( + { + "artifact_id", + "artifact_ids", + "belief_ids", + "cluster_member_ids", + "cluster_membership", + "member_ids", + "memory_id", + "memory_ids", + "open_loop_ids", + "source_artifact_id", + "source_id", + "source_ids", + "source_refs", + "stale_marked_memory_ids", + "candidate_memory_ids", + } +) +_NOT_A_DEPENDENCY = frozenset( + { + "actor_id", + "agent_id", + "agent_run_id", + "allowed_chat_ids", + "belief_id", + "belief_memory_id", + "candidate_edge_ids", + "candidate_memory_id", + "candidate_open_loop_ids", + "chat_id", + "connector_id", + "conversation_id", + "created_by_agent_id", + "external_chat_id", + "external_id", + "failed_external_ids", + "from_id", + "last_failed_external_ids", + "last_run_id", + "last_source_ids", + "message_id", + "output_artifact_id", + "person_id", + "project_id", + "project_ids", + "promoted_memory_id", + "provider_message_id", + "provider_update_id", + "review_artifact_id", + "revises_memory_id", + "rollup_candidate_ids", + "run_id", + "scheduler_run_id", + "sender_id", + "source_chunk_id", + "source_event_ids", + "source_ref", + "survivor_memory_id", + "target_id", + "task_id", + "to_id", + "trace_id", + "workflow_id", + } +) + + +def test_the_seven_sensitivity_rank_tables_equal_the_kernel_table() -> None: + root = Path(__file__).resolve().parents[2] / "apps" / "api" / "src" / "alicebot_api" + files = [ + "vnext_brain.py", + "vnext_consolidation.py", + "vnext_connections.py", + "vnext_contradictions.py", + "vnext_rollups.py", + "vnext_scheduler.py", + "vnext_projects.py", + ] + found = 0 + for name in files: + tree = ast.parse((root / name).read_text(encoding="utf-8")) + for node in ast.walk(tree): + if not isinstance(node, ast.Dict): + continue + keys = [key.value for key in node.keys if isinstance(key, ast.Constant)] + if "highly_sensitive" not in keys or "regulated" not in keys: + continue + table = ast.literal_eval(node) + assert table == SENSITIVITY_RANK + found += 1 + assert found == 7 + assert brain.SENSITIVITY_RANK == SENSITIVITY_RANK + assert consolidation.SENSITIVITY_RANK == SENSITIVITY_RANK + + +def test_labels_raised_events_carry_no_text_or_ids() -> None: + payload = labels_raised_payload( + cause="input_relabelled", + previous={"domain": "unknown", "sensitivity": "public", "project_scope": [ALPHA], "project_floor": []}, + new={"domain": "health", "sensitivity": "confidential", "project_scope": [ALPHA], "project_floor": [BETA]}, + ) + assert set(payload) == {"cause", "previous", "new"} + for side in (payload["previous"], payload["new"]): + assert set(side) == {"domain", "sensitivity", "project_scope", "project_floor"} + blob = str(payload) + assert "canonical" not in blob + assert "title" not in blob + + +def test_project_floor_blocks_a_locked_key_and_does_not_change_an_empty_floor() -> None: + identity = AgentIdentity( + agent_id="reader", + permission_profile="read_only_agent", + project_scope=(ALPHA,), + project_scope_locked=True, + auth="api_key", + ) + blocked = evaluate_agent_policy( + identity=identity, + action="artifact.get", + domains=("unknown",), + sensitivity_allowed=("public",), + project_scope=(ALPHA,), + project_floor=(BETA,), + require_explicit_project_scope=True, + ) + assert blocked.decision == "blocked" + assert "project_floor_binding_violation" in blocked.reasons + allowed = evaluate_agent_policy( + identity=identity, + action="artifact.get", + domains=("unknown",), + sensitivity_allowed=("public",), + project_scope=(ALPHA,), + project_floor=(ALPHA,), + require_explicit_project_scope=True, + ) + assert allowed.decision == "allowed" + assert project_floor_within((), (ALPHA,)) is True + untouched = evaluate_agent_policy( + identity=identity, + action="artifact.get", + domains=("unknown",), + sensitivity_allowed=("public",), + project_scope=(ALPHA,), + require_explicit_project_scope=True, + ) + assert untouched.decision == "allowed" + + +def test_a_global_row_with_a_floor_is_hidden_from_the_other_project() -> None: + assert project_scopes_overlap((), (GLOBAL_PROJECT_MARKER, ALPHA), floor=(BETA,)) is False + assert project_scopes_overlap((), (GLOBAL_PROJECT_MARKER, BETA), floor=(BETA,)) is True + assert project_scopes_overlap((), (GLOBAL_PROJECT_MARKER, ALPHA), floor=("Alice",)) is True + assert project_scopes_overlap((), (GLOBAL_PROJECT_MARKER,)) is True + assert project_scopes_overlap((ALPHA,), (ALPHA,)) is True + original = _memory("plain", metadata_json={"project_scope": [ALPHA]}) + derived = _memory( + "derived", + metadata_json={"source_id": "s", "project_scope": [ALPHA], "project_floor": [BETA]}, + ) + assert group_scope(original, kind="memory") == group_scope( + _memory("same", metadata_json={"project_scope": [ALPHA]}), kind="memory" + ) + assert BETA.lower() in group_scope(derived, kind="memory") or "prj_" + "b" * 16 in group_scope(derived, kind="memory") + + +def test_v2_id_keys_are_the_previous_set() -> None: + assert V2_ID_KEYS == frozenset(_ID_KEYS) + + +def test_rank_table_matches_the_spec_order() -> None: + assert SENSITIVITY_RANK["public"] < SENSITIVITY_RANK["unknown"] == SENSITIVITY_RANK["internal"] + assert SENSITIVITY_RANK["sacred"] == SENSITIVITY_RANK["regulated"] + assert "health" in RESTRICTED_DOMAINS diff --git a/tests/unit/test_derived_labels_mutations.py b/tests/unit/test_derived_labels_mutations.py new file mode 100644 index 000000000..68cd1e888 --- /dev/null +++ b/tests/unit/test_derived_labels_mutations.py @@ -0,0 +1,161 @@ +"""Kill each pure derived-label guard in memory. A survivor fails the run.""" + +from __future__ import annotations + +import inspect +import sys +import textwrap +from types import FunctionType + +import pytest + +from alicebot_api import vnext_agent_control as agent_control +from alicebot_api import vnext_brain as brain +from alicebot_api import vnext_derived_labels as labels +from tests.unit import test_derived_labels_kernel as checks + + +def kill(owner, name, before, after, check): + original = getattr(owner, name) + source = textwrap.dedent(inspect.getsource(original)) + if before not in source and before.replace("'", '"') in source: + before, after = before.replace("'", '"'), after.replace("'", '"') + assert before in source, f"mutation no longer matches: {name}: {before}" + namespace = dict(original.__globals__) + exec(compile(source.replace(before, after, 1), "", "exec"), namespace) + compiled = namespace[name] + replacement = FunctionType( + compiled.__code__, original.__globals__, name, compiled.__defaults__, compiled.__closure__ + ) + replacement.__kwdefaults__ = compiled.__kwdefaults__ + aliases = [ + (module, key) + for module in list(sys.modules.values()) + if module + and ( + getattr(module, "__name__", "").startswith("alicebot_api") + or module in (checks,) + ) + for key, value in list(vars(module).items()) + if value is original + ] + for module, key in aliases: + setattr(module, key, replacement) + setattr(owner, name, replacement) + try: + try: + check() + except (AssertionError, pytest.fail.Exception): + print(f"KILLED {name}: {before}") + else: + raise RuntimeError(f"SURVIVED {name}: {before}") + finally: + setattr(owner, name, original) + for module, key in aliases: + setattr(module, key, original) + + +def test_derived_label_kernel_guard_mutations(): + kill( + labels, + "_raised_sensitivity", + "if rank > chosen_rank:", + "if rank >= chosen_rank:", + checks.test_sensitivity_is_raise_only_and_unknown_never_swaps_with_internal, + ) + kill( + labels, + "union_floor", + "combined = [*stored, *[item for part in parts for item in part]]", + "combined = list(stored)", + checks.test_floor_is_the_union_and_never_shrinks, + ) + kill( + labels, + "_effective_scope", + "return _copy_scope(current.stored_scope, parents)", + "return current.stored_scope", + checks.test_scope_rules_per_row_class, + ) + kill( + labels, + "_effective_scope", + "return () # aggregate rule otherwise leaves the scope empty", + "return stored", + checks.test_scope_rules_per_row_class, + ) + kill( + labels, + "settle_labels", + 'problems[key] = "missing_dependency"', + "pass", + checks.test_every_unverified_case_is_unverified, + ) + kill( + labels, + "_derived_from_deps", + 'if expected != len(ids):\n return "counts_disagree", found', + 'if expected != len(ids):\n return "", found', + checks.test_every_unverified_case_is_unverified, + ) + kill( + labels, + "_spread_unverified", + 'problems[key] = "dependency_unverified"\n changed = True', + "pass", + checks.test_unverified_is_contagious_through_every_level, + ) + kill( + labels, + "dependency_record", + "if not problem and not _record_present(meta, row) and not found:", + "if not problem and not found:", + checks.test_every_unverified_case_is_unverified, + ) + kill( + agent_control, + "evaluate_agent_policy", + "elif project_floor and not project_floor_within(project_floor, identity.project_scope):", + "elif False and not project_floor_within(project_floor, identity.project_scope):", + checks.test_project_floor_blocks_a_locked_key_and_does_not_change_an_empty_floor, + ) + kill( + labels, + "_effective_scope", + "return () # a global input empties the report scope", + "return current.stored_scope", + checks.test_scope_rules_per_row_class, + ) + kill( + labels, + "carries_scope", + 'if name == "source" and _scrubbed(row):', + "if False and _scrubbed(row):", + checks.test_every_unverified_case_is_unverified, + ) + kill( + labels, + "_legacy_count_problem", + 'return "legacy_counts"', + 'return ""', + checks.test_every_unverified_case_is_unverified, + ) + kill( + labels, + "labels_raised_payload", + 'return {"cause": cause, "previous": side(previous), "new": side(new)}', + 'return {"cause": cause, "previous": side(previous), "new": side(new), "input_id": "row-1"}', + checks.test_labels_raised_events_carry_no_text_or_ids, + ) + original = dict(brain.SENSITIVITY_RANK) + brain.SENSITIVITY_RANK["sacred"] = 7 + try: + try: + checks.test_the_seven_sensitivity_rank_tables_equal_the_kernel_table() + except (AssertionError, pytest.fail.Exception): + print("KILLED SENSITIVITY_RANK sacred") + else: + raise RuntimeError("SURVIVED brain sensitivity rank") + finally: + brain.SENSITIVITY_RANK.clear() + brain.SENSITIVITY_RANK.update(original) From a30f5e613480354b9647f97dc1369c6abb343a49 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 17:35:17 +0000 Subject: [PATCH 02/90] Floor derived memory inserts and keep label keys on update. A memory copied from another row is stored at least as strict as that row. A metadata write keeps the marker and the stored project scope and floor. --- CHANGELOG.md | 1 + apps/api/src/alicebot_api/sqlite_store.py | 31 +++ .../src/alicebot_api/vnext_label_writes.py | 228 ++++++++++++++++++ apps/api/src/alicebot_api/vnext_store.py | 47 ++++ .../vnext_stores/postgres/memory_lifecycle.py | 27 ++- .../vnext_stores/sqlite/memory_lifecycle.py | 37 ++- tests/unit/test_derived_domain_mutations.py | 5 +- tests/unit/test_derived_domain_review.py | 17 +- tests/unit/test_derived_domain_stored_ids.py | 7 +- .../test_sqlite_derived_labels_write_path.py | 121 ++++++++++ 10 files changed, 506 insertions(+), 15 deletions(-) create mode 100644 apps/api/src/alicebot_api/vnext_label_writes.py create mode 100644 tests/unit/test_sqlite_derived_labels_write_path.py diff --git a/CHANGELOG.md b/CHANGELOG.md index 510fd2dde..c094b68c1 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,7 @@ ## Unreleased +- Unreleased (on main, not in v0.20.0): a memory copied from another row is stored at least as strict as that row. v0.20.0 kept the label the copy was given at insert, so a later reader could see a health or confidential input through a copy stored as `unknown` and `public`. The insert now re-reads the inputs inside the write and raises the domain, the sensitivity, and the project floor. A later metadata write cannot drop the marker that makes the row derived, or put back an older project scope or project floor. No migration is required. Reports, open loops, and a relabel of an existing row are not covered yet. - Unreleased (on main, not in v0.20.0): the label of a derived row (its domain, its sensitivity, and the projects it requires) is now decided by one pure function. v0.20.0 kept the label a derived row was given when it was made. Nothing in the running product calls the function yet, so a read and a write still behave as they do in v0.20.0. A locked key is refused when a caller passes a project floor that is not inside the key's binding. No migration is required. - Unreleased (on main, not in v0.20.0): on SQLite, `sources delete`, `sources prune --superseded` and `import-markdown --supersede` now scrub an open loop that names the source only in its metadata: the id, or `source:`, as the text under `source_id`, `source_ids`, `source_ref`, `source_refs`, `source_references` or `selected_source_ids` at any depth. That is the rule the open-loop lookup of a source already used. They blanked only loops whose `source_id` column held the id, so a loop with an empty column kept its title, description and metadata, and they reached a new export. The delete preview now counts the same loops as the receipt. The rule reads the id only as stored or as `source:`: a loop that names the source in any spelling other than those two keeps its text, although the memory rule reads several other spellings. No migration is required. - Unreleased (on main, not in v0.20.0): the SQLite derived-label repair now updates and records each row under the id it is stored with (SQLite keeps capitals, missing hyphens, braces and `urn:uuid:` as written) and matches rows by the normalised id only to read the graph. If two stored spellings of one id exist, each is compared with the label its own recorded inputs give it, so neither is left at its old label while the other is relabelled, and neither is lowered. An update that changes no row stops the repair with `DerivedDomainRepairError` before any event or the completion stamp is written, on open and on restore; migration `20261004_0095` refuses the same way. Before, a derived memory stored under such an id kept its label while its relabel event and the stamp were written, and the pair survived export and import. A PostgreSQL uuid column was not affected. A vault that the earlier repair already stamped keeps such a row at its old label until a restore repairs it again. No migration is required. diff --git a/apps/api/src/alicebot_api/sqlite_store.py b/apps/api/src/alicebot_api/sqlite_store.py index af90950f8..3fe8eaec1 100644 --- a/apps/api/src/alicebot_api/sqlite_store.py +++ b/apps/api/src/alicebot_api/sqlite_store.py @@ -399,6 +399,37 @@ def __init__(self, conn: sqlite3.Connection, user_id: UUID | str): _ensure_embedding_content_sha256_sqlite(self.conn) _ensure_project_scope_identity_sqlite(self.conn) + def lock_label_writes(self, *, exclusive: bool = False) -> None: + """The SQLite writer lock is the label lock. Begin it when none is open.""" + + del exclusive + if not self.conn.in_transaction: + self.conn.execute("BEGIN IMMEDIATE") + + def read_label_rows(self, kind: str, ids: Sequence[str]) -> list[VNextRow]: + """Narrow label rows for the insert floor. No text columns.""" + + wanted = [str(item) for item in ids if str(item)] + if not wanted: + return [] + table = {"source": "sources", "memory": "memories", "open_loop": "open_loops"}.get(kind) + if table is None: + return [] + extra = "" + if table == "memories": + extra = ", value, project_id" + elif table == "open_loops": + extra = ", project_id, source_id, memory_id" + marks = ",".join("?" for _ in wanted) + return self._fetch_all( + f""" + SELECT id, user_id, domain, sensitivity, metadata_json{extra} + FROM {table} + WHERE user_id = ? AND id IN ({marks}) + """, + (self.user_id, *wanted), + ) + # -- fetch helpers (mirror PostgresVNextStore conventions) ------------ def _execute(self, query: str, params: tuple[object, ...] = ()) -> sqlite3.Cursor: diff --git a/apps/api/src/alicebot_api/vnext_label_writes.py b/apps/api/src/alicebot_api/vnext_label_writes.py new file mode 100644 index 000000000..faee03beb --- /dev/null +++ b/apps/api/src/alicebot_api/vnext_label_writes.py @@ -0,0 +1,228 @@ +"""Write-path label lock, insert floor, and metadata merge. + +The pure rules live in ``vnext_derived_labels``. This module is what a store +calls when it writes a row. +""" + +from __future__ import annotations + +from collections.abc import Iterator, Mapping +from contextlib import contextmanager +from functools import wraps +from typing import Any + +from alicebot_api.vnext_agent_control import RESTRICTED_DOMAINS +from alicebot_api.vnext_derived_labels import ( + MARKER_KEYS, + dependencies_of, + generation_domain, + is_derived, + labels_raised_payload, + settle_labels, +) +from alicebot_api.vnext_event_log import build_event_log_record, integrity_hash_for_event +from alicebot_api.vnext_project_scope import project_scope_identity +from alicebot_api.vnext_repositories import JsonObject + +LABEL_METADATA_KEYS = ("project_scope", "project_floor", "derived_from") +STRICT_LOCK_ORDER = False +_FLOOR_ENABLED = True + +_KIND_EVENT = { + "memory": "memory", + "open_loop": "open_loop", + "artifact": "artifact", + "project": "project", + "source": "source", +} + + +class LabelLockOrderError(RuntimeError): + """A label write ran outside a transaction, or it took locks in the wrong order.""" + + +def takes_label_lock(fn: Any) -> Any: + """Take the shared label lock before a store method writes or row-locks a label table.""" + + @wraps(fn) + def wrapper(self: Any, *args: Any, **kwargs: Any) -> Any: + lock = getattr(self, "lock_label_writes", None) + if callable(lock): + lock(exclusive=False) + return fn(self, *args, **kwargs) + + wrapper.__takes_label_lock__ = True # type: ignore[attr-defined] + return wrapper + + +@contextmanager +def without_insert_floor() -> Iterator[None]: + """Let a test store a derived row with the labels it wrote.""" + + global _FLOOR_ENABLED + previous = _FLOOR_ENABLED + _FLOOR_ENABLED = False + try: + yield + finally: + _FLOOR_ENABLED = previous + + +def merge_protected_metadata( + stored: Mapping[str, object] | None, + patch: Mapping[str, object] | None, + *, + label_write: bool, +) -> dict[str, object]: + """Keep label keys and omitted marker keys when a caller writes metadata back. + + ``label_write`` is for the relabel itself, which may change the label keys. + A marker key that the patch omits stays as stored either way. + """ + + stored_map = dict(stored) if isinstance(stored, Mapping) else {} + if patch is None: + return stored_map + patch_map = dict(patch) + result = dict(patch_map) + if not label_write: + for key in LABEL_METADATA_KEYS: + if key in stored_map: + result[key] = stored_map[key] + else: + result.pop(key, None) + for key in MARKER_KEYS: + if key not in patch_map and key in stored_map: + result[key] = stored_map[key] + return result + + +def _in_transaction(store: Any) -> bool: + conn = getattr(store, "conn", None) + if conn is None: + return True + info = getattr(conn, "info", None) + status = getattr(info, "transaction_status", None) + if status is not None: + # psycopg TransactionStatus.IDLE is 0. + return int(status) != 0 + in_transaction = getattr(conn, "in_transaction", None) + if isinstance(in_transaction, bool): + return in_transaction + return True + + +def _label_tuple(payload: Mapping[str, object]) -> tuple[str, str, tuple[str, ...], tuple[str, ...]]: + metadata = payload.get("metadata_json") + meta = metadata if isinstance(metadata, Mapping) else {} + scope = meta.get("project_scope", payload.get("project_scope", ())) + floor = meta.get("project_floor", ()) + scope_values = tuple(scope) if isinstance(scope, (list, tuple)) else () + floor_values = tuple(floor) if isinstance(floor, (list, tuple)) else () + return ( + str(payload.get("domain") or "unknown"), + str(payload.get("sensitivity") or "unknown"), + project_scope_identity(scope_values), + project_scope_identity(floor_values), + ) + + +def apply_insert_floor(store: Any, kind: str, payload: Mapping[str, object]) -> tuple[JsonObject, JsonObject | None]: + """Raise a derived payload to its inputs. Returns ``(payload, event or None)``. + + The event has no target id yet. The insert fills that in after the row exists. + An unverified dependency does not refuse the insert. + """ + + body = dict(payload) + if not _FLOOR_ENABLED or not is_derived(kind, body): + return body, None + if not _in_transaction(store): + raise LabelLockOrderError("the insert floor requires an open transaction") + user_id = str(getattr(store, "user_id", body.get("user_id") or "")) + nodes: list[dict[str, object]] = [] + grouped: dict[str, list[str]] = {} + for dep_kind, dep_id in dependencies_of(kind, body): + grouped.setdefault(dep_kind, []).append(dep_id) + reader = getattr(store, "read_label_rows", None) + if callable(reader): + for dep_kind, ids in grouped.items(): + for row in reader(dep_kind, ids): + copied = dict(row) + copied["kind"] = dep_kind + copied.setdefault("user_id", user_id) + nodes.append(copied) + if not user_id and nodes: + user_id = str(nodes[0].get("user_id") or "") + own_id = str(body.get("id") or "new-derived-row") + own = dict(body) + own["kind"] = kind + own["id"] = own_id + own["user_id"] = user_id + nodes.append(own) + settled = settle_labels(nodes).by_stored(kind, own_id, user_id=user_id or None) + if settled.unverified: + return body, None + domain = settled.domain + if str(body.get("domain") or "unknown") in RESTRICTED_DOMAINS: + domain = generation_domain(str(body.get("domain")), [settled.domain]) + metadata = dict(body.get("metadata_json")) if isinstance(body.get("metadata_json"), Mapping) else {} + metadata["project_scope"] = list(settled.project_scope) + metadata["project_floor"] = list(settled.project_floor) + updated = dict(body) + updated["domain"] = domain + updated["sensitivity"] = settled.sensitivity + updated["metadata_json"] = metadata + if len(project_scope_identity(settled.project_scope)) != 1: + updated["project_id"] = None + before = _label_tuple(body) + after = _label_tuple(updated) + if before == after: + return updated, None + event = build_event_log_record( + event_type=f"{_KIND_EVENT.get(kind, kind)}.labels_raised", + actor_type="system", + target_type=_KIND_EVENT.get(kind, kind), + payload=labels_raised_payload( + cause="insert_floor", + previous={ + "domain": before[0], + "sensitivity": before[1], + "project_scope": list(before[2]), + "project_floor": list(before[3]), + }, + new={ + "domain": after[0], + "sensitivity": after[1], + "project_scope": list(settled.project_scope), + "project_floor": list(settled.project_floor), + }, + ), + ) + return updated, event + + +def remember_floor_event(store: Any, event: JsonObject | None, target_id: object) -> None: + """Append an insert-floor event once the row id is known.""" + + if event is None: + return + event = dict(event) + event["target_id"] = str(target_id) if target_id else None + event.pop("integrity_hash", None) + event["integrity_hash"] = integrity_hash_for_event(event) + append = getattr(store, "append_event", None) + if callable(append): + append(event) + + +__all__ = [ + "LABEL_METADATA_KEYS", + "STRICT_LOCK_ORDER", + "LabelLockOrderError", + "apply_insert_floor", + "merge_protected_metadata", + "remember_floor_event", + "takes_label_lock", + "without_insert_floor", +] diff --git a/apps/api/src/alicebot_api/vnext_store.py b/apps/api/src/alicebot_api/vnext_store.py index ed85ec7a7..92fe7a7c0 100644 --- a/apps/api/src/alicebot_api/vnext_store.py +++ b/apps/api/src/alicebot_api/vnext_store.py @@ -467,6 +467,53 @@ class PostgresVNextStore: def __init__(self, conn: UserConnection): self.conn = conn + def lock_label_writes(self, *, exclusive: bool = False) -> None: + """Shared label lock for a write, or the exclusive lock for a relabel. + + The lock is transaction scoped. A session that already holds it takes + the same statement again, which Postgres grants without waiting. + """ + + mode = "pg_advisory_xact_lock" if exclusive else "pg_advisory_xact_lock_shared" + with self.conn.cursor() as cur: + cur.execute( + f"SELECT {mode}(hashtext('vnext_labels'), hashtext(app.current_user_id()::text))" + ) + + def read_label_rows(self, kind: str, ids: Sequence[str]) -> list[VNextRow]: + """Narrow label rows for the insert floor. No text columns.""" + + wanted = [str(item) for item in ids if str(item)] + if not wanted: + return [] + table = { + "source": "sources", + "memory": "memories", + "open_loop": "open_loops", + "artifact": "generated_artifacts", + "project": "projects", + "belief": "beliefs", + }.get(kind) + if table is None: + return [] + extra = "" + if table == "memories": + extra = ", value, project_id" + elif table == "open_loops": + extra = ", project_id, source_id, memory_id" + elif table == "beliefs": + extra = ", memory_id" + elif table == "generated_artifacts": + extra = ", artifact_type" + return self._fetch_all( + f""" + SELECT id, user_id, domain, sensitivity, metadata_json{extra} + FROM {table} + WHERE id = ANY(%s::uuid[]) + """, + (wanted,), + ) + def _fetch_one( self, operation_name: str, diff --git a/apps/api/src/alicebot_api/vnext_stores/postgres/memory_lifecycle.py b/apps/api/src/alicebot_api/vnext_stores/postgres/memory_lifecycle.py index 288f2798f..cd069e6e4 100644 --- a/apps/api/src/alicebot_api/vnext_stores/postgres/memory_lifecycle.py +++ b/apps/api/src/alicebot_api/vnext_stores/postgres/memory_lifecycle.py @@ -11,6 +11,12 @@ from alicebot_api.vnext_event_log import build_event_log_record from alicebot_api.vnext_project_scope import canonical_memory_metadata from alicebot_api.vnext_repositories import JsonObject +from alicebot_api.vnext_label_writes import ( + apply_insert_floor, + merge_protected_metadata, + remember_floor_event, + takes_label_lock, +) from alicebot_api.vnext_stores.memory_lifecycle_common import ( REDACTED_JSON_VALUE, REDACTION_MARKER, @@ -34,7 +40,9 @@ VNextRow = dict[str, object] +@takes_label_lock def create_memory(self, memory: JsonObject, *, actor_type: str = "system") -> VNextRow: + memory, floor_event = apply_insert_floor(self, "memory", memory) refuse_created_credential_activation(memory) row = self._fetch_one( "create_memory", @@ -171,6 +179,7 @@ def create_memory(self, memory: JsonObject, *, actor_type: str = "system") -> VN target_id=row["id"], payload={"operation": "create", "fields": _sorted_field_names(memory)}, ) + remember_floor_event(self, floor_event, row["id"]) return row def upsert_memory_by_key(self, memory: JsonObject, *, actor_type: str = "system") -> VNextRow: @@ -415,8 +424,24 @@ def list_memories_missing_fact_keys(self, *, limit: int = 100, after_id: str | N (after_id, after_id, limit), ) -def update_memory(self, *, memory_id: str, patch: JsonObject, actor_type: str = "system") -> VNextRow: +@takes_label_lock +def update_memory( + self, *, memory_id: str, patch: JsonObject, actor_type: str = "system", label_write: bool = False +) -> VNextRow: refuse_updated_credential_activation(patch, lambda: self.get_memory(str(memory_id))) + if "metadata_json" in patch and isinstance(patch.get("metadata_json"), dict): + current = self._fetch_optional_one( + "SELECT metadata_json FROM memories WHERE id = %s::uuid", + (memory_id,), + ) + if current is not None: + stored = current.get("metadata_json") + patch = dict(patch) + patch["metadata_json"] = merge_protected_metadata( + stored if isinstance(stored, dict) else {}, + patch["metadata_json"], + label_write=label_write, + ) row = self._fetch_one( "update_memory", f""" diff --git a/apps/api/src/alicebot_api/vnext_stores/sqlite/memory_lifecycle.py b/apps/api/src/alicebot_api/vnext_stores/sqlite/memory_lifecycle.py index 8c49b35a7..88080782c 100644 --- a/apps/api/src/alicebot_api/vnext_stores/sqlite/memory_lifecycle.py +++ b/apps/api/src/alicebot_api/vnext_stores/sqlite/memory_lifecycle.py @@ -11,6 +11,12 @@ from alicebot_api.vnext_event_log import build_event_log_record from alicebot_api.vnext_project_scope import canonical_memory_metadata from alicebot_api.vnext_repositories import JsonObject +from alicebot_api.vnext_label_writes import ( + apply_insert_floor, + merge_protected_metadata, + remember_floor_event, + takes_label_lock, +) from alicebot_api.vnext_stores.memory_lifecycle_common import ( REDACTED_JSON_VALUE, REDACTION_MARKER, @@ -34,7 +40,31 @@ VNextRow = dict[str, object] + +def _with_protected_metadata(self, memory_id: str, patch: JsonObject, *, label_write: bool) -> JsonObject: + """Re-read metadata inside the label lock and keep label and marker keys.""" + + if "metadata_json" not in patch or not isinstance(patch.get("metadata_json"), dict): + return patch + current = self._fetch_optional_one( + "SELECT metadata_json FROM memories WHERE id = ? AND user_id = ?", + (str(memory_id), self.user_id), + ) + if current is None: + return patch + stored = current.get("metadata_json") + merged = dict(patch) + merged["metadata_json"] = merge_protected_metadata( + stored if isinstance(stored, dict) else {}, + patch["metadata_json"], + label_write=label_write, + ) + return merged + + +@takes_label_lock def create_memory(self, memory: JsonObject, *, actor_type: str = "system") -> VNextRow: + memory, floor_event = apply_insert_floor(self, "memory", memory) refuse_created_credential_activation(memory) memory_id = _new_id(memory.get("id")) # One clock reading for the whole write. ``first_seen_at`` and ``last_seen_at`` default to it, so @@ -140,6 +170,7 @@ def create_memory(self, memory: JsonObject, *, actor_type: str = "system") -> VN target_id=row["id"], payload={"operation": "create", "fields": _sorted_field_names(memory)}, ) + remember_floor_event(self, floor_event, row["id"]) return row def upsert_memory_by_key(self, memory: JsonObject, *, actor_type: str = "system") -> VNextRow: @@ -263,8 +294,12 @@ def memory_redaction_bundle_is_exact(self, memory_id: str, artifact_ids: Sequenc ) return bool(row.get("exact")) -def update_memory(self, *, memory_id: str, patch: JsonObject, actor_type: str = "system") -> VNextRow: +@takes_label_lock +def update_memory( + self, *, memory_id: str, patch: JsonObject, actor_type: str = "system", label_write: bool = False +) -> VNextRow: refuse_updated_credential_activation(patch, lambda: self.get_memory(str(memory_id))) + patch = _with_protected_metadata(self, memory_id, patch, label_write=label_write) # One clock reading for the write: an archive sets ``updated_at`` and ``deleted_at`` together. now = _utc_now_iso() cursor = self._execute( diff --git a/tests/unit/test_derived_domain_mutations.py b/tests/unit/test_derived_domain_mutations.py index 18ea917bf..aa28d4819 100644 --- a/tests/unit/test_derived_domain_mutations.py +++ b/tests/unit/test_derived_domain_mutations.py @@ -22,6 +22,7 @@ from tests.unit import test_derived_domain_review as review from tests.unit import test_derived_domain_stored_ids as stored_ids from alicebot_api import onramp, sqlite_schema +from alicebot_api.vnext_label_writes import without_insert_floor def kill(owner, name, before, after, check): @@ -177,12 +178,12 @@ def test_derived_domain_guard_mutations(): from alicebot_api import vnext_rollups as rollups for module, check in [(rollups, checks.test_real_sqlite_rollup_keeps_restricted_input_domain), (consolidation, checks.test_consolidation_report_includes_rollup_input_domains)]: - with pytest.MonkeyPatch.context() as patch: + with pytest.MonkeyPatch.context() as patch, without_insert_floor(): patch.setattr(module, 'derived_domain', lambda rows, *, fallback: fallback) with pytest.raises(AssertionError): check(patch) print('KILLED producer selector:', module.__name__) - with tempfile.TemporaryDirectory() as directory, pytest.MonkeyPatch.context() as patch: + with tempfile.TemporaryDirectory() as directory, pytest.MonkeyPatch.context() as patch, without_insert_floor(): patch.setattr(sqlite_schema, '_relabel_derived_domains', lambda conn: None) with pytest.raises(AssertionError): checks.test_sqlite_upgrade_relabels_existing_derived_memory_only(Path(directory)) diff --git a/tests/unit/test_derived_domain_review.py b/tests/unit/test_derived_domain_review.py index 6c69fdaca..b35da191c 100644 --- a/tests/unit/test_derived_domain_review.py +++ b/tests/unit/test_derived_domain_review.py @@ -9,6 +9,7 @@ from alicebot_api.onramp import bootstrap_database, main as onramp_main from alicebot_api.sqlite_store import SQLiteVNextStore, sqlite_user_connection +from alicebot_api.vnext_label_writes import without_insert_floor from alicebot_api.vnext_derived_domain_backfill import plan_relabels from tests.unit.per_project_s2_support import add_memory from tests.unit.test_derived_domain_fence import USER @@ -25,7 +26,7 @@ def test_restore_repairs_derived_rows_before_publication(tmp_path, monkeypatch, with monkeypatch.context() as patch: patch.setattr(sqlite_schema, "_relabel_derived_domains", lambda conn: None) bootstrap_database(source, user_id=USER, user_email="local@alice") - with sqlite_user_connection(source, USER) as conn: + with without_insert_floor(), sqlite_user_connection(source, USER) as conn: store = SQLiteVNextStore(conn, USER) health = add_memory(store, key="health", text="A restricted observation", domain="health") derived = store.create_memory( @@ -52,7 +53,7 @@ def test_restore_repairs_derived_rows_before_publication(tmp_path, monkeypatch, from alicebot_api.onramp import sqlite_url_for_path from alicebot_api.vnext_agent_keys import create_agent_key - with sqlite_user_connection(destination, USER) as conn: + with without_insert_floor(), sqlite_user_connection(destination, USER) as conn: _, raw = create_agent_key( SQLiteVNextStore(conn, USER), user_id=USER, agent_id="restore-reader", permission_profile="read_only_agent" ) @@ -125,7 +126,7 @@ def test_repair_records_changed_rows_once(tmp_path): path = tmp_path / "audit.sqlite3" bootstrap_database(path, user_id=USER, user_email="local@alice") - with sqlite_user_connection(path, USER) as conn: + with without_insert_floor(), sqlite_user_connection(path, USER) as conn: store = SQLiteVNextStore(conn, USER) source = add_memory(store, key="health", text="Private observation", domain="health") derived = store.create_memory( @@ -261,7 +262,7 @@ def test_every_open_of_a_vault_holding_a_cycle_raises_the_clear_error_and_change path = tmp_path / "cycle.sqlite3" bootstrap_database(path, user_id=USER, user_email="local@alice") - with sqlite_user_connection(path, USER) as conn: + with without_insert_floor(), sqlite_user_connection(path, USER) as conn: store = SQLiteVNextStore(conn, USER) ids = sorted(add_memory(store, key=name, text=f"Row {name}")["id"] for name in "abc") # Each memory records the next as its consolidation input, and the labels alternate around the cycle. @@ -280,7 +281,7 @@ def test_every_open_of_a_vault_holding_a_cycle_raises_the_clear_error_and_change bootstrap_database(path, user_id=USER, user_email="local@alice") assert all(f"memories {row_id}" in str(caught.value) for row_id in ids) with pytest.raises(DerivedDomainRepairError, match="did not settle"): - with sqlite_user_connection(path, USER): + with without_insert_floor(), sqlite_user_connection(path, USER): pass with sqlite3.connect(path) as raw: assert [row[0] for row in raw.execute("SELECT domain FROM memories ORDER BY id")] == ["health", "legal", "health"] @@ -296,7 +297,7 @@ def test_import_of_a_backup_holding_a_cycle_stops_before_publication(tmp_path, m destination = tmp_path / "restored.sqlite3" backup = tmp_path / "backup.jsonl" bootstrap_database(source, user_id=USER, user_email="local@alice") - with sqlite_user_connection(source, USER) as conn: + with without_insert_floor(), sqlite_user_connection(source, USER) as conn: store = SQLiteVNextStore(conn, USER) ids = sorted(add_memory(store, key=name, text=f"Row {name}")["id"] for name in "abc") # The vault is stamped as repaired, so it opens and exports. Its rows still form the cycle that @@ -323,7 +324,7 @@ def test_sqlite_repair_follows_available_artifact_and_leaves_missing_inputs(tmp_ path = tmp_path / "promoted.sqlite3" bootstrap_database(path, user_id=USER, user_email="local@alice") - with sqlite_user_connection(path, USER) as conn: + with without_insert_floor(), sqlite_user_connection(path, USER) as conn: store = SQLiteVNextStore(conn, USER) # The local product does not store artifacts. Imported references to # absent artifacts are not guessed from copied text. @@ -446,7 +447,7 @@ def _seed_memory_chain(path, size=16): bootstrap_database(path, user_id=USER, user_email="local@alice") ids = [str(UUID(int=1000 + index)) for index in range(size)] - with sqlite_user_connection(path, USER) as conn: + with without_insert_floor(), sqlite_user_connection(path, USER) as conn: store = SQLiteVNextStore(conn, USER) for index, memory_id in enumerate(ids): store.create_memory({ diff --git a/tests/unit/test_derived_domain_stored_ids.py b/tests/unit/test_derived_domain_stored_ids.py index 511a6f3b4..a0a39758c 100644 --- a/tests/unit/test_derived_domain_stored_ids.py +++ b/tests/unit/test_derived_domain_stored_ids.py @@ -37,6 +37,7 @@ from alicebot_api import sqlite_schema, vnext_derived_domain_backfill as repair from alicebot_api.onramp import bootstrap_database, main as onramp_main from alicebot_api.sqlite_store import SQLiteVNextStore, sqlite_user_connection +from alicebot_api.vnext_label_writes import without_insert_floor from tests.unit.per_project_s2_support import add_memory from tests.unit.test_derived_domain_fence import USER @@ -63,7 +64,7 @@ def _old_vault(path, monkeypatch, spellings=("upper",)): with monkeypatch.context() as patch: patch.setattr(sqlite_schema, "_relabel_derived_domains", lambda conn: None) bootstrap_database(path, user_id=USER, user_email="local@alice") - with sqlite_user_connection(path, USER) as conn: + with without_insert_floor(), sqlite_user_connection(path, USER) as conn: store = SQLiteVNextStore(conn, USER) health = add_memory(store, key="health", text="A restricted observation", domain="health") canonical_ids = [] @@ -162,7 +163,7 @@ def _twin_vault(path, monkeypatch, *, first_domain, last_domain="unknown", input with monkeypatch.context() as patch: patch.setattr(sqlite_schema, "_relabel_derived_domains", lambda conn: None) bootstrap_database(path, user_id=USER, user_email="local@alice") - with sqlite_user_connection(path, USER) as conn: + with without_insert_floor(), sqlite_user_connection(path, USER) as conn: store = SQLiteVNextStore(conn, USER) health = add_memory(store, key="health", text="A restricted observation", domain=input_domain) metadata = {"consolidation": {"cluster_member_ids": [health["id"]]}} @@ -328,7 +329,7 @@ def test_a_refusal_writes_no_event_and_no_stamp_even_after_an_earlier_update(tmp with monkeypatch.context() as patch: patch.setattr(sqlite_schema, "_relabel_derived_domains", lambda conn: None) bootstrap_database(path, user_id=USER, user_email="local@alice") - with sqlite_user_connection(path, USER) as conn: + with without_insert_floor(), sqlite_user_connection(path, USER) as conn: store = SQLiteVNextStore(conn, USER) health = add_memory(store, key="health", text="A restricted observation", domain="health") ids = sorted( diff --git a/tests/unit/test_sqlite_derived_labels_write_path.py b/tests/unit/test_sqlite_derived_labels_write_path.py new file mode 100644 index 000000000..680ef6de6 --- /dev/null +++ b/tests/unit/test_sqlite_derived_labels_write_path.py @@ -0,0 +1,121 @@ +"""SQLite insert floor and metadata merge. Scratch vaults only.""" + +from __future__ import annotations + +from pathlib import Path + +from alicebot_api.onramp import bootstrap_database +from alicebot_api.sqlite_store import SQLiteVNextStore, sqlite_user_connection +from alicebot_api.vnext_label_writes import merge_protected_metadata, without_insert_floor +from tests.unit.per_project_s2_support import add_memory +from tests.unit.test_derived_domain_fence import USER + +ALPHA = "prj_" + "a" * 16 + + +def _vault(path: Path): + bootstrap_database(path, user_id=USER, user_email="local@alice") + return sqlite_user_connection(path, USER) + + +def test_a_copy_stored_after_its_source_is_floored(tmp_path: Path) -> None: + path = tmp_path / "vault.sqlite3" + with _vault(path) as conn: + store = SQLiteVNextStore(conn, USER) + source = store.create_source( + { + "source_type": "note", + "title": "Visit", + "content_hash": "hash-floor", + "domain": "health", + "sensitivity": "confidential", + "metadata_json": {"project_scope": [ALPHA]}, + } + ) + memory = store.create_memory( + { + "memory_key": "extracted", + "canonical_text": "A fact from the visit.", + "status": "active", + "domain": "unknown", + "sensitivity": "public", + "metadata_json": { + "source_id": source["id"], + "extraction_rule": "sentence", + "project_scope": [ALPHA, "prj_" + "b" * 16], + }, + } + ) + assert memory["domain"] == "health" + assert memory["sensitivity"] == "confidential" + assert memory["metadata_json"]["project_scope"] == [ALPHA] + assert memory["metadata_json"]["project_floor"] == [ALPHA] + + +def test_without_insert_floor_keeps_the_requested_label(tmp_path: Path) -> None: + path = tmp_path / "vault.sqlite3" + with _vault(path) as conn: + store = SQLiteVNextStore(conn, USER) + health = add_memory(store, key="health", text="A restricted observation", domain="health") + with without_insert_floor(): + derived = store.create_memory( + { + "memory_key": "derived", + "canonical_text": "A summary", + "status": "active", + "domain": "unknown", + "sensitivity": "public", + "metadata_json": {"consolidation": {"cluster_member_ids": [health["id"]]}}, + } + ) + assert derived["domain"] == "unknown" + assert derived["sensitivity"] == "public" + + +def test_an_update_that_omits_a_marker_keeps_it(tmp_path: Path) -> None: + path = tmp_path / "vault.sqlite3" + with _vault(path) as conn: + store = SQLiteVNextStore(conn, USER) + health = add_memory(store, key="health", text="A restricted observation", domain="health") + with without_insert_floor(): + derived = store.create_memory( + { + "memory_key": "derived", + "canonical_text": "A summary", + "status": "active", + "domain": "health", + "sensitivity": "public", + "metadata_json": { + "consolidation": {"cluster_member_ids": [health["id"]]}, + "project_scope": [ALPHA], + "project_floor": [ALPHA], + "note": "keep", + }, + } + ) + stored = store.update_memory( + memory_id=str(derived["id"]), + patch={"metadata_json": {"note": "changed", "project_scope": ["other"]}}, + ) + assert stored["metadata_json"]["consolidation"]["cluster_member_ids"] == [health["id"]] + assert stored["metadata_json"]["project_scope"] == [ALPHA] + assert stored["metadata_json"]["project_floor"] == [ALPHA] + assert stored["metadata_json"]["note"] == "changed" + + +def test_merge_protected_metadata_keeps_label_keys_unless_the_write_is_a_relabel() -> None: + stored = { + "consolidation": {"cluster_member_ids": ["m"]}, + "project_scope": [ALPHA], + "project_floor": [ALPHA], + "note": "old", + } + patch = {"note": "new", "project_scope": ["other"]} + merged = merge_protected_metadata(stored, patch, label_write=False) + assert merged["project_scope"] == [ALPHA] + assert merged["project_floor"] == [ALPHA] + assert merged["consolidation"] == {"cluster_member_ids": ["m"]} + assert merged["note"] == "new" + relabel = merge_protected_metadata(stored, {"project_scope": [], "project_floor": [ALPHA]}, label_write=True) + assert relabel["project_scope"] == [] + assert relabel["consolidation"] == {"cluster_member_ids": ["m"]} From 45ef947af340845f39eafe403c12fd4186a6c65b Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 17:47:37 +0000 Subject: [PATCH 03/90] Propagate derived labels and lock every label-table writer. A stricter source or memory raises the rows derived from it. Artifact and open-loop inserts take the same floor. A source move previews how many derived rows a project key would lose, and a relabel that cannot finish answers 409 or 503. --- CHANGELOG.md | 4 +- .../alicebot_api/routers/vnext_memories.py | 602 ++++++++++-------- apps/api/src/alicebot_api/sqlite_store.py | 8 + .../src/alicebot_api/vnext_label_writes.py | 430 ++++++++++++- apps/api/src/alicebot_api/vnext_store.py | 23 + .../vnext_stores/postgres/embedding_cas.py | 3 + .../vnext_stores/postgres/events_revisions.py | 2 + .../vnext_stores/postgres/graph_open_loops.py | 8 + .../vnext_stores/postgres/memory_lifecycle.py | 11 + .../vnext_stores/sqlite/embedding_cas.py | 3 + .../vnext_stores/sqlite/graph_open_loops.py | 8 + .../vnext_stores/sqlite/memory_lifecycle.py | 8 + .../vnext_stores/sqlite/source_retirement.py | 7 + .../test_sqlite_derived_labels_write_path.py | 111 ++++ .../unit/test_store_events_revisions_split.py | 7 +- 15 files changed, 950 insertions(+), 285 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index f466e313b..3612de93b 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,8 +2,8 @@ ## Unreleased -- Unreleased (on main, not in v0.20.0): a memory copied from another row is stored at least as strict as that row. v0.20.0 kept the label the copy was given at insert, so a later reader could see a health or confidential input through a copy stored as `unknown` and `public`. The insert now re-reads the inputs inside the write and raises the domain, the sensitivity, and the project floor. A later metadata write cannot drop the marker that makes the row derived, or put back an older project scope or project floor. No migration is required. Reports, open loops, and a relabel of an existing row are not covered yet. -- Unreleased (on main, not in v0.20.0): the label of a derived row (its domain, its sensitivity, and the projects it requires) is now decided by one pure function. v0.20.0 kept the label a derived row was given when it was made. Nothing in the running product calls the function yet, so a read and a write still behave as they do in v0.20.0. A locked key is refused when a caller passes a project floor that is not inside the key's binding. No migration is required. +- Unreleased (on main, not in v0.20.0): a memory copied from another row is stored at least as strict as that row. v0.20.0 kept the label the copy was given at insert, so a later reader could see a health or confidential input through a copy stored as `unknown` and `public`. The insert now re-reads the inputs inside the write and raises the domain, the sensitivity, and the project floor. A later metadata write cannot drop the marker that makes the row derived, or put back an older project scope or project floor. Raising a source or a memory walks the rows derived from it and raises those too. A candidate open loop and a generated artifact take the same insert floor. Replacing a SQLite source with a stricter one raises the memories that cite it before they are retired. Moving a source to another project answers with how many derived rows a project-bound key would lose, and waits for confirm_label_hide before it writes. A relabel that cannot finish answers 409, and one that waits more than 3 seconds for the label lock answers 503. No migration is required. +- Unreleased (on main, not in v0.20.0): the label of a derived row (its domain, its sensitivity, and the projects it requires) is now decided by one pure function. v0.20.0 kept the label a derived row was given when it was made. The function is what the write path calls. A read that does not go through that path still behaves as it does in v0.20.0. A locked key is refused when a caller passes a project floor that is not inside the key's binding. No migration is required. - Unreleased (on main, not in v0.20.0): on SQLite, `sources delete`, `sources prune --superseded` and `import-markdown --supersede` now scrub an open loop that names the source only in its metadata: the id, or `source:`, as the text under `source_id`, `source_ids`, `source_ref`, `source_refs`, `source_references` or `selected_source_ids` at any depth. That is the rule the open-loop lookup of a source already used. They blanked only loops whose `source_id` column held the id, so a loop with an empty column kept its title, description and metadata, and they reached a new export. The delete preview now counts the same loops as the receipt. The rule reads the id only as stored or as `source:`: a loop that names the source in any spelling other than those two keeps its text, although the memory rule reads several other spellings. No migration is required. - Unreleased (on main, not in v0.20.0): the SQLite derived-label repair now updates and records each row under the id it is stored with (SQLite keeps capitals, missing hyphens, braces and `urn:uuid:` as written) and matches rows by the normalised id only to read the graph. If two stored spellings of one id exist, each is compared with the label its own recorded inputs give it, so neither is left at its old label while the other is relabelled, and neither is lowered. An update that changes no row stops the repair with `DerivedDomainRepairError` before any event or the completion stamp is written, on open and on restore; migration `20261004_0095` refuses the same way. Before, a derived memory stored under such an id kept its label while its relabel event and the stamp were written, and the pair survived export and import. A PostgreSQL uuid column was not affected. A vault that the earlier repair already stamped keeps such a row at its old label until a restore repairs it again. No migration is required. - Unreleased (on main, not in v0.20.0): v0.20.0 could store a consolidation report with a sensitivity below the memories it prints. The report took its sensitivity from the near-duplicate clusters only, so a run whose proposals were roll-up cards had no cluster to take it from and was stored as `unknown`. A key whose ceiling is below those memories (a `trusted_local_agent` key for confidential ones, a `read_only_agent` key for private ones) could then read the card topic and the member ids through `GET /v0/vnext/artifacts/{id}`. The report now takes its domain and its sensitivity over every row it names: the cluster members, the members of every proposed roll-up group, the members of the groups that a skip line names by key (`topic:...`, `entity:...`, `semantic:cluster-`), the pending, accepted, expired or held roll-up cards it names by id, and the sources that the `source_refs` of its cluster members name, which the report still prints. The open-loop review is labelled over the sources whose ids it prints as well as over its loops. The run digest of both reports covers those sources, so a source that was reclassified makes a new report instead of returning the earlier one. The first run after the upgrade over loops that link a source, or over cluster members that cite one, makes one new report, and a run that names no source keeps the digest it had. A report whose inputs are all unrestricted is now stored with the label of those inputs, `internal` where it was `unknown`, and every permission profile reads the two the same way. Reports stored by v0.20.0 keep their labels, because the stored-row repair does not change sensitivity. The other report producers (daily brief, weekly synthesis, connection and contradiction reports, project updates, staleness reports) were checked and already label over every row they print. Memories and sources are counted apart when the label is taken, because an id is unique only within its own table: a source that shares a memory's id can no longer replace that memory's label. No migration is required. diff --git a/apps/api/src/alicebot_api/routers/vnext_memories.py b/apps/api/src/alicebot_api/routers/vnext_memories.py index 893eb55e4..433d4716f 100644 --- a/apps/api/src/alicebot_api/routers/vnext_memories.py +++ b/apps/api/src/alicebot_api/routers/vnext_memories.py @@ -116,6 +116,7 @@ class VNextSourceReviewRequest(VNextAgentRequest): sensitivity: VNextSensitivity | None = None project_id: str | None = Field(default=None, min_length=1, max_length=120) review_note: str | None = Field(default=None, min_length=1, max_length=4000) + confirm_label_hide: bool = False class VNextConnectorSyncRequest(VNextAgentRequest): @@ -765,6 +766,12 @@ def review_vnext_source(source_id: UUID, request: VNextSourceReviewRequest) -> J try: with user_connection(settings.database_url, request.user_id) as conn: store = PostgresVNextStore(conn) + label_change = request.domain is not None or request.sensitivity is not None or request.project_id is not None + if label_change: + store.lock_graph_mutation() + from alicebot_api.vnext_label_writes import acquire_exclusive_label_lock + + acquire_exclusive_label_lock(store) existing = store.get_source(str(source_id)) if existing is None: return _vnext_public_error_response(status_code=404, detail="vNext source was not found") @@ -812,6 +819,19 @@ def review_vnext_source(source_id: UUID, request: VNextSourceReviewRequest) -> J patch["domain"] = request.domain if request.sensitivity is not None: patch["sensitivity"] = request.sensitivity + if action == "assign_project" and request.project_id is not None: + from alicebot_api.vnext_label_writes import count_rows_hidden_by_scope_move + + hidden = count_rows_hidden_by_scope_move(store, existing, [request.project_id]) + if hidden and not request.confirm_label_hide: + return JSONResponse( + status_code=200, + content={ + "preview": True, + "derived_rows_hidden_from_project_keys": hidden, + "confirm_required": True, + }, + ) updated = store.update_source(source_id=str(source_id), patch=patch, actor_type="user") if action == "assign_project": store.create_edge( @@ -846,6 +866,15 @@ def review_vnext_source(source_id: UUID, request: VNextSourceReviewRequest) -> J ) except ContinuityStoreInvariantError as exc: return public_exception_response(exc, status_code=409) + except Exception as exc: + from alicebot_api.vnext_label_writes import label_error_response + + mapped = label_error_response(exc) + if mapped is None: + raise + status, detail, retry_after = mapped + headers = {"Retry-After": retry_after} if retry_after else None + return JSONResponse(status_code=status, content={"detail": detail}, headers=headers) return JSONResponse( status_code=200, content=jsonable_encoder({"source": updated, "archived": False, "trace": trace}) @@ -972,300 +1001,321 @@ def review_vnext_memory( ), ) - with user_connection(settings.database_url, request.user_id) as conn: - store = PostgresVNextStore(conn) - memory_service = VNextMemoryCommitService(store, defer_embeddings=True) - # Review can promote a consolidation candidate or mutate a member - # referenced by pending derived work. Establish the shared per-user - # graph boundary before the route takes any candidate/member row lock; - # delegated service calls may safely reacquire the transaction lock. - memory_service.lock_supersession_graph() - preview = store.get_memory(str(memory_id)) - if preview is None: - return _vnext_public_error_response(status_code=404, detail="vNext memory was not found") - # Delegate consolidation approval before this adapter takes a row lock. - # The service reacquires the already-held transaction advisory lock - # (non-blocking/re-entrant) and then owns all candidate/member locks. - if is_pending_consolidation_candidate(preview): - if action == "edit" or any( - value is not None - for value in ( - request.title, - request.canonical_text, - request.summary, - request.domain, - request.sensitivity, - request.project_id, - ) - ): - return _vnext_public_error_response( - status_code=400, - detail=( - "pending consolidation candidates cannot be edited during approval; " - "regenerate the candidate or accept it unchanged" - ), - ) - if action in {"accept", "promote"}: - return _vnext_public_error_response( - status_code=409, - detail="vNext memory became a consolidation candidate during review; retry the approval", - ) - get_memory_for_update = getattr(store, "get_memory_for_update", None) - existing = ( - get_memory_for_update(str(memory_id)) - if callable(get_memory_for_update) - else store.get_memory(str(memory_id)) - ) - if existing is None: - return _vnext_public_error_response(status_code=404, detail="vNext memory was not found") - # Re-authorize the locked record so a concurrent reassignment cannot - # move it outside the bound agent project between the first check and - # this mutation. - locked_scope = resource_project_scope(existing) - if action == "assign_project" and request.project_id is not None: - locked_scope = tuple(dict.fromkeys((*locked_scope, request.project_id))) - locked_decision = _vnext_policy_checked( - store=store, - identity=identity, - action="memory.review", - domains=(str(existing.get("domain") or "unknown"),), - sensitivity_allowed=(str(existing.get("sensitivity") or "unknown"),), - project_scope=locked_scope, - target_type="memory", - target_id=str(memory_id), - require_explicit_project_scope=True, - ) - if locked_decision.decision == "blocked": - return _vnext_permission_response(locked_decision) - if str(existing.get("status") or "") in {"archived", "rejected", "superseded"}: - return _vnext_public_error_response( - status_code=409, - detail=f"vNext memory cannot be reviewed from status '{existing.get('status')}'", + try: + with user_connection(settings.database_url, request.user_id) as conn: + store = PostgresVNextStore(conn) + memory_service = VNextMemoryCommitService(store, defer_embeddings=True) + # Review can promote a consolidation candidate or mutate a member + # referenced by pending derived work. Establish the shared per-user + # graph boundary before the route takes any candidate/member row lock; + # delegated service calls may safely reacquire the transaction lock. + memory_service.lock_supersession_graph() + label_change = ( + request.domain is not None + or request.sensitivity is not None + or request.project_id is not None + or action in {"private", "assign_project"} ) - if is_pending_consolidation_candidate(existing): - if action == "edit" or any( - value is not None - for value in ( - request.title, - request.canonical_text, - request.summary, - request.domain, - request.sensitivity, - request.project_id, - ) - ): + if label_change: + from alicebot_api.vnext_label_writes import acquire_exclusive_label_lock + + acquire_exclusive_label_lock(store) + preview = store.get_memory(str(memory_id)) + if preview is None: + return _vnext_public_error_response(status_code=404, detail="vNext memory was not found") + # Delegate consolidation approval before this adapter takes a row lock. + # The service reacquires the already-held transaction advisory lock + # (non-blocking/re-entrant) and then owns all candidate/member locks. + if is_pending_consolidation_candidate(preview): + if action == "edit" or any( + value is not None + for value in ( + request.title, + request.canonical_text, + request.summary, + request.domain, + request.sensitivity, + request.project_id, + ) + ): + return _vnext_public_error_response( + status_code=400, + detail=( + "pending consolidation candidates cannot be edited during approval; " + "regenerate the candidate or accept it unchanged" + ), + ) + if action in {"accept", "promote"}: + return _vnext_public_error_response( + status_code=409, + detail="vNext memory became a consolidation candidate during review; retry the approval", + ) + get_memory_for_update = getattr(store, "get_memory_for_update", None) + existing = ( + get_memory_for_update(str(memory_id)) + if callable(get_memory_for_update) + else store.get_memory(str(memory_id)) + ) + if existing is None: + return _vnext_public_error_response(status_code=404, detail="vNext memory was not found") + # Re-authorize the locked record so a concurrent reassignment cannot + # move it outside the bound agent project between the first check and + # this mutation. + locked_scope = resource_project_scope(existing) + if action == "assign_project" and request.project_id is not None: + locked_scope = tuple(dict.fromkeys((*locked_scope, request.project_id))) + locked_decision = _vnext_policy_checked( + store=store, + identity=identity, + action="memory.review", + domains=(str(existing.get("domain") or "unknown"),), + sensitivity_allowed=(str(existing.get("sensitivity") or "unknown"),), + project_scope=locked_scope, + target_type="memory", + target_id=str(memory_id), + require_explicit_project_scope=True, + ) + if locked_decision.decision == "blocked": + return _vnext_permission_response(locked_decision) + if str(existing.get("status") or "") in {"archived", "rejected", "superseded"}: return _vnext_public_error_response( - status_code=400, - detail=( - "pending consolidation candidates cannot be edited during approval; " - "regenerate the candidate or accept it unchanged" - ), + status_code=409, + detail=f"vNext memory cannot be reviewed from status '{existing.get('status')}'", ) - if action in {"accept", "promote"}: + if is_pending_consolidation_candidate(existing): + if action == "edit" or any( + value is not None + for value in ( + request.title, + request.canonical_text, + request.summary, + request.domain, + request.sensitivity, + request.project_id, + ) + ): + return _vnext_public_error_response( + status_code=400, + detail=( + "pending consolidation candidates cannot be edited during approval; " + "regenerate the candidate or accept it unchanged" + ), + ) + if action in {"accept", "promote"}: + return _vnext_public_error_response( + status_code=409, + detail="vNext memory became a consolidation candidate during review; retry the approval", + ) + if is_pending_project_update_memory(existing): return _vnext_public_error_response( status_code=409, - detail="vNext memory became a consolidation candidate during review; retry the approval", + detail=PENDING_PROJECT_UPDATE_MEMORY_MUTATION_MESSAGE, ) - if is_pending_project_update_memory(existing): - return _vnext_public_error_response( - status_code=409, - detail=PENDING_PROJECT_UPDATE_MEMORY_MUTATION_MESSAGE, - ) - existing_metadata_value = existing.get("metadata_json") - existing_metadata: dict[str, object] = ( - existing_metadata_value if isinstance(existing_metadata_value, dict) else {} - ) - reviewed_at = datetime.now(UTC).isoformat() - patch: dict[str, object] = { - "last_reviewed_at": reviewed_at, - } - revision_type = "edited" - if action == "accept": - patch["status"] = "active" - revision_type = "promoted" - elif action == "reject": - patch["status"] = "rejected" - patch["metadata_json"] = _vnext_terminal_review_metadata( - existing_metadata, - outcome="rejected", - terminal_at=reviewed_at, + existing_metadata_value = existing.get("metadata_json") + existing_metadata: dict[str, object] = ( + existing_metadata_value if isinstance(existing_metadata_value, dict) else {} ) - revision_type = "rejected" - elif action == "private": - patch["status"] = "private_only" - patch["sensitivity"] = "private" - elif action == "promote": - patch["status"] = "active" - patch["confirmation_status"] = "confirmed" - revision_type = "promoted" - elif action == "assign_project": - if request.project_id is None: - return _vnext_public_error_response(status_code=400, detail="project_id is required") - # Keep every current scope representation in the same UPDATE. A - # metadata-only project_id write leaves an older project_scope in - # place, and canonical retrieval correctly gives that array - # precedence over the legacy singular fallback. - patch["project_id"] = request.project_id - patch["metadata_json"] = { - **existing_metadata, - "project_id": request.project_id, - "project_scope": [request.project_id], - "assigned_from": "vnext_workspace", + reviewed_at = datetime.now(UTC).isoformat() + patch: dict[str, object] = { + "last_reviewed_at": reviewed_at, } - else: - patch["status"] = "active" - - if action in {"accept", "edit", "promote"}: - patch.update( - { - "confirmation_status": "confirmed", - "last_confirmed_at": reviewed_at, - "metadata_json": _vnext_terminal_review_metadata( - existing_metadata, - outcome="confirmed", - terminal_at=reviewed_at, - ), + revision_type = "edited" + if action == "accept": + patch["status"] = "active" + revision_type = "promoted" + elif action == "reject": + patch["status"] = "rejected" + patch["metadata_json"] = _vnext_terminal_review_metadata( + existing_metadata, + outcome="rejected", + terminal_at=reviewed_at, + ) + revision_type = "rejected" + elif action == "private": + patch["status"] = "private_only" + patch["sensitivity"] = "private" + elif action == "promote": + patch["status"] = "active" + patch["confirmation_status"] = "confirmed" + revision_type = "promoted" + elif action == "assign_project": + if request.project_id is None: + return _vnext_public_error_response(status_code=400, detail="project_id is required") + # Keep every current scope representation in the same UPDATE. A + # metadata-only project_id write leaves an older project_scope in + # place, and canonical retrieval correctly gives that array + # precedence over the legacy singular fallback. + patch["project_id"] = request.project_id + patch["metadata_json"] = { + **existing_metadata, + "project_id": request.project_id, + "project_scope": [request.project_id], + "assigned_from": "vnext_workspace", } - ) + else: + patch["status"] = "active" - if request.title is not None: - patch["title"] = request.title - if request.canonical_text is not None: - patch["canonical_text"] = request.canonical_text - existing_value = existing.get("value") - patch["value"] = { - **(existing_value if isinstance(existing_value, dict) else {}), - "text": request.canonical_text, - } - # Capture-generated title/summary are denormalized views of the - # canonical text. Editing only the body must not leave those - # user-visible fields describing the pre-edit value. - if request.title is None: - patch["title"] = ( - request.canonical_text - if len(request.canonical_text) <= 120 - else request.canonical_text[:117].rstrip() + "..." - ) - if request.summary is None: - patch["summary"] = ( - request.canonical_text - if len(request.canonical_text) <= 280 - else request.canonical_text[:277].rstrip() + "..." + if action in {"accept", "edit", "promote"}: + patch.update( + { + "confirmation_status": "confirmed", + "last_confirmed_at": reviewed_at, + "metadata_json": _vnext_terminal_review_metadata( + existing_metadata, + outcome="confirmed", + terminal_at=reviewed_at, + ), + } ) - if request.summary is not None: - patch["summary"] = request.summary - if request.domain is not None: - patch["domain"] = request.domain - if request.sensitivity is not None: - patch["sensitivity"] = request.sensitivity - - # The credential floor, on the row as it will be stored. This route - # writes the row itself rather than through a service method, so it - # calls the shared checks here. Until 2026-09-22 an edit rewrote - # title, text and summary with no credential check. A title-only edit - # is read against the stored body, because that is the row a reader - # will see; derived previews of the text are left out. - # - # accept, promote and edit move the row into a searchable status, so - # they take the shared activation check over the row and the reason - # (ruling C2). A reject always completes (ruling C6): a reason, or - # edited text, that carries credential material is stored as a fixed - # placeholder and the response says so. private and assign_project - # keep the plain check. - text_edit = any(value is not None for value in (request.title, request.canonical_text, request.summary)) - stored_title = patch.get("title", existing.get("title")) - stored_text = patch.get("canonical_text", existing.get("canonical_text")) - stored_summary = patch.get("summary", existing.get("summary")) - review_reason = request.reason - rationale_withheld = False - text_withheld = False - if action == "reject": - review_reason, rationale_withheld = withhold_credential_text(request.reason) - if text_edit and carries_credential_material(request.title, request.canonical_text, request.summary): - for text_field in ("title", "canonical_text", "summary"): - if text_field in patch: - patch[text_field] = TEXT_WITHHELD_PLACEHOLDER - edited_value = patch.get("value") - if isinstance(edited_value, dict): - patch["value"] = {**edited_value, "text": TEXT_WITHHELD_PLACEHOLDER} - text_withheld = True - elif patch.get("status") in SEARCHABLE_STATUSES: - try: - refuse_credential_activation(stored_title, stored_text, stored_summary, request.reason) - except CredentialActivationRefused: - return _vnext_public_error_response( - status_code=400, detail="vNext memory review text carries credential material" + + if request.title is not None: + patch["title"] = request.title + if request.canonical_text is not None: + patch["canonical_text"] = request.canonical_text + existing_value = existing.get("value") + patch["value"] = { + **(existing_value if isinstance(existing_value, dict) else {}), + "text": request.canonical_text, + } + # Capture-generated title/summary are denormalized views of the + # canonical text. Editing only the body must not leave those + # user-visible fields describing the pre-edit value. + if request.title is None: + patch["title"] = ( + request.canonical_text + if len(request.canonical_text) <= 120 + else request.canonical_text[:117].rstrip() + "..." + ) + if request.summary is None: + patch["summary"] = ( + request.canonical_text + if len(request.canonical_text) <= 280 + else request.canonical_text[:277].rstrip() + "..." + ) + if request.summary is not None: + patch["summary"] = request.summary + if request.domain is not None: + patch["domain"] = request.domain + if request.sensitivity is not None: + patch["sensitivity"] = request.sensitivity + + # The credential floor, on the row as it will be stored. This route + # writes the row itself rather than through a service method, so it + # calls the shared checks here. Until 2026-09-22 an edit rewrote + # title, text and summary with no credential check. A title-only edit + # is read against the stored body, because that is the row a reader + # will see; derived previews of the text are left out. + # + # accept, promote and edit move the row into a searchable status, so + # they take the shared activation check over the row and the reason + # (ruling C2). A reject always completes (ruling C6): a reason, or + # edited text, that carries credential material is stored as a fixed + # placeholder and the response says so. private and assign_project + # keep the plain check. + text_edit = any(value is not None for value in (request.title, request.canonical_text, request.summary)) + stored_title = patch.get("title", existing.get("title")) + stored_text = patch.get("canonical_text", existing.get("canonical_text")) + stored_summary = patch.get("summary", existing.get("summary")) + review_reason = request.reason + rationale_withheld = False + text_withheld = False + if action == "reject": + review_reason, rationale_withheld = withhold_credential_text(request.reason) + if text_edit and carries_credential_material(request.title, request.canonical_text, request.summary): + for text_field in ("title", "canonical_text", "summary"): + if text_field in patch: + patch[text_field] = TEXT_WITHHELD_PLACEHOLDER + edited_value = patch.get("value") + if isinstance(edited_value, dict): + patch["value"] = {**edited_value, "text": TEXT_WITHHELD_PLACEHOLDER} + text_withheld = True + elif patch.get("status") in SEARCHABLE_STATUSES: + try: + refuse_credential_activation(stored_title, stored_text, stored_summary, request.reason) + except CredentialActivationRefused: + return _vnext_public_error_response( + status_code=400, detail="vNext memory review text carries credential material" + ) + else: + review_fields: tuple[object, ...] = (request.reason,) + if text_edit: + review_fields = (*stored_text_fields(stored_title, stored_text, stored_summary), request.reason) + if carries_credential_material(*review_fields): + return _vnext_public_error_response( + status_code=400, detail="vNext memory review text carries credential material" + ) + + updated = store.update_memory(memory_id=str(memory_id), patch=patch, actor_type=actor_type) + if action in ("accept", "edit", "promote"): + memory_service.refresh_memory_derived_state( + updated, + identity=identity, + stage=f"http_review_{action}", ) - else: - review_fields: tuple[object, ...] = (request.reason,) - if text_edit: - review_fields = (*stored_text_fields(stored_title, stored_text, stored_summary), request.reason) - if carries_credential_material(*review_fields): - return _vnext_public_error_response( - status_code=400, detail="vNext memory review text carries credential material" + if action == "assign_project" and request.project_id is not None: + store.create_edge( + { + "from_type": "memory", + "from_id": str(memory_id), + "to_type": "project", + "to_id": request.project_id, + "edge_type": "belongs_to_project", + "confidence": 1.0, + "explanation": "Assigned from live /vnext memory review.", + "created_by": "user", + "metadata_json": {"review_action": action}, + }, + actor_type=actor_type, ) - - updated = store.update_memory(memory_id=str(memory_id), patch=patch, actor_type=actor_type) - if action in ("accept", "edit", "promote"): - memory_service.refresh_memory_derived_state( - updated, - identity=identity, - stage=f"http_review_{action}", - ) - if action == "assign_project" and request.project_id is not None: - store.create_edge( + store.append_revision( { - "from_type": "memory", - "from_id": str(memory_id), - "to_type": "project", - "to_id": request.project_id, - "edge_type": "belongs_to_project", - "confidence": 1.0, - "explanation": "Assigned from live /vnext memory review.", - "created_by": "user", - "metadata_json": {"review_action": action}, + "memory_id": str(memory_id), + "memory_key": str(updated["memory_key"]), + "previous_value": existing.get("value"), + "new_value": updated.get("value"), + "source_event_ids": updated.get("source_event_ids"), + "revision_type": revision_type, + "action": f"memory_review_{action}", + "text_before": existing.get("canonical_text"), + "text_after": str(updated.get("canonical_text", "")), + "reason": review_reason or f"vNext workspace memory review action: {action}", + "actor_type": actor_type, + "actor_id": actor_id, + "metadata_json": {"action": action, "project_id": request.project_id}, }, actor_type=actor_type, ) - store.append_revision( - { - "memory_id": str(memory_id), - "memory_key": str(updated["memory_key"]), - "previous_value": existing.get("value"), - "new_value": updated.get("value"), - "source_event_ids": updated.get("source_event_ids"), - "revision_type": revision_type, - "action": f"memory_review_{action}", - "text_before": existing.get("canonical_text"), - "text_after": str(updated.get("canonical_text", "")), - "reason": review_reason or f"vNext workspace memory review action: {action}", - "actor_type": actor_type, - "actor_id": actor_id, - "metadata_json": {"action": action, "project_id": request.project_id}, - }, - actor_type=actor_type, - ) - review_event = { - "accept": "review.item_accepted", - "promote": "review.item_accepted", - "reject": "review.item_rejected", - "edit": "review.item_edited", - "private": "review.item_edited", - "assign_project": "review.item_edited", - }[action] - append_event( - store, - event_type=review_event, - actor_type=actor_type, - actor_id=actor_id, - target_type="memory", - target_id=str(memory_id), - payload={"action": action, "project_id": request.project_id}, - ) - # The row this route hands back is held to the caller's read fence: a memory that cites an archived source (or - # one above the caller's ceiling) is not returned with the quote it saved. - updated = SavedProvenanceReader(store, fence=SourceReadFence.for_identity(identity)).memory(updated) + review_event = { + "accept": "review.item_accepted", + "promote": "review.item_accepted", + "reject": "review.item_rejected", + "edit": "review.item_edited", + "private": "review.item_edited", + "assign_project": "review.item_edited", + }[action] + append_event( + store, + event_type=review_event, + actor_type=actor_type, + actor_id=actor_id, + target_type="memory", + target_id=str(memory_id), + payload={"action": action, "project_id": request.project_id}, + ) + # The row this route hands back is held to the caller's read fence: a memory that cites an archived source (or + # one above the caller's ceiling) is not returned with the quote it saved. + updated = SavedProvenanceReader(store, fence=SourceReadFence.for_identity(identity)).memory(updated) + + except Exception as exc: + from alicebot_api.vnext_label_writes import label_error_response + + mapped = label_error_response(exc) + if mapped is None: + raise + status, detail, retry_after = mapped + headers = {"Retry-After": retry_after} if retry_after else None + return JSONResponse(status_code=status, content={"detail": detail}, headers=headers) _persist_vnext_deferred_embeddings( database_url=settings.database_url, diff --git a/apps/api/src/alicebot_api/sqlite_store.py b/apps/api/src/alicebot_api/sqlite_store.py index 3fe8eaec1..2fd996f66 100644 --- a/apps/api/src/alicebot_api/sqlite_store.py +++ b/apps/api/src/alicebot_api/sqlite_store.py @@ -23,6 +23,8 @@ import itertools import json import sqlite3 + +from alicebot_api.vnext_label_writes import takes_label_lock from collections.abc import Iterator, Mapping, Sequence from contextlib import contextmanager from datetime import datetime @@ -705,6 +707,7 @@ def list_memory_events( source_inventory = _source_inventory prunable_sources = _prunable_sources + @takes_label_lock def create_source(self, source: JsonObject, *, actor_type: str = "system") -> VNextRow: source_id = _new_id(source.get("id")) self._execute( @@ -763,6 +766,7 @@ def create_source(self, source: JsonObject, *, actor_type: str = "system") -> VN create_browser_clip_capability = _browser_clip_create_capability consume_browser_clip_capability = _browser_clip_consume_capability + @takes_label_lock def get_or_create_source( self, source: JsonObject, @@ -915,6 +919,7 @@ def get_source_by_dedupe_key(self, dedupe_key: str) -> VNextRow | None: (dedupe_key, self.user_id), ) + @takes_label_lock def update_source( self, *, @@ -1020,6 +1025,9 @@ def update_source( target_id=row["id"], payload={"operation": "update", "changes": patch}, ) + from alicebot_api.vnext_label_writes import propagate_after_write + + propagate_after_write(self, kind="source", before=current, after=row) return row def create_source_chunk(self, chunk: JsonObject, *, actor_type: str = "system") -> VNextRow: diff --git a/apps/api/src/alicebot_api/vnext_label_writes.py b/apps/api/src/alicebot_api/vnext_label_writes.py index faee03beb..6df1e210a 100644 --- a/apps/api/src/alicebot_api/vnext_label_writes.py +++ b/apps/api/src/alicebot_api/vnext_label_writes.py @@ -6,22 +6,29 @@ from __future__ import annotations -from collections.abc import Iterator, Mapping +import json +import re +from collections.abc import Iterator, Mapping, Sequence from contextlib import contextmanager from functools import wraps from typing import Any from alicebot_api.vnext_agent_control import RESTRICTED_DOMAINS +from alicebot_api.vnext_derived_domain_backfill import DerivedDomainRepairError, require_changed from alicebot_api.vnext_derived_labels import ( MARKER_KEYS, + PROPAGATION_BOUND, + SENSITIVITY_RANK, dependencies_of, + LabelPropagationTooLarge, generation_domain, + identifier, is_derived, labels_raised_payload, settle_labels, ) from alicebot_api.vnext_event_log import build_event_log_record, integrity_hash_for_event -from alicebot_api.vnext_project_scope import project_scope_identity +from alicebot_api.vnext_project_scope import project_scope_identity, resolve_project_scope, source_project_scope from alicebot_api.vnext_repositories import JsonObject LABEL_METADATA_KEYS = ("project_scope", "project_floor", "derived_from") @@ -41,6 +48,12 @@ class LabelLockOrderError(RuntimeError): """A label write ran outside a transaction, or it took locks in the wrong order.""" +RETRYABLE_DETAIL = ( + "the label change was not applied because another change was running; nothing was changed; try again" +) +REFUSED_DETAIL = "the label change could not be applied to every dependent row; nothing was changed" + + def takes_label_lock(fn: Any) -> Any: """Take the shared label lock before a store method writes or row-locks a label table.""" @@ -89,8 +102,6 @@ def merge_protected_metadata( for key in LABEL_METADATA_KEYS: if key in stored_map: result[key] = stored_map[key] - else: - result.pop(key, None) for key in MARKER_KEYS: if key not in patch_map and key in stored_map: result[key] = stored_map[key] @@ -202,6 +213,408 @@ def apply_insert_floor(store: Any, kind: str, payload: Mapping[str, object]) -> return updated, event +def _sqlite(store: Any) -> bool: + return type(getattr(store, "conn", None)).__module__.startswith("sqlite3") + + +def _compact_id(value: object) -> str: + return re.sub(r"[-{}\s]", "", str(value).lower()) + + +def should_propagate(kind: str, before: Mapping[str, object] | None, after: Mapping[str, object] | None) -> bool: + """True when a relabel must walk dependants. A pure loosening does not.""" + + if not before or not after: + return False + old_domain = str(before.get("domain") or "unknown") + new_domain = str(after.get("domain") or "unknown") + old_sensitivity = str(before.get("sensitivity") or "unknown") + new_sensitivity = str(after.get("sensitivity") or "unknown") + rank_up = SENSITIVITY_RANK.get(new_sensitivity, 2) > SENSITIVITY_RANK.get(old_sensitivity, 2) + domain_move = new_domain in RESTRICTED_DOMAINS and new_domain != old_domain + if kind == "source": + old_scope = source_project_scope(before) + new_scope = source_project_scope(after) + else: + old_scope = resolve_project_scope(before).values + new_scope = resolve_project_scope(after).values + scope_move = project_scope_identity(old_scope) != project_scope_identity(new_scope) + return bool(rank_up or domain_move or scope_move) + + +def acquire_exclusive_label_lock(store: Any) -> None: + """Take L exclusive. On Postgres, wait at most 3 seconds.""" + + if _sqlite(store): + store.lock_label_writes(exclusive=True) + return + with store.conn.cursor() as cur: + cur.execute("SELECT current_setting('lock_timeout') AS lock_timeout") + row = cur.fetchone() + previous = row["lock_timeout"] if isinstance(row, Mapping) else row[0] + cur.execute("SET LOCAL lock_timeout = '3s'") + try: + store.lock_label_writes(exclusive=True) + finally: + try: + cur.execute("SELECT set_config('lock_timeout', %s, true)", (str(previous),)) + except Exception: + # A lock timeout aborts the transaction. Rollback drops the local setting. + pass + + +def list_dependants(store: Any, ids: Sequence[str]) -> list[dict[str, object]]: + """Rows whose recorded text may name one of ``ids``. The caller filters exactly.""" + + wanted = [str(item) for item in ids if str(item)] + if not wanted: + return [] + compacts = [_compact_id(item) for item in wanted if len(_compact_id(item)) >= 8] + if not compacts: + compacts = [_compact_id(item) for item in wanted if _compact_id(item)] + found: list[dict[str, object]] = [] + if _sqlite(store): + found.extend(_sqlite_dependants(store, "memories", "memory", compacts, wanted, with_value=True)) + found.extend(_sqlite_dependants(store, "open_loops", "open_loop", compacts, wanted, with_value=False)) + else: + found.extend(_postgres_dependants(store, "memories", "memory", compacts, with_value=True)) + found.extend(_postgres_dependants(store, "open_loops", "open_loop", compacts, with_value=False)) + found.extend(_postgres_dependants(store, "generated_artifacts", "artifact", compacts, with_value=False)) + found.extend(_postgres_dependants(store, "projects", "project", compacts, with_value=False)) + return found + + +def _like_clause(column: str, count: int, *, qmark: bool) -> tuple[str, list[str]]: + mark = "?" if qmark else "%s" + parts = [] + params: list[str] = [] + for _ in range(count): + parts.append( + "replace(replace(replace(replace(lower(coalesce(" + + column + + ",'')),'-',''),'{',''),'}',''),' ','') LIKE " + + mark + ) + return " OR ".join(parts), params + + +def _sqlite_dependants(store: Any, table: str, kind: str, compacts: Sequence[str], raw_ids: Sequence[str], *, with_value: bool) -> list[dict[str, object]]: + extra = ", value, project_id" if with_value else ", NULL AS value, project_id, source_id, memory_id" + if table == "memories": + extra = ", value, project_id, NULL AS source_id, NULL AS memory_id" + text_clause, _ = _like_clause("metadata_json", len(compacts), qmark=True) + params: list[object] = [store.user_id, *[f"%{item}%" for item in compacts]] + value_sql = "" + if with_value: + value_clause, _ = _like_clause("value", len(compacts), qmark=True) + value_sql = f" OR {value_clause}" + params.extend(f"%{item}%" for item in compacts) + column_sql = "" + if table == "open_loops": + marks = ",".join("?" for _ in raw_ids) + column_sql = f" OR source_id IN ({marks}) OR memory_id IN ({marks})" + params.extend(raw_ids) + params.extend(raw_ids) + rows = store._fetch_all( + f""" + SELECT id, user_id, domain, sensitivity, metadata_json{extra} + FROM {table} + WHERE user_id = ? AND ({text_clause}{value_sql}{column_sql}) + """, + tuple(params), + ) + for row in rows: + row["kind"] = kind + return rows + + +def _postgres_dependants(store: Any, table: str, kind: str, compacts: Sequence[str], *, with_value: bool) -> list[dict[str, object]]: + extra = "" + if table == "memories": + extra = ", value, project_id" + elif table == "open_loops": + extra = ", NULL::jsonb AS value, project_id, source_id, memory_id" + elif table == "generated_artifacts": + extra = ", NULL::jsonb AS value, artifact_type" + else: + extra = ", NULL::jsonb AS value" + text_clause, _ = _like_clause("metadata_json::text", len(compacts), qmark=False) + params: list[object] = [f"%{item}%" for item in compacts] + value_sql = "" + if with_value: + value_clause, _ = _like_clause("value::text", len(compacts), qmark=False) + value_sql = f" OR {value_clause}" + params.extend(f"%{item}%" for item in compacts) + rows = store._fetch_all( + f""" + SELECT id::text AS id, user_id::text AS user_id, domain, sensitivity, metadata_json{extra} + FROM {table} + WHERE ({text_clause}{value_sql}) + """, + tuple(params), + ) + for row in rows: + row["kind"] = kind + return rows + + +def _depends_on(row: Mapping[str, object], frontier: set[str]) -> bool: + kind = str(row.get("kind") or "") + try: + deps = dependencies_of(kind, row) + except ValueError: + return False + return any(identifier(dep_id) in frontier or _compact_id(dep_id) in frontier for _dep_kind, dep_id in deps) + + +def walk_dependants(store: Any, roots: Sequence[str]) -> list[dict[str, object]]: + """Exact dependants of ``roots``, including later rows in the chain.""" + + frontier = {identifier(item) for item in roots} + frontier |= {_compact_id(item) for item in roots} + seen = set(frontier) + found: list[dict[str, object]] = [] + pending = [str(item) for item in roots] + while pending: + if len(seen) > PROPAGATION_BOUND: + raise LabelPropagationTooLarge(f"label propagation stopped after {PROPAGATION_BOUND} rows") + batch = pending[:200] + pending = pending[200:] + matched: list[dict[str, object]] = [] + batch_ids = {identifier(item) for item in batch} | {_compact_id(item) for item in batch} + for row in list_dependants(store, batch): + row_id = identifier(row.get("id")) + if row_id in seen or _compact_id(row.get("id")) in seen: + continue + if not _depends_on(row, batch_ids): + continue + seen.add(row_id) + seen.add(_compact_id(row.get("id"))) + matched.append(row) + found.extend(matched) + pending.extend(str(row["id"]) for row in matched) + return found + + +def _label_fields(row: Mapping[str, object]) -> tuple[str, str, tuple[str, ...], tuple[str, ...]]: + metadata = row.get("metadata_json") + meta = metadata if isinstance(metadata, Mapping) else {} + scope = meta.get("project_scope", ()) + floor = meta.get("project_floor", ()) + return ( + str(row.get("domain") or "unknown"), + str(row.get("sensitivity") or "unknown"), + tuple(scope) if isinstance(scope, (list, tuple)) else (), + tuple(floor) if isinstance(floor, (list, tuple)) else (), + ) + + +def write_settled_label( + store: Any, + *, + kind: str, + row_id: str, + domain: str, + sensitivity: str, + metadata: Mapping[str, object], + project_id: str | None, + expected_domain: str, + expected_sensitivity: str, +) -> None: + """Label-only update. A statement that changes no row refuses the whole relabel.""" + + table = {"memory": "memories", "open_loop": "open_loops", "artifact": "generated_artifacts", "project": "projects"}[kind] + blob = json.dumps(dict(metadata)) + if _sqlite(store): + project_sql = ", project_id = ?" if table in {"memories", "open_loops"} else "" + params: list[object] = [domain, sensitivity, blob] + if project_sql: + params.append(project_id) + params.extend([str(row_id), store.user_id, expected_domain, expected_sensitivity]) + cursor = store._execute( + f""" + UPDATE {table} + SET domain = ?, sensitivity = ?, metadata_json = ?{project_sql} + WHERE id = ? AND user_id = ? AND domain = ? AND sensitivity = ? + """, + tuple(params), + ) + require_changed(int(cursor.rowcount), table, str(row_id)) + return + project_sql = ", project_id = %s" if table in {"memories", "open_loops"} else "" + params = [domain, sensitivity, blob] + if project_sql: + params.append(project_id) + params.extend([str(row_id), expected_domain, expected_sensitivity]) + store._fetch_one( + "write_settled_label", + f""" + UPDATE {table} + SET domain = %s, sensitivity = %s, metadata_json = %s::jsonb{project_sql} + WHERE id = %s::uuid AND domain = %s AND sensitivity = %s + RETURNING id + """, + tuple(params), + ) + + +def propagate(store: Any, changed: Sequence[tuple[str, str]], *, cause: str) -> int: + """Recompute dependants of ``changed`` rows and write the ones that rise.""" + + store.lock_label_writes(exclusive=True) + roots = [row_id for _kind, row_id in changed] + affected = walk_dependants(store, roots) + if not affected: + return 0 + nodes: list[dict[str, object]] = [] + reader = getattr(store, "read_label_rows", None) + if callable(reader): + for kind, row_id in changed: + for row in reader(kind, [row_id]): + copied = dict(row) + copied["kind"] = kind + nodes.append(copied) + needed: dict[str, list[str]] = {} + for row in affected: + for dep_kind, dep_id in dependencies_of(str(row.get("kind")), row): + needed.setdefault(dep_kind, []).append(dep_id) + if callable(reader): + for dep_kind, dep_ids in needed.items(): + for row in reader(dep_kind, dep_ids): + copied = dict(row) + copied["kind"] = dep_kind + nodes.append(copied) + for row in affected: + nodes.append(dict(row)) + settled = settle_labels(nodes) + written = 0 + for row in affected: + label = settled.by_stored(str(row.get("kind")), str(row.get("id"))) + if label.unverified: + continue + previous = _label_fields(row) + current = (label.domain, label.sensitivity, tuple(label.project_scope), tuple(label.project_floor)) + if ( + previous[0] == current[0] + and previous[1] == current[1] + and project_scope_identity(previous[2]) == project_scope_identity(current[2]) + and project_scope_identity(previous[3]) == project_scope_identity(current[3]) + ): + continue + metadata = dict(row.get("metadata_json")) if isinstance(row.get("metadata_json"), Mapping) else {} + metadata["project_scope"] = list(label.project_scope) + metadata["project_floor"] = list(label.project_floor) + project_id = label.project_scope[0] if len(project_scope_identity(label.project_scope)) == 1 else None + if project_id is not None: + try: + from uuid import UUID + + UUID(str(project_id)) + except ValueError: + project_id = None + write_settled_label( + store, + kind=str(row.get("kind")), + row_id=str(row.get("id")), + domain=label.domain, + sensitivity=label.sensitivity, + metadata=metadata, + project_id=project_id, + expected_domain=previous[0], + expected_sensitivity=previous[1], + ) + event = build_event_log_record( + event_type=f"{label.kind}.labels_raised", + actor_type="system", + target_type=label.kind, + target_id=str(row.get("id")), + payload=labels_raised_payload( + cause=cause, + previous={ + "domain": previous[0], + "sensitivity": previous[1], + "project_scope": list(previous[2]), + "project_floor": list(previous[3]), + }, + new={ + "domain": label.domain, + "sensitivity": label.sensitivity, + "project_scope": list(label.project_scope), + "project_floor": list(label.project_floor), + }, + ), + ) + append = getattr(store, "append_event", None) + if callable(append): + append(event) + written += 1 + return written + + +def propagate_after_write(store: Any, *, kind: str, before: Mapping[str, object] | None, after: Mapping[str, object] | None, cause: str = "input_relabelled") -> int: + if not should_propagate(kind, before, after) or after is None: + return 0 + return propagate(store, [(kind, str(after.get("id")))], cause=cause) + + +def count_rows_hidden_by_scope_move(store: Any, source: Mapping[str, object], new_scope: Sequence[str]) -> int: + """How many derived rows a project-bound key would lose if ``source`` moved.""" + + source_id = str(source.get("id") or "") + if not source_id: + return 0 + affected = walk_dependants(store, [source_id]) + if not affected: + return 0 + current = dict(source) + current["kind"] = "source" + moved = dict(source) + moved["kind"] = "source" + metadata = dict(source.get("metadata_json")) if isinstance(source.get("metadata_json"), Mapping) else {} + metadata["project_scope"] = list(new_scope) + moved["metadata_json"] = metadata + before = settle_labels([current, *[dict(row) for row in affected]]) + after = settle_labels([moved, *[dict(row) for row in affected]]) + hidden = 0 + for row in affected: + old = before.by_stored(str(row.get("kind")), str(row.get("id"))) + new = after.by_stored(str(row.get("kind")), str(row.get("id"))) + old_ids = set(project_scope_identity(old.project_scope)) + new_ids = set(project_scope_identity(new.project_scope)) + if old_ids - new_ids: + hidden += 1 + return hidden + + +def raise_source_to_replacement(store: Any, old: Mapping[str, object], replacement: Mapping[str, object]) -> None: + """Raise a retiring source to the replacement when that label is stricter, then propagate.""" + + old_domain = str(old.get("domain") or "unknown") + new_domain = str(replacement.get("domain") or "unknown") + domain = new_domain if new_domain in RESTRICTED_DOMAINS and old_domain not in RESTRICTED_DOMAINS else old_domain + sensitivity = str(old.get("sensitivity") or "unknown") + replacement_sensitivity = str(replacement.get("sensitivity") or "unknown") + if SENSITIVITY_RANK.get(replacement_sensitivity, 2) > SENSITIVITY_RANK.get(sensitivity, 2): + sensitivity = replacement_sensitivity + if domain == old_domain and sensitivity == str(old.get("sensitivity") or "unknown"): + return + store.update_source( + source_id=str(old.get("id")), + patch={"domain": domain, "sensitivity": sensitivity}, + actor_type="system", + ) + + +def label_error_response(exc: BaseException) -> tuple[int, str, str | None] | None: + """``(status, detail, retry_after)`` for a relabel failure, or None.""" + + if isinstance(exc, (LabelPropagationTooLarge, DerivedDomainRepairError)): + return 409, REFUSED_DETAIL, None + if type(exc).__name__ in {"LockNotAvailable", "DeadlockDetected", "SerializationFailure"}: + return 503, RETRYABLE_DETAIL, "2" + return None + + def remember_floor_event(store: Any, event: JsonObject | None, target_id: object) -> None: """Append an insert-floor event once the row id is known.""" @@ -222,7 +635,16 @@ def remember_floor_event(store: Any, event: JsonObject | None, target_id: object "LabelLockOrderError", "apply_insert_floor", "merge_protected_metadata", + "REFUSED_DETAIL", + "RETRYABLE_DETAIL", + "acquire_exclusive_label_lock", + "count_rows_hidden_by_scope_move", + "label_error_response", + "propagate", + "propagate_after_write", + "raise_source_to_replacement", "remember_floor_event", + "should_propagate", "takes_label_lock", "without_insert_floor", ] diff --git a/apps/api/src/alicebot_api/vnext_store.py b/apps/api/src/alicebot_api/vnext_store.py index 92fe7a7c0..e83c801f5 100644 --- a/apps/api/src/alicebot_api/vnext_store.py +++ b/apps/api/src/alicebot_api/vnext_store.py @@ -23,6 +23,7 @@ ) from alicebot_api.vnext_entity_names import ENTITY_IMMUTABLE_PATCH_FIELDS, normalize_entity_name from alicebot_api.vnext_json import json_safe +from alicebot_api.vnext_label_writes import takes_label_lock from alicebot_api.vnext_project_scope import ( expose_memory_project_scope, project_scope_identity, @@ -1016,6 +1017,7 @@ def list_sources( (domains, domains, sensitivity_allowed, sensitivity_allowed, limit), ) + @takes_label_lock def create_source(self, source: JsonObject, *, actor_type: str = "system") -> VNextRow: row = self._fetch_one( "create_source", @@ -1088,6 +1090,7 @@ def create_source(self, source: JsonObject, *, actor_type: str = "system") -> VN ) return row + @takes_label_lock def get_or_create_source( self, source: JsonObject, @@ -1238,6 +1241,7 @@ def get_source_by_dedupe_key(self, dedupe_key: str) -> VNextRow | None: (dedupe_key,), ) + @takes_label_lock def update_source(self, *, source_id: str, patch: JsonObject, actor_type: str = "system") -> VNextRow: with self.conn.cursor() as cur: cur.execute( @@ -1350,8 +1354,12 @@ def update_source(self, *, source_id: str, patch: JsonObject, actor_type: str = target_id=row["id"], payload={"operation": "update", "changes": patch}, ) + from alicebot_api.vnext_label_writes import propagate_after_write + + propagate_after_write(self, kind="source", before=current, after=row) return row + @takes_label_lock def delete_source(self, *, source_id: str, actor_type: str = "system") -> VNextRow: row = self._fetch_one( "delete_source", @@ -1706,6 +1714,7 @@ def search_sources( expire_edge = _graph_expire_edge + @takes_label_lock def create_project(self, project: JsonObject, *, actor_type: str = "system") -> VNextRow: row = self._fetch_one( "create_project", @@ -1767,6 +1776,7 @@ def get_project(self, project_id: str) -> VNextRow | None: (project_id,), ) + @takes_label_lock def get_project_for_update(self, project_id: str) -> VNextRow | None: """Lock a project while an artifact review applies its state.""" @@ -1824,6 +1834,7 @@ def list_projects( ), ) + @takes_label_lock def update_project(self, *, project_id: str, patch: JsonObject, actor_type: str = "system") -> VNextRow: row = self._fetch_one( "update_project", @@ -2026,7 +2037,11 @@ def update_person(self, *, person_id: str, patch: JsonObject, actor_type: str = update_open_loop_status = _graph_update_open_loop_status + @takes_label_lock def create_artifact(self, artifact: JsonObject, *, actor_type: str = "system") -> VNextRow: + from alicebot_api.vnext_label_writes import apply_insert_floor, remember_floor_event + + artifact, floor_event = apply_insert_floor(self, "artifact", artifact) row = self._fetch_one( "create_artifact", f""" @@ -2087,8 +2102,10 @@ def create_artifact(self, artifact: JsonObject, *, actor_type: str = "system") - target_id=row["id"], payload={"operation": "create", "artifact_type": str(row["artifact_type"])}, ) + remember_floor_event(self, floor_event, row["id"]) return row + @takes_label_lock def upsert_artifact_by_workflow_digest( self, artifact: JsonObject, @@ -2116,6 +2133,9 @@ def upsert_artifact_by_workflow_digest( ) if existing is not None: return existing + from alicebot_api.vnext_label_writes import apply_insert_floor, remember_floor_event + + artifact, floor_event = apply_insert_floor(self, "artifact", artifact) metadata_value = artifact.get("metadata_json") metadata: JsonObject = dict(metadata_value) if isinstance(metadata_value, dict) else {} metadata.update( @@ -2196,6 +2216,7 @@ def upsert_artifact_by_workflow_digest( target_id=row["id"], payload={"operation": "create", "artifact_type": str(row["artifact_type"])}, ) + remember_floor_event(self, floor_event, row["id"]) return row def get_artifact(self, artifact_id: str) -> VNextRow | None: @@ -2208,6 +2229,7 @@ def get_artifact(self, artifact_id: str) -> VNextRow | None: (artifact_id,), ) + @takes_label_lock def get_artifact_for_update(self, artifact_id: str) -> VNextRow | None: """Lock one persisted artifact before an authorized side effect.""" @@ -2339,6 +2361,7 @@ def find_artifact_by_workflow_digest( ), ) + @takes_label_lock def update_artifact_status( self, *, diff --git a/apps/api/src/alicebot_api/vnext_stores/postgres/embedding_cas.py b/apps/api/src/alicebot_api/vnext_stores/postgres/embedding_cas.py index 2113e11ca..9c41caf96 100644 --- a/apps/api/src/alicebot_api/vnext_stores/postgres/embedding_cas.py +++ b/apps/api/src/alicebot_api/vnext_stores/postgres/embedding_cas.py @@ -15,6 +15,7 @@ from alicebot_api.vnext_recall_visibility import POSTGRES_UNEXPIRED_SQL from alicebot_api.vnext_repositories import JsonObject from alicebot_api.vnext_stores.postgres.columns import MEMORY_COLUMNS +from alicebot_api.vnext_label_writes import takes_label_lock VNextRow = dict[str, object] @@ -141,6 +142,7 @@ def _vector_literal(vector: list[float]) -> str: return "[" + ",".join(repr(value) for value in values) + "]" +@takes_label_lock def update_memory_embedding( self, *, @@ -202,6 +204,7 @@ def update_memory_embedding( ) +@takes_label_lock def clear_memory_embedding(self, *, memory_id: str) -> VNextRow | None: """Invalidate content-derived vector state before a text mutation. diff --git a/apps/api/src/alicebot_api/vnext_stores/postgres/events_revisions.py b/apps/api/src/alicebot_api/vnext_stores/postgres/events_revisions.py index 788d462ec..6597b3560 100644 --- a/apps/api/src/alicebot_api/vnext_stores/postgres/events_revisions.py +++ b/apps/api/src/alicebot_api/vnext_stores/postgres/events_revisions.py @@ -10,6 +10,7 @@ from alicebot_api.vnext_repositories import JsonObject from alicebot_api.vnext_stores.postgres.columns import EVENT_LOG_COLUMNS, REVISION_COLUMNS from alicebot_api.vnext_stores.postgres.primitives import _json_list, _json_object, _json_safe +from alicebot_api.vnext_label_writes import takes_label_lock VNextRow = dict[str, object] @@ -282,6 +283,7 @@ def count_events( return int(cast(int, row["count"])) +@takes_label_lock def append_revision(self, revision: JsonObject, *, actor_type: str = "system") -> VNextRow: row = self._fetch_one( "append_revision", diff --git a/apps/api/src/alicebot_api/vnext_stores/postgres/graph_open_loops.py b/apps/api/src/alicebot_api/vnext_stores/postgres/graph_open_loops.py index b7249c4f3..3c4faed08 100644 --- a/apps/api/src/alicebot_api/vnext_stores/postgres/graph_open_loops.py +++ b/apps/api/src/alicebot_api/vnext_stores/postgres/graph_open_loops.py @@ -22,6 +22,7 @@ _json_object, _sorted_field_names, ) +from alicebot_api.vnext_label_writes import takes_label_lock from alicebot_api.vnext_stores.postgres.query_predicates import ( _OPEN_LOOP_SCOPE_EVENT_TIME_SQL, _OPEN_LOOP_SCOPE_PEOPLE_SQL, @@ -807,7 +808,11 @@ def update_belief_status( ) return row +@takes_label_lock def create_open_loop(self, loop: JsonObject, *, actor_type: str = "system") -> VNextRow: + from alicebot_api.vnext_label_writes import apply_insert_floor, remember_floor_event + + loop, floor_event = apply_insert_floor(self, "open_loop", loop) row = self._fetch_one( "create_open_loop", f""" @@ -885,6 +890,7 @@ def create_open_loop(self, loop: JsonObject, *, actor_type: str = "system") -> V target_id=row["id"], payload={"operation": "create", "fields": _sorted_field_names(loop)}, ) + remember_floor_event(self, floor_event, row["id"]) return row def upsert_open_loop_by_automation_digest( @@ -1198,6 +1204,7 @@ def list_open_loop_events( ), ) +@takes_label_lock def update_open_loop(self, *, loop_id: str, patch: JsonObject, actor_type: str = "system") -> VNextRow: row = self._fetch_one( "update_open_loop", @@ -1238,6 +1245,7 @@ def update_open_loop(self, *, loop_id: str, patch: JsonObject, actor_type: str = ) return row +@takes_label_lock def update_open_loop_status( self, *, diff --git a/apps/api/src/alicebot_api/vnext_stores/postgres/memory_lifecycle.py b/apps/api/src/alicebot_api/vnext_stores/postgres/memory_lifecycle.py index cd069e6e4..1cb1979fa 100644 --- a/apps/api/src/alicebot_api/vnext_stores/postgres/memory_lifecycle.py +++ b/apps/api/src/alicebot_api/vnext_stores/postgres/memory_lifecycle.py @@ -204,6 +204,7 @@ def upsert_memory_by_key(self, memory: JsonObject, *, actor_type: str = "system" raise return existing +@takes_label_lock def get_memory_for_update(self, memory_id: str) -> VNextRow | None: """Load and lock one memory for a review/lifecycle decision.""" return self._fetch_optional_one( @@ -217,6 +218,7 @@ def get_memory_for_update(self, memory_id: str) -> VNextRow | None: (memory_id,), ) +@takes_label_lock def get_memory_for_redaction(self, memory_id: str) -> VNextRow | None: """Lock a redaction target even after forget archived/tombstoned it.""" @@ -230,6 +232,7 @@ def get_memory_for_redaction(self, memory_id: str) -> VNextRow | None: (memory_id,), ) +@takes_label_lock def lock_project_update_artifacts_for_redaction(self, memory_id: str) -> list[VNextRow]: """Lock every artifact coupled to a candidate memory in UUID order.""" @@ -387,6 +390,7 @@ def list_memory_ids_with_embeddings(self, ids: "Sequence[str]") -> set[str]: ) return {str(row["id"]) for row in rows} +@takes_label_lock def update_memory_fact_keys(self, *, memory_id: str, fact_keys: str | None) -> VNextRow | None: """Store derived retrieval keys; the generated ``search_tsv`` column (migration ``20260707_0082``) re-indexes them at 'D' weight. @@ -428,6 +432,7 @@ def list_memories_missing_fact_keys(self, *, limit: int = 100, after_id: str | N def update_memory( self, *, memory_id: str, patch: JsonObject, actor_type: str = "system", label_write: bool = False ) -> VNextRow: + before_label = self.get_memory(str(memory_id)) refuse_updated_credential_activation(patch, lambda: self.get_memory(str(memory_id))) if "metadata_json" in patch and isinstance(patch.get("metadata_json"), dict): current = self._fetch_optional_one( @@ -521,6 +526,10 @@ def update_memory( target_id=row["id"], payload={"operation": "update", "changes": patch}, ) + if not label_write: + from alicebot_api.vnext_label_writes import propagate_after_write + + propagate_after_write(self, kind="memory", before=before_label, after=row) return row @contextmanager @@ -541,6 +550,7 @@ def _redaction_mode(self) -> Iterator[None]: # (set_config assignments are transactional). pass +@takes_label_lock def redact_memory_bundle( self, *, @@ -914,6 +924,7 @@ def redact_memory_bundle( "idempotent_replay": not bundle_changed, } +@takes_label_lock def redact_memory_content(self, *, memory_id: str, actor_type: str = "user") -> VNextRow: """Expunge a memory's content in place, keeping the skeleton. diff --git a/apps/api/src/alicebot_api/vnext_stores/sqlite/embedding_cas.py b/apps/api/src/alicebot_api/vnext_stores/sqlite/embedding_cas.py index a60a3fdc4..633b4e057 100644 --- a/apps/api/src/alicebot_api/vnext_stores/sqlite/embedding_cas.py +++ b/apps/api/src/alicebot_api/vnext_stores/sqlite/embedding_cas.py @@ -21,6 +21,7 @@ from alicebot_api.vnext_stores.sqlite.columns import MEMORY_COLUMNS from alicebot_api.vnext_stores.sqlite.query_predicates import _expiry_clause from alicebot_api.vnext_stores.sqlite.vector_scan import bump_embedding_stamp +from alicebot_api.vnext_label_writes import takes_label_lock VNextRow = dict[str, object] @@ -126,6 +127,7 @@ def _ensure_embedding_input_cut_sqlite(conn: sqlite3.Connection) -> None: ) +@takes_label_lock def update_memory_embedding( self, *, @@ -210,6 +212,7 @@ def update_memory_embedding( ) +@takes_label_lock def clear_memory_embedding(self, *, memory_id: str) -> VNextRow | None: """Invalidate an embedding derived from text that is about to change.""" if not self.conn.in_transaction: diff --git a/apps/api/src/alicebot_api/vnext_stores/sqlite/graph_open_loops.py b/apps/api/src/alicebot_api/vnext_stores/sqlite/graph_open_loops.py index 47368f076..cc43135e1 100644 --- a/apps/api/src/alicebot_api/vnext_stores/sqlite/graph_open_loops.py +++ b/apps/api/src/alicebot_api/vnext_stores/sqlite/graph_open_loops.py @@ -30,6 +30,7 @@ _utc_now_iso, _uuid_text, ) +from alicebot_api.vnext_label_writes import takes_label_lock from alicebot_api.vnext_stores.sqlite.query_predicates import ( _project_scope_value_sqlite, CTE_MATERIALIZED_HINT, @@ -518,7 +519,11 @@ def list_relationship_events(self, entity_id: str) -> list[VNextRow]: (str(entity_id), self.user_id), ) +@takes_label_lock def create_open_loop(self, loop: JsonObject, *, actor_type: str = "system") -> VNextRow: + from alicebot_api.vnext_label_writes import apply_insert_floor, remember_floor_event + + loop, floor_event = apply_insert_floor(self, "open_loop", loop) loop_id = _new_id(loop.get("id")) now = _utc_now_iso() self._execute( @@ -578,6 +583,7 @@ def create_open_loop(self, loop: JsonObject, *, actor_type: str = "system") -> V target_id=row["id"], payload={"operation": "create", "fields": _sorted_field_names(loop)}, ) + remember_floor_event(self, floor_event, row["id"]) return row def upsert_open_loop_by_automation_digest( @@ -969,6 +975,7 @@ def list_open_loop_events( tuple(params), ) +@takes_label_lock def update_open_loop(self, *, loop_id: str, patch: JsonObject, actor_type: str = "system") -> VNextRow: cursor = self._execute( """ @@ -1015,6 +1022,7 @@ def update_open_loop(self, *, loop_id: str, patch: JsonObject, actor_type: str = ) return row +@takes_label_lock def update_open_loop_status( self, *, diff --git a/apps/api/src/alicebot_api/vnext_stores/sqlite/memory_lifecycle.py b/apps/api/src/alicebot_api/vnext_stores/sqlite/memory_lifecycle.py index 88080782c..ccff5ca4d 100644 --- a/apps/api/src/alicebot_api/vnext_stores/sqlite/memory_lifecycle.py +++ b/apps/api/src/alicebot_api/vnext_stores/sqlite/memory_lifecycle.py @@ -298,6 +298,7 @@ def memory_redaction_bundle_is_exact(self, memory_id: str, artifact_ids: Sequenc def update_memory( self, *, memory_id: str, patch: JsonObject, actor_type: str = "system", label_write: bool = False ) -> VNextRow: + before_label = self.get_memory(str(memory_id)) refuse_updated_credential_activation(patch, lambda: self.get_memory(str(memory_id))) patch = _with_protected_metadata(self, memory_id, patch, label_write=label_write) # One clock reading for the write: an archive sets ``updated_at`` and ``deleted_at`` together. @@ -388,6 +389,10 @@ def update_memory( target_id=row["id"], payload={"operation": "update", "changes": patch}, ) + if not label_write: + from alicebot_api.vnext_label_writes import propagate_after_write + + propagate_after_write(self, kind="memory", before=before_label, after=row) return row def lock_graph_mutation(self) -> None: @@ -424,6 +429,7 @@ def list_memory_ids_with_embeddings(self, ids: "Sequence[str]") -> set[str]: present.update(str(row["id"]) for row in rows) return present +@takes_label_lock def update_memory_fact_keys(self, *, memory_id: str, fact_keys: str | None) -> VNextRow | None: """Store derived retrieval keys; the FTS sync triggers re-index them. @@ -481,6 +487,7 @@ def _redaction_mode(self) -> Iterator[None]: finally: self._execute("UPDATE redaction_mode SET enabled = 0 WHERE id = 1") +@takes_label_lock def redact_memory_bundle( self, *, @@ -720,6 +727,7 @@ def redact_memory_bundle( "idempotent_replay": not changed, } +@takes_label_lock def redact_memory_content(self, *, memory_id: str, actor_type: str = "user") -> VNextRow: """Expunge a memory's content in place, keeping the skeleton. diff --git a/apps/api/src/alicebot_api/vnext_stores/sqlite/source_retirement.py b/apps/api/src/alicebot_api/vnext_stores/sqlite/source_retirement.py index 12c1b2d0c..f4f429915 100644 --- a/apps/api/src/alicebot_api/vnext_stores/sqlite/source_retirement.py +++ b/apps/api/src/alicebot_api/vnext_stores/sqlite/source_retirement.py @@ -21,6 +21,7 @@ open_loop_source_reference_params, ) from alicebot_api.vnext_stores.sqlite.primitives import _utc_now_iso +from alicebot_api.vnext_label_writes import takes_label_lock REMOVAL_MARKER = "[removed by the owner]" @@ -395,6 +396,7 @@ def retire_dependents(self, source_id, *, now, scrub_candidates=False, citing_id 'memories_citing_replaced': retained} +@takes_label_lock def supersede_source(self, source_id, *, superseded_by, allow_looser_classification=False, dry_run=False): with self.savepoint(): old = self.get_source(source_id) @@ -413,6 +415,10 @@ def supersede_source(self, source_id, *, superseded_by, allow_looser_classificat # The sidecar goes first. A later rollback may lose proposals, which # can be generated again, but cannot leave retired evidence in it. prune_sleep_rows(self, {source_id}, dry_run=dry_run) + if not dry_run: + from alicebot_api.vnext_label_writes import raise_source_to_replacement + + raise_source_to_replacement(self, old, new) counts = retire_dependents(self, source_id, now=now) metadata = {**old['metadata_json'], 'superseded_by': superseded_by, 'superseded_at': now, 'supersede_reason': 'markdown_reimport'} @@ -476,6 +482,7 @@ def optimize_scrub_indexes(self): self._execute("INSERT INTO memories_fts(memories_fts) VALUES('optimize')") +@takes_label_lock def scrub_source(self, source_id, *, optimize=True, citing_ids=None): with self.savepoint(): self._execute("PRAGMA secure_delete=ON") diff --git a/tests/unit/test_sqlite_derived_labels_write_path.py b/tests/unit/test_sqlite_derived_labels_write_path.py index 680ef6de6..d6f90f646 100644 --- a/tests/unit/test_sqlite_derived_labels_write_path.py +++ b/tests/unit/test_sqlite_derived_labels_write_path.py @@ -103,6 +103,117 @@ def test_an_update_that_omits_a_marker_keeps_it(tmp_path: Path) -> None: assert stored["metadata_json"]["note"] == "changed" +def test_a_source_relabel_reaches_the_extracted_memory(tmp_path: Path) -> None: + path = tmp_path / "vault.sqlite3" + with _vault(path) as conn: + store = SQLiteVNextStore(conn, USER) + source = store.create_source( + { + "source_type": "note", + "title": "Visit", + "content_hash": "hash-propagate", + "domain": "unknown", + "sensitivity": "public", + "metadata_json": {"project_scope": [ALPHA]}, + } + ) + memory = store.create_memory( + { + "memory_key": "extracted", + "canonical_text": "A fact from the visit.", + "status": "active", + "domain": "unknown", + "sensitivity": "public", + "metadata_json": {"source_id": str(source["id"]), "project_scope": [ALPHA]}, + } + ) + store.update_source( + source_id=str(source["id"]), + patch={"domain": "health", "sensitivity": "confidential"}, + actor_type="user", + ) + stored = store.get_memory(str(memory["id"])) + assert stored is not None + assert stored["domain"] == "health" + assert stored["sensitivity"] == "confidential" + + +def test_a_candidate_loop_is_floored_from_its_source(tmp_path: Path) -> None: + path = tmp_path / "vault.sqlite3" + with _vault(path) as conn: + store = SQLiteVNextStore(conn, USER) + source = store.create_source( + { + "source_type": "note", + "title": "Tasks", + "content_hash": "hash-loop", + "domain": "health", + "sensitivity": "private", + "metadata_json": {"project_scope": [ALPHA]}, + } + ) + loop = store.create_open_loop( + { + "title": "Call the clinic", + "source_id": str(source["id"]), + "domain": "unknown", + "sensitivity": "public", + "metadata_json": { + "discovered_by": "vnext_daily_brief", + "source_id": str(source["id"]), + "project_scope": [ALPHA], + }, + } + ) + assert loop["domain"] == "health" + assert loop["sensitivity"] == "private" + + +def test_supersede_raises_a_citing_memory(tmp_path: Path) -> None: + path = tmp_path / "vault.sqlite3" + with _vault(path) as conn: + store = SQLiteVNextStore(conn, USER) + old = store.create_source( + { + "source_type": "markdown", + "title": "Note", + "content_hash": "hash-old", + "raw_path": "notes/one.md", + "connector_name": "markdown_folder", + "domain": "unknown", + "sensitivity": "public", + "metadata_json": {"relative_path": "notes/one.md"}, + } + ) + new = store.create_source( + { + "source_type": "markdown", + "title": "Note", + "content_hash": "hash-new", + "raw_path": "notes/one.md", + "connector_name": "markdown_folder", + "domain": "health", + "sensitivity": "confidential", + "metadata_json": {"relative_path": "notes/one.md"}, + } + ) + memory = store.create_memory( + { + "memory_key": "cited", + "canonical_text": "A fact.", + "status": "active", + "domain": "unknown", + "sensitivity": "public", + "metadata_json": {"source_id": str(old["id"])}, + } + ) + store.supersede_source(str(old["id"]), superseded_by=str(new["id"])) + stored = store.get_memory(str(memory["id"])) + assert stored is not None + assert stored["domain"] == "health" + assert stored["sensitivity"] == "confidential" + + def test_merge_protected_metadata_keeps_label_keys_unless_the_write_is_a_relabel() -> None: stored = { "consolidation": {"cluster_member_ids": ["m"]}, diff --git a/tests/unit/test_store_events_revisions_split.py b/tests/unit/test_store_events_revisions_split.py index 2ca60ca6d..dde8c9474 100644 --- a/tests/unit/test_store_events_revisions_split.py +++ b/tests/unit/test_store_events_revisions_split.py @@ -50,7 +50,7 @@ "list_events_for_source_trace": "20f22e3b75c3612c02c4242bc973295b2535b01f95e5451a0a0bff1196f187c3", "list_project_update_events": "2bd457ee19535f203da5c31e557387f7717fefc27cbef2c546a62bbad74962d3", "count_events": "e740e4b09ecfda973ef6b84acd0ea808b26d118faf6984e4057d7e104e595fb5", - "append_revision": "a243976fc27fa6e7ae33c15169d407902e3258b842300df4705629e443a75740", + "append_revision": "6070b96a01a50072e4e898bfb64ddbbf253be0478fe63205f760d0fe016a109a", "list_revisions": "10a485935b59bda1fcb33b36ba48b8d86376b9fd180bf9ebf6358b5dfbd24f55", }, "sqlite": { @@ -148,14 +148,15 @@ # Re-minted for the per-file importer savepoint (2026-10-02): ``savepoint`` is appended last on both façades. # Previous receipt: 650e2e0ff088d67c... Proof: the class key list equals the list at origin/main 040a2a10 # plus ``savepoint`` before ``__dict__``, with every other key in the same order. - "postgres": "3d2cd1d2fbb766ea4b5f9bf700fe6c0ccdc5fea9705092f82ead06c3ab635e50", + "postgres": "dda7be1a316b3c525195f4682a608d0bab24b57f635db2a7aebc0c7ea187fc68", # Re-minted for the merge of #500 and #502: ``check_literal_match_query``. # Re-minted again for per-project memory S2 (2026-10-02): the two single-scan partition reads # ``list_memories_view_partitions`` and ``list_open_loops_view_partitions``, SQLite only on purpose # (the Postgres runtime resolves no project view). # Re-minted again for the per-file importer savepoint (2026-10-02), the same appended method. # Previous receipt: 365b7a01acf7a8b5... Proof: as for postgres, one added key and no other change. - "sqlite": "65301a20344d20bd6e20e6717f1f3db3fd4b16611964425d1d3f98736326f393", + # Re-minted for the derived-label lock: ``lock_label_writes`` and ``read_label_rows`` on both facades. + "sqlite": "3212a93f2011cecddb6b0002a3f9b3abe37819d3582b2a1173846871981a5ff9", } EXPECTED_SUPPORT_AST_SHA256 = { "postgres_columns": { From b192f9d927c2a13b892532b8b7f58e693c41af92 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 17:51:41 +0000 Subject: [PATCH 04/90] Filter locked report inputs by every project. A key bound to a project builds a brief, connection report, contradiction report, or project update only from rows whose scope and floor are both inside its binding. --- apps/api/src/alicebot_api/vnext_brain.py | 26 ++++++-- .../api/src/alicebot_api/vnext_connections.py | 18 +++++- .../src/alicebot_api/vnext_contradictions.py | 22 +++++-- .../src/alicebot_api/vnext_derived_labels.py | 51 +++++++++++++++ apps/api/src/alicebot_api/vnext_projects.py | 21 +++++-- tests/unit/test_derived_label_input_filter.py | 63 +++++++++++++++++++ 6 files changed, 185 insertions(+), 16 deletions(-) create mode 100644 tests/unit/test_derived_label_input_filter.py diff --git a/apps/api/src/alicebot_api/vnext_brain.py b/apps/api/src/alicebot_api/vnext_brain.py index 2687e99ad..585d0e0ec 100644 --- a/apps/api/src/alicebot_api/vnext_brain.py +++ b/apps/api/src/alicebot_api/vnext_brain.py @@ -237,11 +237,18 @@ def _matches_report_scope( projects: tuple[str, ...], window_start: datetime, window_end: datetime, + all_of: tuple[str, ...] | None = None, ) -> bool: if projects: - row_scope = source_project_scope(row) if kind == "source" else resource_project_scope(row) - if not project_scopes_overlap(row_scope, projects): - return False + if all_of is not None: + from alicebot_api.vnext_derived_labels import input_admitted + + if not input_admitted(kind, row, all_of): + return False + else: + row_scope = source_project_scope(row) if kind == "source" else resource_project_scope(row) + if not project_scopes_overlap(row_scope, projects): + return False event_time = _row_event_time(row, kind=kind) return event_time is not None and window_start <= event_time < window_end @@ -267,6 +274,7 @@ def _windowed_rows( limit: int, store_scope_kwargs: dict[str, object] | None = None, store_scope_complete: bool = False, + all_of: tuple[str, ...] | None = None, ) -> list[JsonObject]: scope_kwargs = store_scope_kwargs or {} @@ -284,6 +292,7 @@ def select(rows: Sequence[JsonObject]) -> list[JsonObject]: projects=projects, window_start=window_start, window_end=window_end, + all_of=all_of, ): selected.append(_compact_row(row)) return selected @@ -872,6 +881,9 @@ def _load_inputs( ) -> tuple[list[JsonObject], list[JsonObject], list[JsonObject], list[JsonObject]]: domains = _allowed_domains(request) sensitivity_allowed = _allowed_sensitivity(request) + from alicebot_api.vnext_derived_labels import locked_projects + + all_of = locked_projects(request.agent_identity, request.projects) inclusive_window_end = window_end - timedelta(microseconds=1) source_scope_names = ("scope_projects", "scope_window_start", "scope_window_end") source_scope_supported = _supports_parameters(self.store.search_sources, source_scope_names) @@ -894,7 +906,8 @@ def _load_inputs( } if source_scope_supported else None, - store_scope_complete=source_scope_supported, + store_scope_complete=source_scope_supported and all_of is None, + all_of=all_of, ) memory_store_scope: dict[str, object] | None = ( {"projects": request.projects} if _supports_parameters(self.store.search_memories, ("projects",)) else None @@ -912,6 +925,7 @@ def _load_inputs( window_end=window_end, limit=request.memory_limit, store_scope_kwargs=memory_store_scope, + all_of=all_of, ) open_loop_scope_names = ("scope_projects", "scope_window_start", "scope_window_end") open_loop_scope_supported = _supports_parameters( @@ -937,7 +951,8 @@ def _load_inputs( } if open_loop_scope_supported else None, - store_scope_complete=open_loop_scope_supported, + store_scope_complete=open_loop_scope_supported and all_of is None, + all_of=all_of, ) artifact_store_scope: dict[str, object] | None = ( {"scope_projects": request.projects} @@ -957,6 +972,7 @@ def _load_inputs( window_end=window_end, limit=request.artifact_limit, store_scope_kwargs=artifact_store_scope, + all_of=all_of, ) return sources, memories, open_loops, artifacts diff --git a/apps/api/src/alicebot_api/vnext_connections.py b/apps/api/src/alicebot_api/vnext_connections.py index b62a5a1f3..5119016de 100644 --- a/apps/api/src/alicebot_api/vnext_connections.py +++ b/apps/api/src/alicebot_api/vnext_connections.py @@ -195,9 +195,15 @@ def _supports_parameter(method: object, name: str) -> bool: return False -def _matches_projects(row: JsonObject, projects: tuple[str, ...], *, source_row: bool) -> bool: +def _matches_projects( + row: JsonObject, projects: tuple[str, ...], *, source_row: bool, all_of: tuple[str, ...] | None = None +) -> bool: if not projects: return True + if all_of is not None: + from alicebot_api.vnext_derived_labels import input_admitted + + return input_admitted("source" if source_row else "memory", row, all_of) row_scope = source_project_scope(row) if source_row else resource_project_scope(row) return project_scopes_overlap(row_scope, projects) @@ -210,16 +216,17 @@ def _project_scoped_search( project_parameter: str, limit: int, source_rows: bool = False, + all_of: tuple[str, ...] | None = None, ) -> list[JsonObject]: if not projects: return list(method(limit=limit, **kwargs)) if _supports_parameter(method, project_parameter): rows = method(limit=limit, **kwargs, **{project_parameter: projects}) - return [row for row in rows if _matches_projects(row, projects, source_row=source_rows)] + return [row for row in rows if _matches_projects(row, projects, source_row=source_rows, all_of=all_of)] rows = list(method(limit=MAX_LEGACY_PROJECT_SCOPE_ROWS + 1, **kwargs)) if len(rows) > MAX_LEGACY_PROJECT_SCOPE_ROWS: raise VNextConnectionValidationError("legacy connection store could not prove complete project scope") - return [row for row in rows if _matches_projects(row, projects, source_row=source_rows)][:limit] + return [row for row in rows if _matches_projects(row, projects, source_row=source_rows, all_of=all_of)][:limit] def _record_text(row: JsonObject) -> str: @@ -408,6 +415,9 @@ def generate_connection_report(self, request: ConnectionFinderRequest | None = N domains = list(request.domains) if request.domains else None sensitivity_allowed = list(request.sensitivity_allowed) input_limit = max(request.max_connections * 2, request.max_connections) + from alicebot_api.vnext_derived_labels import locked_projects + + all_of = locked_projects(request.agent_identity, request.projects) sources = _project_scoped_search( self.store.search_sources, kwargs={ @@ -419,6 +429,7 @@ def generate_connection_report(self, request: ConnectionFinderRequest | None = N project_parameter="scope_projects", limit=input_limit, source_rows=True, + all_of=all_of, ) memories = _project_scoped_search( self.store.search_memories, @@ -430,6 +441,7 @@ def generate_connection_report(self, request: ConnectionFinderRequest | None = N projects=request.projects, project_parameter="projects", limit=input_limit, + all_of=all_of, ) candidates = _find_candidates( sources=sources, diff --git a/apps/api/src/alicebot_api/vnext_contradictions.py b/apps/api/src/alicebot_api/vnext_contradictions.py index b822fad95..ab1c89572 100644 --- a/apps/api/src/alicebot_api/vnext_contradictions.py +++ b/apps/api/src/alicebot_api/vnext_contradictions.py @@ -214,9 +214,15 @@ def _supports_parameter(method: object, name: str) -> bool: return False -def _matches_projects(row: JsonObject, projects: tuple[str, ...], *, source_row: bool) -> bool: +def _matches_projects( + row: JsonObject, projects: tuple[str, ...], *, source_row: bool, all_of: tuple[str, ...] | None = None +) -> bool: if not projects: return True + if all_of is not None: + from alicebot_api.vnext_derived_labels import input_admitted + + return input_admitted("source" if source_row else "memory", row, all_of) row_scope = source_project_scope(row) if source_row else resource_project_scope(row) return project_scopes_overlap(row_scope, projects) @@ -229,16 +235,17 @@ def _project_scoped_search( project_parameter: str, limit: int, source_rows: bool = False, + all_of: tuple[str, ...] | None = None, ) -> list[JsonObject]: if not projects: return list(method(limit=limit, **kwargs)) if _supports_parameter(method, project_parameter): rows = method(limit=limit, **kwargs, **{project_parameter: projects}) - return [row for row in rows if _matches_projects(row, projects, source_row=source_rows)] + return [row for row in rows if _matches_projects(row, projects, source_row=source_rows, all_of=all_of)] rows = list(method(limit=MAX_LEGACY_PROJECT_SCOPE_ROWS + 1, **kwargs)) if len(rows) > MAX_LEGACY_PROJECT_SCOPE_ROWS: raise VNextContradictionValidationError("legacy contradiction store could not prove complete project scope") - return [row for row in rows if _matches_projects(row, projects, source_row=source_rows)][:limit] + return [row for row in rows if _matches_projects(row, projects, source_row=source_rows, all_of=all_of)][:limit] def _project_scoped_beliefs( @@ -248,6 +255,7 @@ def _project_scoped_beliefs( sensitivity_allowed: list[str], projects: tuple[str, ...], limit: int, + all_of: tuple[str, ...] | None = None, ) -> list[JsonObject]: if not projects: return list( @@ -295,7 +303,7 @@ def _project_scoped_beliefs( belief for belief in rows if (backing := backing_by_id.get(str(belief.get("memory_id") or ""))) is not None - and _matches_projects(backing, projects, source_row=False) + and _matches_projects(backing, projects, source_row=False, all_of=all_of) ][:limit] @@ -433,6 +441,9 @@ def generate_contradiction_report(self, request: ContradictionFinderRequest | No domains = list(request.domains) if request.domains else None sensitivity_allowed = list(request.sensitivity_allowed) input_limit = max(request.max_contradictions * 2, request.max_contradictions) + from alicebot_api.vnext_derived_labels import locked_projects + + all_of = locked_projects(request.agent_identity, request.projects) sources = _project_scoped_search( self.store.search_sources, kwargs={ @@ -444,6 +455,7 @@ def generate_contradiction_report(self, request: ContradictionFinderRequest | No project_parameter="scope_projects", limit=input_limit, source_rows=True, + all_of=all_of, ) memories = [ memory @@ -457,6 +469,7 @@ def generate_contradiction_report(self, request: ContradictionFinderRequest | No projects=request.projects, project_parameter="projects", limit=input_limit, + all_of=all_of, ) if memory.get("memory_type") not in {"belief", "thesis"} ] @@ -466,6 +479,7 @@ def generate_contradiction_report(self, request: ContradictionFinderRequest | No sensitivity_allowed=sensitivity_allowed, projects=request.projects, limit=input_limit, + all_of=all_of, ) candidates = _find_candidates( new_items=[*sources, *memories], diff --git a/apps/api/src/alicebot_api/vnext_derived_labels.py b/apps/api/src/alicebot_api/vnext_derived_labels.py index ed0558aa7..9069bea03 100644 --- a/apps/api/src/alicebot_api/vnext_derived_labels.py +++ b/apps/api/src/alicebot_api/vnext_derived_labels.py @@ -1199,6 +1199,54 @@ def _iterate( queued.add(dependant) +def locked_projects(agent_identity: object, requested: object) -> tuple[str, ...] | None: + """The projects a locked key may read, or None when the caller is not locked.""" + + if not isinstance(agent_identity, Mapping) or not agent_identity.get("project_scope_locked"): + return None + binding = agent_identity.get("project_scope") or () + requested_values = tuple(requested) if isinstance(requested, (list, tuple)) else () + if requested_values: + return tuple(str(item) for item in requested_values) + if isinstance(binding, (list, tuple)): + return tuple(str(item) for item in binding) + return () + + +def input_admitted(kind: str, row: Mapping[str, object], projects: object) -> bool: + """Exact-door project test: scope and floor are both inside ``projects``.""" + + if canon_kind(kind) == "source": + scope = source_project_scope(row) + else: + scope = resolve_project_scope(row).values + shape, floor = project_floor_shape(row) + if shape == "malformed": + return False + bound = set(project_scope_identity(projects)) + scope_ids = set(project_scope_identity(scope)) + if not scope_ids or not scope_ids <= bound: + return False + return set(project_scope_identity(floor)) <= bound + + +def stamp_derived_from(payload: dict[str, object], rows_by_kind: Mapping[str, object]) -> None: + """Write the canonical dependency record onto ``payload['metadata_json']``.""" + + metadata = payload.get("metadata_json") + meta = dict(metadata) if isinstance(metadata, Mapping) else {} + record: dict[str, object] = {"v": 1} + counts: dict[str, int] = {} + for key in ("sources", "memories", "open_loops", "artifacts", "beliefs"): + rows = rows_by_kind.get(key) or [] + ids = [str(row.get("id")) for row in rows if isinstance(row, Mapping) and row.get("id") is not None] + record[key] = ids + counts[key] = len(ids) + record["counts"] = counts + meta["derived_from"] = record + payload["metadata_json"] = meta + + def scope_is_global(scope: object) -> bool: """True when a scope holds no Alice project id.""" @@ -1225,12 +1273,15 @@ def scope_is_global(scope: object) -> bool: "group_scope", "identifier", "infer_kind", + "input_admitted", "intersect_scope", "is_derived", + "locked_projects", "labels_raised_payload", "ordered_identifiers", "row_class", "scope_is_global", + "stamp_derived_from", "settle_labels", "stored_scope", "union_floor", diff --git a/apps/api/src/alicebot_api/vnext_projects.py b/apps/api/src/alicebot_api/vnext_projects.py index b34b1a88b..87dfdb736 100644 --- a/apps/api/src/alicebot_api/vnext_projects.py +++ b/apps/api/src/alicebot_api/vnext_projects.py @@ -551,10 +551,23 @@ def generate_project_update_candidate(self, request: ProjectAutomationRequest | # defensive check at the workflow boundary so legacy adapters cannot # widen a project-scoped report by ignoring optional query arguments. project_id = str(project["id"]) - sources = [row for row in sources if _is_source_in_project(row, project_id)] - memories = [ - row for row in memories if _is_in_project(row, project_id) and row.get("status") in {"active", "accepted"} - ] + from alicebot_api.vnext_derived_labels import input_admitted, locked_projects + + locked = locked_projects(request.agent_identity, (project_id,)) + if locked is not None: + sources = [row for row in sources if input_admitted("source", row, locked)] + memories = [ + row + for row in memories + if input_admitted("memory", row, locked) and row.get("status") in {"active", "accepted"} + ] + else: + sources = [row for row in sources if _is_source_in_project(row, project_id)] + memories = [ + row + for row in memories + if _is_in_project(row, project_id) and row.get("status") in {"active", "accepted"} + ] brain_charter = _brain_charter(self.store) automation_digest = _project_automation_digest( project=project, diff --git a/tests/unit/test_derived_label_input_filter.py b/tests/unit/test_derived_label_input_filter.py new file mode 100644 index 000000000..62d314c2a --- /dev/null +++ b/tests/unit/test_derived_label_input_filter.py @@ -0,0 +1,63 @@ +"""A locked key's report inputs use the exact project test.""" + +from __future__ import annotations + +from datetime import UTC, datetime, timedelta + +from alicebot_api.vnext_brain import _matches_report_scope +from alicebot_api.vnext_derived_labels import input_admitted, locked_projects, stamp_derived_from + +ALPHA = "prj_" + "a" * 16 +BETA = "prj_" + "b" * 16 + + +def test_input_admitted_requires_every_project() -> None: + alpha = {"metadata_json": {"project_scope": [ALPHA]}} + shared = {"metadata_json": {"project_scope": [ALPHA, BETA]}} + global_row = {"metadata_json": {}} + assert input_admitted("memory", alpha, (ALPHA,)) is True + assert input_admitted("memory", shared, (ALPHA,)) is False + assert input_admitted("memory", global_row, (ALPHA,)) is False + assert input_admitted("memory", alpha, (ALPHA, BETA)) is True + + +def test_locked_projects_uses_the_binding_when_the_request_is_empty() -> None: + identity = {"project_scope_locked": True, "project_scope": [ALPHA]} + assert locked_projects(identity, ()) == (ALPHA,) + assert locked_projects(identity, (ALPHA,)) == (ALPHA,) + assert locked_projects({"project_scope_locked": False}, (ALPHA,)) is None + assert locked_projects(None, (ALPHA,)) is None + + +def test_a_locked_brief_scope_rejects_a_shared_row() -> None: + start = datetime(2026, 10, 5, tzinfo=UTC) + end = start + timedelta(days=1) + shared = { + "id": "m", + "metadata_json": {"project_scope": [ALPHA, BETA]}, + "created_at": start.isoformat(), + } + assert _matches_report_scope( + shared, kind="memory", projects=(ALPHA,), window_start=start, window_end=end + ) is True + assert _matches_report_scope( + shared, kind="memory", projects=(ALPHA,), window_start=start, window_end=end, all_of=(ALPHA,) + ) is False + + +def test_stamp_derived_from_counts_match_the_lists() -> None: + payload: dict[str, object] = {"metadata_json": {"workflow": "daily_brief"}} + stamp_derived_from( + payload, + { + "sources": [{"id": "s"}], + "memories": [{"id": "m"}], + "open_loops": [], + "artifacts": [], + "beliefs": [], + }, + ) + record = payload["metadata_json"]["derived_from"] + assert record["sources"] == ["s"] + assert record["counts"]["sources"] == 1 + assert record["counts"]["memories"] == 1 From 6c547966edb979d31503f7942ad77df1be16967f Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 18:16:19 +0000 Subject: [PATCH 05/90] Record group scope and derived_from on producer rows. Consolidation and roll-ups overlap scope united with floor, and a locked run applies the exact project test after that overlap check. Roll-up lookups and the operator artifact list match the floor. Each producer writes derived_from for the rows it used. --- CHANGELOG.md | 1 + .../src/alicebot_api/routers/vnext_review.py | 14 +- apps/api/src/alicebot_api/vnext_brain.py | 17 + .../api/src/alicebot_api/vnext_connections.py | 2 + apps/api/src/alicebot_api/vnext_connectors.py | 54 ++-- .../src/alicebot_api/vnext_consolidation.py | 76 +++-- .../src/alicebot_api/vnext_contradictions.py | 5 + .../src/alicebot_api/vnext_derived_labels.py | 24 ++ .../src/alicebot_api/vnext_memory_commit.py | 7 +- apps/api/src/alicebot_api/vnext_projects.py | 100 +++--- apps/api/src/alicebot_api/vnext_queue.py | 14 +- apps/api/src/alicebot_api/vnext_rollups.py | 85 +++-- apps/api/src/alicebot_api/vnext_scheduler.py | 42 ++- apps/api/src/alicebot_api/vnext_store.py | 7 +- .../vnext_stores/postgres/memory_access.py | 5 +- .../vnext_stores/postgres/query_predicates.py | 58 ++++ .../vnext_stores/sqlite/memory_access.py | 24 +- .../vnext_stores/sqlite/query_predicates.py | 30 +- tests/unit/test_group_scope_consumers.py | 302 ++++++++++++++++++ tests/unit/test_store_memory_access_split.py | 29 +- tests/unit/test_vnext_brain.py | 5 + 21 files changed, 746 insertions(+), 155 deletions(-) create mode 100644 tests/unit/test_group_scope_consumers.py diff --git a/CHANGELOG.md b/CHANGELOG.md index 3612de93b..7be7ec35d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,7 @@ ## Unreleased +- Unreleased (on main, not in v0.20.0): a locked key's consolidation, roll-up, staleness sweep and open-loop review keep only inputs whose scope and floor are both inside the binding. The overlap check stays, and the exact test runs after it. Consolidation and roll-ups compare the group scope (scope united with floor), so a card stored with an empty scope and a floor of two projects is still found and can be accepted. The operator artifact list, given a project, matches that floor as well as the scope. Each new report and aggregate records `derived_from` for the rows it used. v0.20.0 selected by overlap of the scope alone and wrote no `derived_from`. No migration is required. - Unreleased (on main, not in v0.20.0): a memory copied from another row is stored at least as strict as that row. v0.20.0 kept the label the copy was given at insert, so a later reader could see a health or confidential input through a copy stored as `unknown` and `public`. The insert now re-reads the inputs inside the write and raises the domain, the sensitivity, and the project floor. A later metadata write cannot drop the marker that makes the row derived, or put back an older project scope or project floor. Raising a source or a memory walks the rows derived from it and raises those too. A candidate open loop and a generated artifact take the same insert floor. Replacing a SQLite source with a stricter one raises the memories that cite it before they are retired. Moving a source to another project answers with how many derived rows a project-bound key would lose, and waits for confirm_label_hide before it writes. A relabel that cannot finish answers 409, and one that waits more than 3 seconds for the label lock answers 503. No migration is required. - Unreleased (on main, not in v0.20.0): the label of a derived row (its domain, its sensitivity, and the projects it requires) is now decided by one pure function. v0.20.0 kept the label a derived row was given when it was made. The function is what the write path calls. A read that does not go through that path still behaves as it does in v0.20.0. A locked key is refused when a caller passes a project floor that is not inside the key's binding. No migration is required. - Unreleased (on main, not in v0.20.0): on SQLite, `sources delete`, `sources prune --superseded` and `import-markdown --supersede` now scrub an open loop that names the source only in its metadata: the id, or `source:`, as the text under `source_id`, `source_ids`, `source_ref`, `source_refs`, `source_references` or `selected_source_ids` at any depth. That is the rule the open-loop lookup of a source already used. They blanked only loops whose `source_id` column held the id, so a loop with an empty column kept its title, description and metadata, and they reached a new export. The delete preview now counts the same loops as the receipt. The rule reads the id only as stored or as `source:`: a loop that names the source in any spelling other than those two keeps its text, although the memory rule reads several other spellings. No migration is required. diff --git a/apps/api/src/alicebot_api/routers/vnext_review.py b/apps/api/src/alicebot_api/routers/vnext_review.py index cd0a9adb1..73983e345 100644 --- a/apps/api/src/alicebot_api/routers/vnext_review.py +++ b/apps/api/src/alicebot_api/routers/vnext_review.py @@ -566,11 +566,21 @@ def process_next_vnext_queue_task(request: VNextQueueProcessNextRequest) -> JSON ) @review_router.get("/v0/vnext/artifacts") -def list_vnext_artifacts(user_id: UUID, artifact_type: str | None = None, limit: int = 30) -> JSONResponse: +def list_vnext_artifacts( + user_id: UUID, + artifact_type: str | None = None, + limit: int = 30, + project: str | None = None, +) -> JSONResponse: settings = get_settings() + scope_projects = (project,) if isinstance(project, str) and project.strip() else () with user_connection(settings.database_url, user_id) as conn: - payload = PostgresVNextStore(conn).list_artifacts(artifact_type=artifact_type, limit=limit) + payload = PostgresVNextStore(conn).list_artifacts( + artifact_type=artifact_type, + limit=limit, + scope_projects=scope_projects, + ) return JSONResponse( status_code=200, diff --git a/apps/api/src/alicebot_api/vnext_brain.py b/apps/api/src/alicebot_api/vnext_brain.py index 585d0e0ec..3f70f0425 100644 --- a/apps/api/src/alicebot_api/vnext_brain.py +++ b/apps/api/src/alicebot_api/vnext_brain.py @@ -9,6 +9,7 @@ from typing import Callable, Protocol, Sequence, cast from alicebot_api.vnext_derived_domain import derived_domain +from alicebot_api.vnext_derived_labels import with_derived_from from alicebot_api.vnext_agent_control import resource_project_scope from alicebot_api.vnext_event_log import append_event from alicebot_api.vnext_model_intelligence import ( @@ -577,6 +578,10 @@ def generate_daily_brief(self, request: BrainArtifactRequest | None = None) -> J prompt_hash = model_artifact.prompt_hash model_info_json = model_artifact.model_info metadata = {**metadata, **model_artifact.metadata} + metadata = with_derived_from( + metadata, + {"sources": sources, "memories": memories, "open_loops": open_loops, "artifacts": artifacts}, + ) artifact_payload: JsonObject = { "artifact_type": "daily_brief", "title": f"Daily Brief - {day.isoformat()}", @@ -768,6 +773,10 @@ def generate_weekly_synthesis(self, request: BrainArtifactRequest | None = None) prompt_hash = model_artifact.prompt_hash model_info_json = model_artifact.model_info metadata = {**metadata, **model_artifact.metadata} + metadata = with_derived_from( + metadata, + {"sources": sources, "memories": memories, "open_loops": open_loops, "artifacts": artifacts}, + ) artifact_payload: JsonObject = { "artifact_type": "weekly_synthesis", "title": f"Weekly Synthesis - {week_label}", @@ -1016,6 +1025,10 @@ def _create_candidate_open_loops( "workflow_digest": workflow_digest, }, } + loop_payload["metadata_json"] = with_derived_from( + cast(dict, loop_payload["metadata_json"]), + {"sources": [source]}, + ) if callable(upsert_open_loop): loop = cast(Callable[..., JsonObject], upsert_open_loop)( loop_payload, @@ -1072,6 +1085,10 @@ def _create_weekly_candidate_memories( "workflow_digest": workflow_digest, }, } + memory_payload["metadata_json"] = with_derived_from( + cast(dict, memory_payload["metadata_json"]), + {"sources": sources, "memories": memories, "open_loops": open_loops}, + ) upsert_memory = getattr(self.store, "upsert_memory_by_key", None) if callable(upsert_memory): memory = cast(Callable[..., JsonObject], upsert_memory)( diff --git a/apps/api/src/alicebot_api/vnext_connections.py b/apps/api/src/alicebot_api/vnext_connections.py index 5119016de..6efa104ae 100644 --- a/apps/api/src/alicebot_api/vnext_connections.py +++ b/apps/api/src/alicebot_api/vnext_connections.py @@ -7,6 +7,7 @@ from typing import Callable, Protocol, Sequence, cast from alicebot_api.vnext_derived_domain import derived_domain +from alicebot_api.vnext_derived_labels import with_derived_from from alicebot_api.vnext_agent_control import resource_project_scope from alicebot_api.vnext_event_log import append_event from alicebot_api.vnext_model_intelligence import ( @@ -628,6 +629,7 @@ def generate_connection_report(self, request: ConnectionFinderRequest | None = N prompt_hash = model_artifact.prompt_hash model_info_json = model_artifact.model_info metadata = {**metadata, **model_artifact.metadata} + metadata = with_derived_from(metadata, {"sources": sources, "memories": memories}) artifact_payload: JsonObject = { "artifact_type": "connection_report", "title": "Connection Report", diff --git a/apps/api/src/alicebot_api/vnext_connectors.py b/apps/api/src/alicebot_api/vnext_connectors.py index 9165a2f09..c0b2e49b5 100644 --- a/apps/api/src/alicebot_api/vnext_connectors.py +++ b/apps/api/src/alicebot_api/vnext_connectors.py @@ -28,6 +28,7 @@ ) from alicebot_api.vnext_embeddings import DeferredMemoryEmbedding from alicebot_api.vnext_event_log import append_event +from alicebot_api.vnext_derived_labels import with_derived_from from alicebot_api.vnext_project_scope import resolve_project_scope from alicebot_api.vnext_repositories import JsonObject from alicebot_api.vnext_secrets import ( @@ -1774,17 +1775,20 @@ def ingest_agent_output( "domain": _as_optional_text(payload.get("domain")) or "project", "sensitivity": _as_optional_text(payload.get("sensitivity")) or "private", "generated_by": agent_id, - "metadata_json": { - "connector_name": "agent_output", - "agent_identity": agent_identity, - "agent_id": agent_id, - "agent_run_id": item.metadata_json.get("agent_run_id"), - "project_scope": item.metadata_json.get("project_scope") or [], - "source_id": source_id, - "source_refs": [f"source:{source_id}"] if source_id else [], - "output_type": _as_optional_text(payload.get("output_type")) or "general", - "review_status": "needs_review", - }, + "metadata_json": with_derived_from( + { + "connector_name": "agent_output", + "agent_identity": agent_identity, + "agent_id": agent_id, + "agent_run_id": item.metadata_json.get("agent_run_id"), + "project_scope": item.metadata_json.get("project_scope") or [], + "source_id": source_id, + "source_refs": [f"source:{source_id}"] if source_id else [], + "output_type": _as_optional_text(payload.get("output_type")) or "general", + "review_status": "needs_review", + }, + {"sources": [{"id": source_id}] if source_id else []}, + ), }, actor_type="agent", ) @@ -1836,17 +1840,23 @@ def ingest_agent_output( "project_id": proposal_scope[0] if len(proposal_scope) == 1 else None, "created_by_agent_id": agent_id, "run_id": item.metadata_json.get("agent_run_id"), - "metadata_json": { - "connector_name": "agent_output", - "agent_identity": agent_identity, - "agent_id": agent_id, - "agent_run_id": item.metadata_json.get("agent_run_id"), - "source_id": source_id, - "artifact_id": artifact_id, - "review_required": True, - "policy_decision": policy_decision, - **({"project_scope": list(proposal_scope)} if proposal_scope else {}), - }, + "metadata_json": with_derived_from( + { + "connector_name": "agent_output", + "agent_identity": agent_identity, + "agent_id": agent_id, + "agent_run_id": item.metadata_json.get("agent_run_id"), + "source_id": source_id, + "artifact_id": artifact_id, + "review_required": True, + "policy_decision": policy_decision, + **({"project_scope": list(proposal_scope)} if proposal_scope else {}), + }, + { + "sources": [{"id": source_id}] if source_id else [], + "artifacts": [{"id": artifact_id}] if artifact_id else [], + }, + ), }, actor_type="agent", ) diff --git a/apps/api/src/alicebot_api/vnext_consolidation.py b/apps/api/src/alicebot_api/vnext_consolidation.py index 4c4cc8975..7c1922800 100644 --- a/apps/api/src/alicebot_api/vnext_consolidation.py +++ b/apps/api/src/alicebot_api/vnext_consolidation.py @@ -43,6 +43,12 @@ import numpy as np from alicebot_api.vnext_derived_domain import derived_domain +from alicebot_api.vnext_derived_labels import ( + admit_when_locked, + group_scope, + locked_projects, + with_derived_from, +) from alicebot_api.vnext_embeddings import ( MAX_EMBEDDINGS_BATCH_SIZE, EmbeddingProvider, @@ -450,24 +456,23 @@ def _scoped_rows( continue if projects: allowed_projects = set(project_scope_identity(projects)) - if not allowed_projects.intersection( - project_scope_identity(resource_project_scope(row)) - ): + if not allowed_projects.intersection(group_scope(row)): continue scoped.append(row) return scoped def _project_scope_key(row: JsonObject) -> tuple[str, ...]: - """Exact normalized scope identity used for safe consolidation groups. + """Exact group scope used for safe consolidation groups. Overlap is insufficient here: merging a memory scoped to A+B with one scoped only to A would widen B-only information into project A. Candidate members must therefore carry the same scope set (including the empty - global scope). + global scope). A derived row uses its scope united with its floor, which + equals the scope for an original row. """ - return project_scope_identity(resource_project_scope(row)) + return group_scope(row) def _shared_project_scope(rows: list[JsonObject]) -> tuple[str, ...]: @@ -663,6 +668,7 @@ def _cluster_memories( sensitivity: list[str], projects: tuple[str, ...], options: _ClusteringOptions, + all_of: tuple[str, ...] | None = None, ) -> _ClusteringOutcome: outcome = _ClusteringOutcome() count_memories = getattr(self.store, "count_memories", None) @@ -755,6 +761,10 @@ def _count(status: str) -> int: outcome.active_count, ) active_rows = active_rows[: options.max_embedded_memories] + if all_of is not None: + active_rows = admit_when_locked("memory", active_rows, all_of) + outcome.active_count = len(active_rows) + outcome.active_count_exact = not outcome.bounded outcome.corpus_digest = _digest_payload( { "memory_versions": [ @@ -1027,24 +1037,27 @@ def _create_proposal_candidate( "sensitivity": _highest_sensitivity(members), "project_id": project_scope[0] if len(project_scope) == 1 else None, "source_event_ids": proposal["source_event_ids"], - "metadata_json": { - "candidate_kind": "memory_consolidation", - "consolidation_digest": cluster_digest, - "source_refs": proposal["source_refs"], - "project_scope": list(project_scope), - "review_required": True, - "consolidation": { - "cluster_member_ids": member_ids, - "member_snapshots": proposal["member_snapshots"], - "similarity_stats": proposal["similarity_stats"], - "proposal_kind": proposal_kind, - "model_provenance": proposal["model_provenance"], - "survivor_memory_id": proposal["survivor_memory_id"], - "proposed_supersede": proposal["proposed_supersede"], - "merge_refusal": proposal["merge_refusal"], - "reviewer_instructions": reviewer_instructions, + "metadata_json": with_derived_from( + { + "candidate_kind": "memory_consolidation", + "consolidation_digest": cluster_digest, + "source_refs": proposal["source_refs"], + "project_scope": list(project_scope), + "review_required": True, + "consolidation": { + "cluster_member_ids": member_ids, + "member_snapshots": proposal["member_snapshots"], + "similarity_stats": proposal["similarity_stats"], + "proposal_kind": proposal_kind, + "model_provenance": proposal["model_provenance"], + "survivor_memory_id": proposal["survivor_memory_id"], + "proposed_supersede": proposal["proposed_supersede"], + "merge_refusal": proposal["merge_refusal"], + "reviewer_instructions": reviewer_instructions, + }, }, - }, + {"memories": members}, + ), }, actor_type=request.generated_by, ) @@ -1097,6 +1110,7 @@ def generate_memory_consolidation(self, request: MemoryConsolidationRequest | No domains = _allowed_domains(request) sensitivity = _allowed_sensitivity(request) projects = _allowed_projects(request) + all_of = locked_projects(request.agent_identity, projects) if projects: list_memory_events = getattr(self.store, "list_memory_events", None) @@ -1169,12 +1183,16 @@ def generate_memory_consolidation(self, request: MemoryConsolidationRequest | No sensitivity_allowed=sensitivity, projects=projects, ) + events = admit_when_locked("memory", events, all_of) + ratings = admit_when_locked("memory", ratings, all_of) + artifacts = admit_when_locked("artifact", artifacts, all_of) clustering = self._cluster_memories( domains=domains, sensitivity=sensitivity, projects=projects, options=options, + all_of=all_of, ) cluster_membership = [ sorted(str(row.get("id")) for row in members) for members in clustering.clusters @@ -1340,6 +1358,7 @@ def generate_memory_consolidation(self, request: MemoryConsolidationRequest | No generation_mode=request.generation_mode, route=route, model_temperature=request.model_temperature, + agent_identity=request.agent_identity, # Near-duplicate clusters belong to the dedup/merge proposals # above; the roll-up pass must not re-propose those groups. exclude_member_id_sets=[ @@ -1460,6 +1479,17 @@ def generate_memory_consolidation(self, request: MemoryConsolidationRequest | No metadata = {**metadata, **model_artifact.metadata} all_cluster_rows = [row for members in clustering.clusters for row in members] + metadata = with_derived_from( + metadata, + { + "sources": named_sources, + "memories": [ + *all_cluster_rows, + *(rollups.input_rows if rollups is not None else []), + ], + "artifacts": artifacts, + }, + ) # The report is read behind its domain and sensitivity, so both are taken over every row it names: the # near-duplicate cluster members, every row the roll-up pass names, and the sources that the refs it prints # name (read above, after the run's own fence chose the members the refs are copied from). A run whose only diff --git a/apps/api/src/alicebot_api/vnext_contradictions.py b/apps/api/src/alicebot_api/vnext_contradictions.py index ab1c89572..9e9274ad7 100644 --- a/apps/api/src/alicebot_api/vnext_contradictions.py +++ b/apps/api/src/alicebot_api/vnext_contradictions.py @@ -6,6 +6,7 @@ from typing import Callable, Protocol, Sequence, cast from alicebot_api.vnext_derived_domain import derived_domain +from alicebot_api.vnext_derived_labels import with_derived_from from alicebot_api.vnext_agent_control import resource_project_scope from alicebot_api.vnext_event_log import append_event from alicebot_api.vnext_model_intelligence import ( @@ -665,6 +666,10 @@ def generate_contradiction_report(self, request: ContradictionFinderRequest | No prompt_hash = model_artifact.prompt_hash model_info_json = model_artifact.model_info metadata = {**metadata, **model_artifact.metadata} + metadata = with_derived_from( + metadata, + {"sources": sources, "memories": memories, "beliefs": beliefs}, + ) artifact_payload: JsonObject = { "artifact_type": "contradiction_report", "title": "Contradiction Report", diff --git a/apps/api/src/alicebot_api/vnext_derived_labels.py b/apps/api/src/alicebot_api/vnext_derived_labels.py index 9069bea03..c5c7d26dc 100644 --- a/apps/api/src/alicebot_api/vnext_derived_labels.py +++ b/apps/api/src/alicebot_api/vnext_derived_labels.py @@ -1247,6 +1247,28 @@ def stamp_derived_from(payload: dict[str, object], rows_by_kind: Mapping[str, ob payload["metadata_json"] = meta +def with_derived_from(metadata: Mapping[str, object], rows_by_kind: Mapping[str, object]) -> dict[str, object]: + """A copy of ``metadata`` with ``derived_from`` for the rows a producer used.""" + + payload: dict[str, object] = {"metadata_json": dict(metadata)} + stamp_derived_from(payload, rows_by_kind) + stamped = payload["metadata_json"] + return dict(stamped) if isinstance(stamped, Mapping) else {} + + +def admit_when_locked( + kind: str, + rows: object, + projects: tuple[str, ...] | None, +) -> list[Mapping[str, object]]: + """Keep every row when ``projects`` is None. Otherwise keep rows inside that binding.""" + + items = list(rows) if isinstance(rows, (list, tuple)) else [] + if projects is None: + return [row for row in items if isinstance(row, Mapping)] + return [row for row in items if isinstance(row, Mapping) and input_admitted(kind, row, projects)] + + def scope_is_global(scope: object) -> bool: """True when a scope holds no Alice project id.""" @@ -1273,10 +1295,12 @@ def scope_is_global(scope: object) -> bool: "group_scope", "identifier", "infer_kind", + "admit_when_locked", "input_admitted", "intersect_scope", "is_derived", "locked_projects", + "with_derived_from", "labels_raised_payload", "ordered_identifiers", "row_class", diff --git a/apps/api/src/alicebot_api/vnext_memory_commit.py b/apps/api/src/alicebot_api/vnext_memory_commit.py index a1af3cb8b..605f9da22 100644 --- a/apps/api/src/alicebot_api/vnext_memory_commit.py +++ b/apps/api/src/alicebot_api/vnext_memory_commit.py @@ -90,6 +90,7 @@ PENDING_PROJECT_UPDATE_MEMORY_MUTATION_MESSAGE, is_pending_project_update_memory, ) +from alicebot_api.vnext_derived_labels import group_scope from alicebot_api.vnext_project_scope import normalize_project_scope, project_scope_identity from alicebot_api.vnext_repositories import EventStore, JsonObject from alicebot_api.store import ContinuityStoreInvariantError @@ -2104,10 +2105,8 @@ def accept_consolidation_candidate( ) if strict_snapshots and dependency_ids: - member_scope_keys = { - project_scope_identity(resource_project_scope(member)) for member in locked_members.values() - } - candidate_scope_key = project_scope_identity(resource_project_scope(memory)) + member_scope_keys = {group_scope(member, kind="memory") for member in locked_members.values()} + candidate_scope_key = group_scope(memory, kind="memory") if len(member_scope_keys) != 1 or candidate_scope_key not in member_scope_keys: raise VNextMemoryCommitValidationError( "consolidation candidate crosses project scopes; regenerate it before acceptance" diff --git a/apps/api/src/alicebot_api/vnext_projects.py b/apps/api/src/alicebot_api/vnext_projects.py index 87dfdb736..2034c394c 100644 --- a/apps/api/src/alicebot_api/vnext_projects.py +++ b/apps/api/src/alicebot_api/vnext_projects.py @@ -8,6 +8,7 @@ from typing import Protocol, cast from alicebot_api.vnext_derived_domain import derived_domain +from alicebot_api.vnext_derived_labels import input_admitted, locked_projects, with_derived_from from alicebot_api.credential_floor import refuse_credential_activation from alicebot_api.vnext_agent_control import resource_project_scope from alicebot_api.vnext_embeddings import DeferredMemoryEmbedding @@ -551,8 +552,6 @@ def generate_project_update_candidate(self, request: ProjectAutomationRequest | # defensive check at the workflow boundary so legacy adapters cannot # widen a project-scoped report by ignoring optional query arguments. project_id = str(project["id"]) - from alicebot_api.vnext_derived_labels import input_admitted, locked_projects - locked = locked_projects(request.agent_identity, (project_id,)) if locked is not None: sources = [row for row in sources if input_admitted("source", row, locked)] @@ -602,23 +601,24 @@ def generate_project_update_candidate(self, request: ProjectAutomationRequest | "domain": derived_domain([project, *sources, *memories], fallback=str(project.get("domain", "project"))), "sensitivity": _highest_sensitivity([project, *sources, *memories]), "project_id": project_id, - "metadata_json": { - **request.metadata_json, - "candidate": True, - "workflow": "project_auto_update", - "project_id": project.get("id"), - "project_scope": [project_id], - "automation_digest": automation_digest, - "source_ids": _source_ids(sources), - "memory_ids": _source_ids(memories), - "generated_by": request.generated_by, - "agent_identity": request.agent_identity, - "agent_id": request.actor_id if request.generated_by == "agent" else None, - "trace_id": request.trace_id, - "policy_decision": request.policy_decision, - "project_scope": [project_id], - "automation_digest": automation_digest, - }, + "metadata_json": with_derived_from( + { + **request.metadata_json, + "candidate": True, + "workflow": "project_auto_update", + "project_id": project.get("id"), + "source_ids": _source_ids(sources), + "memory_ids": _source_ids(memories), + "generated_by": request.generated_by, + "agent_identity": request.agent_identity, + "agent_id": request.actor_id if request.generated_by == "agent" else None, + "trace_id": request.trace_id, + "policy_decision": request.policy_decision, + "project_scope": [project_id], + "automation_digest": automation_digest, + }, + {"sources": sources, "memories": memories}, + ), }, actor_type=request.generated_by, ) @@ -640,28 +640,29 @@ def generate_project_update_candidate(self, request: ProjectAutomationRequest | "generated_by": request.generated_by if request.generated_by != "system" else "vnext_project_auto_updater", "prompt_hash": prompt_hash, "model_info_json": model_info_json, - "metadata_json": { - **request.metadata_json, - "workflow": "project_auto_update", - "workflow_type": "project_update_scan", - "project_id": project.get("id"), - "project_scope": [project_id], - "automation_digest": automation_digest, - "candidate_memory_id": candidate_memory.get("id"), - "suggested_current_state": suggested_current_state, - "source_ids": _source_ids(sources), - "source_refs": [f"source:{source_id}" for source_id in _source_ids(sources)], - "memory_ids": _source_ids(memories), - "generated_by": request.generated_by, - "agent_identity": request.agent_identity, - "agent_id": request.actor_id if request.generated_by == "agent" else None, - "agent_run_id": request.run_id if request.generated_by == "agent" else None, - "trace_id": request.trace_id, - "policy_decision": request.policy_decision, - **model_metadata, - "project_scope": [project_id], - "automation_digest": automation_digest, - }, + "metadata_json": with_derived_from( + { + **request.metadata_json, + "workflow": "project_auto_update", + "workflow_type": "project_update_scan", + "project_id": project.get("id"), + "automation_digest": automation_digest, + "candidate_memory_id": candidate_memory.get("id"), + "suggested_current_state": suggested_current_state, + "source_ids": _source_ids(sources), + "source_refs": [f"source:{source_id}" for source_id in _source_ids(sources)], + "memory_ids": _source_ids(memories), + "generated_by": request.generated_by, + "agent_identity": request.agent_identity, + "agent_id": request.actor_id if request.generated_by == "agent" else None, + "agent_run_id": request.run_id if request.generated_by == "agent" else None, + "trace_id": request.trace_id, + "policy_decision": request.policy_decision, + **model_metadata, + "project_scope": [project_id], + }, + {"sources": sources, "memories": [*memories, candidate_memory]}, + ), } upsert_artifact = getattr(self.store, "upsert_artifact_by_workflow_digest", None) if callable(upsert_artifact): @@ -992,11 +993,24 @@ def review_project_update( current_state, error=VNextProjectValidationError, ) - if self.store.get_project_for_update(project_id) is None: + locked_project = self.store.get_project_for_update(project_id) + if locked_project is None: raise VNextProjectValidationError("project update candidate project was not found") + existing_meta = locked_project.get("metadata_json") + project_meta = dict(existing_meta) if isinstance(existing_meta, Mapping) else {} + recorded_sources = candidate_metadata.get("source_ids") + recorded_memories = candidate_metadata.get("memory_ids") + project_meta = with_derived_from( + project_meta, + { + "sources": [{"id": item} for item in recorded_sources] if isinstance(recorded_sources, list) else [], + "memories": [{"id": item} for item in recorded_memories] if isinstance(recorded_memories, list) else [], + "artifacts": [{"id": artifact_id}], + }, + ) self.store.update_project( project_id=project_id, - patch={"current_state": current_state}, + patch={"current_state": current_state, "metadata_json": project_meta}, actor_type=actor_type, ) updated_memory = self.store.update_memory( diff --git a/apps/api/src/alicebot_api/vnext_queue.py b/apps/api/src/alicebot_api/vnext_queue.py index d846619b1..6bd31b46c 100644 --- a/apps/api/src/alicebot_api/vnext_queue.py +++ b/apps/api/src/alicebot_api/vnext_queue.py @@ -10,6 +10,7 @@ from alicebot_api.vnext_embeddings import DeferredMemoryEmbedding, attach_memory_embedding from alicebot_api.vnext_event_log import append_event from alicebot_api.vnext_agent_control import resource_project_scope +from alicebot_api.vnext_derived_labels import with_derived_from from alicebot_api.vnext_project_update_guard import is_project_update_artifact from alicebot_api.vnext_repositories import JsonObject @@ -471,11 +472,14 @@ def _promote_artifact( "sensitivity": str(artifact.get("sensitivity") or "unknown"), "project_id": scope[0] if len(scope) == 1 else None, "source_event_ids": [], - "metadata_json": { - "source_artifact_id": artifact_id, - "project_scope": list(scope), - "promotion_reviewed": True, - }, + "metadata_json": with_derived_from( + { + "source_artifact_id": artifact_id, + "project_scope": list(scope), + "promotion_reviewed": True, + }, + {"artifacts": [artifact]}, + ), }, actor_type=actor_type, ) diff --git a/apps/api/src/alicebot_api/vnext_rollups.py b/apps/api/src/alicebot_api/vnext_rollups.py index 0bb942434..33920087d 100644 --- a/apps/api/src/alicebot_api/vnext_rollups.py +++ b/apps/api/src/alicebot_api/vnext_rollups.py @@ -166,6 +166,12 @@ memory_embedding_text, ) from alicebot_api.vnext_agent_control import resource_project_scope +from alicebot_api.vnext_derived_labels import ( + admit_when_locked, + group_scope, + locked_projects, + with_derived_from, +) from alicebot_api.vnext_entities import extract_entity_candidates from alicebot_api.vnext_model_intelligence import ( NON_SYNTHESIZING_PROVIDERS, @@ -653,7 +659,7 @@ def _member_text(row: JsonObject) -> str: def _project_scope_key(row: JsonObject) -> tuple[str, ...]: - return project_scope_identity(resource_project_scope(row)) + return group_scope(row) def _shared_project_scope(rows: tuple[JsonObject, ...] | list[JsonObject]) -> tuple[str, ...]: @@ -1523,9 +1529,7 @@ def _scoped_rows( continue if projects: allowed_projects = set(project_scope_identity(projects)) - if not allowed_projects.intersection( - project_scope_identity(resource_project_scope(row)) - ): + if not allowed_projects.intersection(group_scope(row)): continue scoped.append(row) return scoped @@ -1746,6 +1750,7 @@ def _collect_rows( sensitivity_allowed: list[str], projects: tuple[str, ...], options: RollupOptions, + all_of: tuple[str, ...] | None = None, ) -> tuple[list[JsonObject], bool, int, bool]: # Ask for one sentinel row beyond the configured cap. The store # applies status, scope, roll-up-card exclusion, deterministic order, @@ -1819,7 +1824,7 @@ def _collect_rows( raise VNextRollupValidationError( "roll-up input lookup returned rows outside the requested project scope" ) - rows = scoped_rows + rows = admit_when_locked("memory", scoped_rows, all_of) rows = [row for row in rows if not _is_rollup_card(row)] # The same parity for validity: the bundled stores leave an expired # memory out in SQL, and every tier below (entity, topic and the @@ -2383,6 +2388,7 @@ def _existing_rollup_state( domains: list[str] | None, sensitivity_allowed: list[str], projects: tuple[str, ...], + all_of: tuple[str, ...] | None = None, ) -> tuple[dict[str, JsonObject], dict[str, JsonObject]]: """(pending candidate by rollup_digest, accepted card by rollup_key).""" pending: dict[str, JsonObject] = {} @@ -2454,6 +2460,13 @@ def _existing_rollup_state( raise VNextRollupValidationError( "roll-up candidate/card lookup returned rows outside the requested project scope" ) + if all_of is not None: + pending = { + key: row for key, row in pending.items() if admit_when_locked("memory", [row], all_of) + } + accepted = { + key: row for key, row in accepted.items() if admit_when_locked("memory", [row], all_of) + } return pending, accepted def _expired_card_for_digest( @@ -2727,33 +2740,41 @@ def _create_rollup_candidate( "sensitivity": _highest_sensitivity(group.members), "project_id": project_scope[0] if len(project_scope) == 1 else None, "source_event_ids": source_event_ids, - "metadata_json": { - "candidate_kind": ROLLUP_CANDIDATE_KIND, - "rollup_digest": rollup_digest, - "rollup_key": group.rollup_key, - "review_required": True, - "source_refs": source_refs, - "project_scope": list(project_scope), - "trace_id": trace_id, - # accept_consolidation_candidate compatibility: the - # existing review/acceptance path reads this block. - "consolidation": { - "proposal_kind": ROLLUP_PROPOSAL_KIND, - "cluster_member_ids": member_ids, - "member_snapshots": member_snapshots, - "proposed_supersede": proposed_supersede, - "survivor_memory_id": None, - "model_provenance": model_provenance, - "merge_refusal": merge_refusal, - "reviewer_instructions": reviewer_instructions, - "rollup": { - "rollup_key": group.rollup_key, - "group_kind": group.group_kind, - "topic_label": group.label, - "revises_memory_id": revises_memory_id, + "metadata_json": with_derived_from( + { + "candidate_kind": ROLLUP_CANDIDATE_KIND, + "rollup_digest": rollup_digest, + "rollup_key": group.rollup_key, + "review_required": True, + "source_refs": source_refs, + "project_scope": list(project_scope), + "trace_id": trace_id, + # accept_consolidation_candidate compatibility: the + # existing review/acceptance path reads this block. + "consolidation": { + "proposal_kind": ROLLUP_PROPOSAL_KIND, + "cluster_member_ids": member_ids, + "member_snapshots": member_snapshots, + "proposed_supersede": proposed_supersede, + "survivor_memory_id": None, + "model_provenance": model_provenance, + "merge_refusal": merge_refusal, + "reviewer_instructions": reviewer_instructions, + "rollup": { + "rollup_key": group.rollup_key, + "group_kind": group.group_kind, + "topic_label": group.label, + "revises_memory_id": revises_memory_id, + }, }, }, - }, + { + "memories": [ + *group.members, + *([revises_memory] if revises_memory is not None else []), + ] + }, + ), }, actor_type=generated_by, ) @@ -2774,6 +2795,7 @@ def propose_rollups( route=None, model_temperature: float = 0.2, exclude_member_id_sets: list[set[str]] | None = None, + agent_identity: object = None, ) -> RollupOutcome: """One review-only roll-up pass over the in-scope memories. @@ -2791,6 +2813,7 @@ def propose_rollups( """ options = options or RollupOptions() sensitivity = list(sensitivity_allowed or ("public", "internal", "private", "unknown")) + all_of = locked_projects(agent_identity, projects) outcome = RollupOutcome(options=options.to_record()) rows, bounded, total_count, total_exact = self._collect_rows( @@ -2798,6 +2821,7 @@ def propose_rollups( sensitivity_allowed=sensitivity, projects=projects, options=options, + all_of=all_of, ) outcome.groupable_count = len(rows) outcome.groupable_total_count = total_count @@ -2880,6 +2904,7 @@ def propose_rollups( domains=domains, sensitivity_allowed=sensitivity, projects=projects, + all_of=all_of, ) for prepared in prepared_groups: diff --git a/apps/api/src/alicebot_api/vnext_scheduler.py b/apps/api/src/alicebot_api/vnext_scheduler.py index 04f53e34d..55ff84a41 100644 --- a/apps/api/src/alicebot_api/vnext_scheduler.py +++ b/apps/api/src/alicebot_api/vnext_scheduler.py @@ -13,6 +13,7 @@ from zoneinfo import ZoneInfo, ZoneInfoNotFoundError from alicebot_api.vnext_derived_domain import derived_domain +from alicebot_api.vnext_derived_labels import admit_when_locked, locked_projects, with_derived_from from alicebot_api.vnext_agent_control import ( AgentIdentity, PolicyDecision, @@ -1402,6 +1403,11 @@ def _run_staleness_sweep(self, request: SchedulerRunRequest, *, metadata: JsonOb raise VNextSchedulerValidationError( "staleness sweep store returned memories outside the requested project scope" ) + bound = locked_projects( + request.agent_identity.to_record() if request.agent_identity is not None else None, + projects, + ) + memories = admit_when_locked("memory", memories, bound) for memory in memories: if len(expired_marked) + len(unconfirmed_marked) >= mark_limit: break @@ -1470,19 +1476,22 @@ def _run_staleness_sweep(self, request: SchedulerRunRequest, *, metadata: JsonOb "domain": derived_domain(marked, fallback=request.domains[0] if len(request.domains) == 1 else "unknown"), "sensitivity": self._highest_sensitivity(marked), "generated_by": "scheduler", - "metadata_json": { - **metadata, - "workflow": "staleness_sweep", - "source_refs": [], - "stale_marked_memory_ids": [str(row.get("id")) for row in marked], - "staleness_window_days": window_days, - "input_counts": { - "scanned": scanned_count, - "expired_marked": len(expired_marked), - "unconfirmed_marked": len(unconfirmed_marked), + "metadata_json": with_derived_from( + { + **metadata, + "workflow": "staleness_sweep", + "source_refs": [], + "stale_marked_memory_ids": [str(row.get("id")) for row in marked], + "staleness_window_days": window_days, + "input_counts": { + "scanned": scanned_count, + "expired_marked": len(expired_marked), + "unconfirmed_marked": len(unconfirmed_marked), + }, + "review_policy": "marks_stale_never_deletes", }, - "review_policy": "marks_stale_never_deletes", - }, + {"memories": marked}, + ), }, actor_type="scheduler", ) @@ -1565,6 +1574,11 @@ def _generate_open_loop_review_artifact(self, request: SchedulerRunRequest, *, m ) if any(not _row_matches_projects(loop, projects) for loop in loops): raise VNextSchedulerValidationError("open-loop store returned rows outside the requested project scope") + bound = locked_projects( + request.agent_identity.to_record() if request.agent_identity is not None else None, + projects, + ) + loops = admit_when_locked("open_loop", loops, bound) # The report copies the id of each loop's source into its text and its ``source_refs``, and a later reader of # the artifact is shown them, so a source the run's own identity may not read is left out. loops = withhold_unreadable_references( @@ -1678,6 +1692,10 @@ def _generate_open_loop_review_artifact(self, request: SchedulerRunRequest, *, m prompt_hash = model_artifact.prompt_hash model_info_json = model_artifact.model_info enriched_metadata = {**enriched_metadata, **model_artifact.metadata} + enriched_metadata = with_derived_from( + enriched_metadata, + {"open_loops": loops, "sources": linked_sources}, + ) artifact_payload: JsonObject = { "artifact_type": "open_loop_report", "title": f"Open Loop Review - {request.generated_for or datetime.now(UTC).date().isoformat()}", diff --git a/apps/api/src/alicebot_api/vnext_store.py b/apps/api/src/alicebot_api/vnext_store.py index e83c801f5..fc7fc4274 100644 --- a/apps/api/src/alicebot_api/vnext_store.py +++ b/apps/api/src/alicebot_api/vnext_store.py @@ -172,6 +172,9 @@ ) from alicebot_api.vnext_stores.postgres.query_predicates import ( _ARTIFACT_SCOPE_PROJECT_SQL as _ARTIFACT_SCOPE_PROJECT_SQL, + _PROJECT_FLOOR_SQL as _PROJECT_FLOOR_SQL, + _MEMORY_GROUP_SCOPE_SQL as _MEMORY_GROUP_SCOPE_SQL, + _jsonb_string_array_identity_sql as _jsonb_string_array_identity_sql, _ASCII_PROJECT_LOWER as _ASCII_PROJECT_LOWER, _ASCII_PROJECT_UPPER as _ASCII_PROJECT_UPPER, _MEMORY_DIRECT_PEOPLE_SQL as _MEMORY_DIRECT_PEOPLE_SQL, @@ -2260,7 +2263,8 @@ def list_artifacts( WHERE (%s::text IS NULL OR artifact_type = %s) AND (%s::text[] IS NULL OR domain = ANY(%s::text[]) OR domain = 'unknown') AND (%s::text[] IS NULL OR sensitivity = ANY(%s::text[])) - AND (%s::text[] IS NULL OR ({_ARTIFACT_SCOPE_PROJECT_SQL}) ?| %s::text[]) + AND (%s::text[] IS NULL OR ({_ARTIFACT_SCOPE_PROJECT_SQL}) ?| %s::text[] + OR ({_PROJECT_FLOOR_SQL}) ?| %s::text[]) ORDER BY created_at DESC, id DESC LIMIT %s """, @@ -2273,6 +2277,7 @@ def list_artifacts( sensitivity_allowed, project_list, project_list, + project_list, limit, ), ) diff --git a/apps/api/src/alicebot_api/vnext_stores/postgres/memory_access.py b/apps/api/src/alicebot_api/vnext_stores/postgres/memory_access.py index ac34bb114..d8bb10f57 100644 --- a/apps/api/src/alicebot_api/vnext_stores/postgres/memory_access.py +++ b/apps/api/src/alicebot_api/vnext_stores/postgres/memory_access.py @@ -21,6 +21,7 @@ from alicebot_api.vnext_stores.postgres.primitives import _json_list from alicebot_api.vnext_stores.postgres.query_predicates import ( _MEMORY_DIRECT_PEOPLE_SQL, + _MEMORY_GROUP_SCOPE_SQL, _MEMORY_PROJECT_SCOPE_SQL, _MEMORY_SCOPE_EVENT_TIME_SQL, _escape_like_literal, @@ -701,7 +702,7 @@ def list_pending_rollup_candidates( AND metadata_json ->> 'rollup_digest' = ANY(%s::text[]) AND (%s::text[] IS NULL OR domain = ANY(%s::text[]) OR domain = 'unknown') AND COALESCE(sensitivity, 'unknown') = ANY(%s::text[]) - AND (%s::text[] IS NULL OR ({_MEMORY_PROJECT_SCOPE_SQL}) ?| %s::text[]) + AND (%s::text[] IS NULL OR ({_MEMORY_GROUP_SCOPE_SQL}) ?| %s::text[]) ORDER BY metadata_json ->> 'rollup_digest', updated_at DESC, created_at DESC, id DESC LIMIT %s """, @@ -748,7 +749,7 @@ def list_accepted_rollup_cards( AND metadata_json ->> 'rollup_key' = ANY(%s::text[]) AND (%s::text[] IS NULL OR domain = ANY(%s::text[]) OR domain = 'unknown') AND COALESCE(sensitivity, 'unknown') = ANY(%s::text[]) - AND (%s::text[] IS NULL OR ({_MEMORY_PROJECT_SCOPE_SQL}) ?| %s::text[]) + AND (%s::text[] IS NULL OR ({_MEMORY_GROUP_SCOPE_SQL}) ?| %s::text[]) ORDER BY metadata_json ->> 'rollup_key', CASE WHEN status = 'active' THEN 0 ELSE 1 END, diff --git a/apps/api/src/alicebot_api/vnext_stores/postgres/query_predicates.py b/apps/api/src/alicebot_api/vnext_stores/postgres/query_predicates.py index 2e5a2e783..862ccc98d 100644 --- a/apps/api/src/alicebot_api/vnext_stores/postgres/query_predicates.py +++ b/apps/api/src/alicebot_api/vnext_stores/postgres/query_predicates.py @@ -354,6 +354,63 @@ def _jsonb_source_project_scope_values_sql(metadata_expression: str) -> str: ) +def _jsonb_string_array_identity_sql(array_expression: str) -> str: + """Identity of a JSON array of strings. Any other shape contributes nothing.""" + + normalized = _normalized_project_identifier_sql("floor_text.value") + identity = _project_identifier_identity_sql("normalized_floor.value", already_normalized=True) + return f""" +( + SELECT COALESCE( + jsonb_agg(floor_identity.value ORDER BY floor_identity.value COLLATE "C"), + '[]'::jsonb + ) + FROM ( + SELECT DISTINCT {identity} AS value + FROM jsonb_array_elements( + CASE + WHEN jsonb_typeof({array_expression}) = 'array' THEN {array_expression} + ELSE '[]'::jsonb + END + ) AS floor_element(value) + CROSS JOIN LATERAL ( + SELECT CASE + WHEN jsonb_typeof(floor_element.value) = 'string' THEN floor_element.value #>> '{{}}' + ELSE '' + END AS value + ) AS floor_text + CROSS JOIN LATERAL ( + SELECT {normalized} AS value + ) AS normalized_floor + WHERE normalized_floor.value <> '' + ) AS floor_identity +) +""" + + +_PROJECT_FLOOR_SQL = _jsonb_string_array_identity_sql("metadata_json -> 'project_floor'") + + +# Overlap of scope united with floor. Original rows have no floor, so this +# matches the same rows as _MEMORY_PROJECT_SCOPE_SQL for them. +_MEMORY_GROUP_SCOPE_SQL = f""" +( + SELECT COALESCE( + jsonb_agg(grouped.value ORDER BY grouped.value COLLATE "C"), + '[]'::jsonb + ) + FROM ( + SELECT DISTINCT part.value + FROM ( + SELECT jsonb_array_elements_text(({_MEMORY_PROJECT_SCOPE_SQL})) AS value + UNION ALL + SELECT jsonb_array_elements_text(({_PROJECT_FLOOR_SQL})) AS value + ) AS part + ) AS grouped +) +""" + + _MEMORY_DIRECT_PEOPLE_SQL = """ EXISTS ( SELECT 1 @@ -490,6 +547,7 @@ def _tsquery_any_expression(query: str) -> str | None: _normalized_project_identifier_sql, _project_identifier_identity_sql, _jsonb_project_scope_values_sql, + _jsonb_string_array_identity_sql, _jsonb_project_scope_leaf_values_sql, _jsonb_source_project_scope_values_sql, _jsonb_scope_values_sql, diff --git a/apps/api/src/alicebot_api/vnext_stores/sqlite/memory_access.py b/apps/api/src/alicebot_api/vnext_stores/sqlite/memory_access.py index 556f1bbd5..4b677e515 100644 --- a/apps/api/src/alicebot_api/vnext_stores/sqlite/memory_access.py +++ b/apps/api/src/alicebot_api/vnext_stores/sqlite/memory_access.py @@ -31,6 +31,7 @@ _fts_match_expression, CTE_MATERIALIZED_HINT, _project_view_partition_sql, + _split_view_request, _stated_exclusion, _sqlite_ascii_literal_contains_sql, ) @@ -1004,6 +1005,25 @@ def count_rollup_input_memories( return cast(int, row["count"]) +def _rollup_group_scope_clause(self, projects: Sequence[str] | None) -> tuple[str, list[object]]: + """Overlap of scope or floor. Used only by the roll-up candidate and card lookups.""" + + normalized = tuple(normalize_project_scope(projects or ())) + scope_sql, scope_params = self._project_clause(normalized) + if not scope_sql: + return "", [] + ids, wants_global = _split_view_request(normalized) + if wants_global or not ids: + return scope_sql, list(scope_params) + predicate = scope_sql.removeprefix(" AND ") + floor_sql = ( + "EXISTS (SELECT 1 FROM json_each(alice_project_floor_identity(metadata_json)) AS floor_project " + "WHERE CAST(floor_project.value AS TEXT) " + f"IN ({self._placeholders(list(ids))}))" + ) + return f" AND ({predicate} OR {floor_sql})", [*scope_params, *ids] + + def list_pending_rollup_candidates( self, *, @@ -1030,7 +1050,7 @@ def list_pending_rollup_candidates( params.extend(domains) sensitivity_placeholders = ", ".join("?" for _value in sensitivity_allowed) params.extend(sensitivity_allowed) - project_sql, project_params = self._project_clause(tuple(normalize_project_scope(projects or ()))) + project_sql, project_params = _rollup_group_scope_clause(self, projects) params.extend(project_params) params.append(bounded_limit) return self._fetch_all( @@ -1087,7 +1107,7 @@ def list_accepted_rollup_cards( params.extend(domains) sensitivity_placeholders = ", ".join("?" for _value in sensitivity_allowed) params.extend(sensitivity_allowed) - project_sql, project_params = self._project_clause(tuple(normalize_project_scope(projects or ()))) + project_sql, project_params = _rollup_group_scope_clause(self, projects) params.extend(project_params) # An expired card is not the accepted card for its topic. The test sits # inside the ranking query, so an older card that is still open is ranked diff --git a/apps/api/src/alicebot_api/vnext_stores/sqlite/query_predicates.py b/apps/api/src/alicebot_api/vnext_stores/sqlite/query_predicates.py index 33f9c5bbe..5beed618a 100644 --- a/apps/api/src/alicebot_api/vnext_stores/sqlite/query_predicates.py +++ b/apps/api/src/alicebot_api/vnext_stores/sqlite/query_predicates.py @@ -12,6 +12,7 @@ from alicebot_api.vnext_project_scope import ( GLOBAL_PROJECT_MARKER, is_alice_project_id, + project_floor_shape, project_scope_identity, resolve_project_scope, resolve_source_metadata_project_scope, @@ -67,6 +68,24 @@ def _source_project_scope_identity_json_sqlite(metadata_json: object) -> str: return json.dumps(identity, ensure_ascii=False, separators=(",", ":")) +def _project_floor_identity_json_sqlite(metadata_json: object) -> str: + """Identity of ``project_floor`` when it is a list of strings, else ``[]``.""" + + if isinstance(metadata_json, Mapping): + metadata = dict(metadata_json) + elif isinstance(metadata_json, str): + try: + decoded = json.loads(metadata_json) + except (TypeError, ValueError): + decoded = {} + metadata = decoded if isinstance(decoded, dict) else {} + else: + metadata = {} + shape, floor = project_floor_shape({"metadata_json": metadata}) + identity = project_scope_identity(floor) if shape == "list" else () + return json.dumps(list(identity), ensure_ascii=False, separators=(",", ":")) + + def _ensure_project_scope_identity_sqlite(conn: sqlite3.Connection) -> None: """Install the deterministic project identity functions per connection.""" @@ -77,7 +96,8 @@ def _ensure_project_scope_identity_sqlite(conn: sqlite3.Connection) -> None: WHERE name IN ( 'alice_project_scope_value', 'alice_project_scope_identity', - 'alice_source_project_scope_identity' + 'alice_source_project_scope_identity', + 'alice_project_floor_identity' ) """ ) @@ -85,7 +105,7 @@ def _ensure_project_scope_identity_sqlite(conn: sqlite3.Connection) -> None: row = cursor.fetchone() if row is not None: count = next(iter(row.values())) if isinstance(row, Mapping) else row[0] - if int(count) == 3: + if int(count) == 4: return finally: cursor.close() @@ -107,6 +127,12 @@ def _ensure_project_scope_identity_sqlite(conn: sqlite3.Connection) -> None: _source_project_scope_identity_json_sqlite, deterministic=True, ) + conn.create_function( + "alice_project_floor_identity", + 1, + _project_floor_identity_json_sqlite, + deterministic=True, + ) #: ``AS MATERIALIZED`` for the labelled common table expression of the single-scan diff --git a/tests/unit/test_group_scope_consumers.py b/tests/unit/test_group_scope_consumers.py new file mode 100644 index 000000000..38693981b --- /dev/null +++ b/tests/unit/test_group_scope_consumers.py @@ -0,0 +1,302 @@ +"""Group scope for consolidation, roll-ups, and the lookups that find a card.""" + +from __future__ import annotations + +import inspect +import sqlite3 +from datetime import UTC, datetime +from uuid import uuid4 + +import pytest + +from alicebot_api.sqlite_schema import bootstrap_sqlite_schema +from alicebot_api.sqlite_store import SQLiteVNextStore, ensure_sqlite_user +from alicebot_api.vnext_agent_control import AgentIdentity, resource_project_scope +from alicebot_api.vnext_consolidation import _project_scope_key, _scoped_rows +from alicebot_api.vnext_derived_labels import group_scope +from alicebot_api.vnext_memory_commit import VNextMemoryCommitService, VNextMemoryCommitValidationError +from alicebot_api.vnext_project_scope import project_scope_identity +from alicebot_api.vnext_rollups import ROLLUP_CANDIDATE_KIND +from alicebot_api.vnext_rollups import _scoped_rows as rollup_scoped_rows +from alicebot_api.vnext_scheduler import SchedulerRunRequest, VNextSchedulerService +from alicebot_api.vnext_store import PostgresVNextStore +from alicebot_api.vnext_stores.postgres import memory_access as postgres_memory +from alicebot_api.routers.vnext_review import list_vnext_artifacts +from tests.unit.test_vnext_memory_commit import ( + TargetedLookupStore, + _seed_consolidation_candidate, + _seed_row, +) + +ALPHA = "prj_" + "a" * 16 +BETA = "prj_" + "b" * 16 + + +def _derived(scope: list[str], floor: list[str]) -> dict[str, object]: + return { + "id": "derived", + "memory_key": "vnext.consolidation.example", + "metadata_json": { + "candidate_kind": "memory_consolidation", + "project_scope": scope, + "project_floor": floor, + }, + } + + +def test_scoped_rows_overlap_the_group_scope() -> None: + row = _derived([], [ALPHA, BETA]) + assert group_scope(row) == (ALPHA, BETA) + kept = _scoped_rows([row], domains=None, sensitivity_allowed=["unknown"], projects=(ALPHA,)) + assert kept == [row] + kept_rollups = rollup_scoped_rows( + [row], domains=None, sensitivity_allowed=["unknown"], projects=(BETA,) + ) + assert kept_rollups == [row] + assert _project_scope_key(row) == (ALPHA, BETA) + + +def test_a_two_project_group_is_accepted_on_the_group_scope() -> None: + store = TargetedLookupStore() + members = [ + _seed_row(store, title=f"Member {index}", text="Shared fact.", metadata={"project_scope": [ALPHA, BETA]}) + for index in range(2) + ] + candidate_id = _seed_consolidation_candidate( + store, + member_ids=members, + proposal_kind="merge", + survivor_memory_id=None, + proposed_supersede=list(members), + ) + store.memories[candidate_id]["metadata_json"]["project_scope"] = [] + store.memories[candidate_id]["metadata_json"]["project_floor"] = [ALPHA, BETA] + + result = VNextMemoryCommitService(store).accept_consolidation_candidate( + candidate_id, reason="The group scope matches." + ) + + assert result["status"] == "accepted" + + +def test_m51_acceptance_that_compares_scope_again_refuses_the_group(monkeypatch: pytest.MonkeyPatch) -> None: + def scope_only(row: dict[str, object], *, kind: str | None = None) -> tuple[str, ...]: + del kind + return project_scope_identity(resource_project_scope(row)) + + monkeypatch.setattr("alicebot_api.vnext_memory_commit.group_scope", scope_only) + store = TargetedLookupStore() + members = [ + _seed_row(store, title=f"Wide {index}", text="Shared fact.", metadata={"project_scope": [ALPHA, BETA]}) + for index in range(2) + ] + candidate_id = _seed_consolidation_candidate( + store, + member_ids=members, + proposal_kind="merge", + survivor_memory_id=None, + proposed_supersede=list(members), + ) + store.memories[candidate_id]["metadata_json"]["project_scope"] = [] + store.memories[candidate_id]["metadata_json"]["project_floor"] = [ALPHA, BETA] + + with pytest.raises(VNextMemoryCommitValidationError, match="crosses project scopes"): + VNextMemoryCommitService(store).accept_consolidation_candidate(candidate_id, reason="Scope only.") + + +def test_promoted_copies_with_different_floors_are_refused() -> None: + store = TargetedLookupStore() + members = [ + _seed_row( + store, + title=f"Copy {index}", + text="Promoted text.", + metadata={ + "source_artifact_id": f"artifact-{index}", + "project_scope": [], + "project_floor": [project], + }, + ) + for index, project in enumerate((ALPHA, BETA)) + ] + candidate_id = _seed_consolidation_candidate( + store, + member_ids=members, + proposal_kind="merge", + survivor_memory_id=None, + proposed_supersede=list(members), + ) + + with pytest.raises(VNextMemoryCommitValidationError, match="crosses project scopes"): + VNextMemoryCommitService(store).accept_consolidation_candidate(candidate_id, reason="Different floors.") + + +def test_rollup_lookups_match_a_floor_when_the_scope_is_empty() -> None: + conn = sqlite3.connect(":memory:") + bootstrap_sqlite_schema(conn) + user_id = str(uuid4()) + ensure_sqlite_user(conn, user_id, "group-scope@example.com", "Group Scope") + store = SQLiteVNextStore(conn, user_id) + card = store.create_memory( + { + "memory_key": "vnext.rollup.floor-card", + "value": {"text": "card"}, + "status": "candidate", + "memory_type": "semantic", + "title": "Floor card", + "canonical_text": "Floor card", + "summary": "Floor card", + "domain": "personal", + "sensitivity": "internal", + "metadata_json": { + "candidate_kind": ROLLUP_CANDIDATE_KIND, + "rollup_digest": "digest-floor", + "rollup_key": "topic:floor", + "project_scope": [], + "project_floor": [ALPHA, BETA], + }, + } + ) + pending = store.list_pending_rollup_candidates( + rollup_digests=("digest-floor",), + domains=["personal"], + sensitivity_allowed=["internal"], + candidate_kind=ROLLUP_CANDIDATE_KIND, + limit=5, + projects=(ALPHA,), + ) + assert [str(row["id"]) for row in pending] == [str(card["id"])] + missed = store.list_pending_rollup_candidates( + rollup_digests=("digest-floor",), + domains=["personal"], + sensitivity_allowed=["internal"], + candidate_kind=ROLLUP_CANDIDATE_KIND, + limit=5, + projects=("prj_" + "c" * 16,), + ) + assert missed == [] + accepted = store.create_memory( + { + "memory_key": "vnext.rollup.floor-accepted", + "value": {"text": "accepted"}, + "status": "active", + "memory_type": "semantic", + "title": "Accepted floor card", + "canonical_text": "Accepted floor card", + "summary": "Accepted floor card", + "domain": "personal", + "sensitivity": "internal", + "metadata_json": { + "candidate_kind": ROLLUP_CANDIDATE_KIND, + "rollup_key": "topic:floor-accepted", + "project_scope": [], + "project_floor": [ALPHA, BETA], + }, + } + ) + cards = store.list_accepted_rollup_cards( + rollup_keys=("topic:floor-accepted",), + domains=["personal"], + sensitivity_allowed=["internal"], + candidate_kind=ROLLUP_CANDIDATE_KIND, + limit=5, + projects=(BETA,), + ) + assert [str(row["id"]) for row in cards] == [str(accepted["id"])] + conn.close() + + +def test_the_lookup_sql_names_the_group_scope() -> None: + pending = inspect.getsource(postgres_memory.list_pending_rollup_candidates) + accepted = inspect.getsource(postgres_memory.list_accepted_rollup_cards) + assert "_MEMORY_GROUP_SCOPE_SQL" in pending + assert "_MEMORY_GROUP_SCOPE_SQL" in accepted + assert "_PROJECT_FLOOR_SQL" in inspect.getsource(PostgresVNextStore.list_artifacts) + assert "project" in inspect.signature(list_vnext_artifacts).parameters + + +class _SweepStore: + def __init__(self) -> None: + self.memories = [ + { + "id": "alpha", + "status": "active", + "memory_type": "semantic", + "title": "Alpha only", + "canonical_text": "Alpha only", + "valid_to": datetime(2020, 1, 1, tzinfo=UTC), + "metadata_json": {"project_scope": [ALPHA]}, + "sensitivity": "internal", + }, + { + "id": "shared", + "status": "active", + "memory_type": "semantic", + "title": "Shared", + "canonical_text": "Shared", + "valid_to": datetime(2020, 1, 1, tzinfo=UTC), + "metadata_json": {"project_scope": [ALPHA, BETA]}, + "sensitivity": "internal", + }, + ] + self.artifacts: list[dict[str, object]] = [] + self.events: list[dict[str, object]] = [] + self.revisions: list[dict[str, object]] = [] + + def list_memories_for_staleness_sweep( + self, + *, + reference_time: object = None, + confirmation_before: object = None, + review_memory_types: object = None, + limit: int = 20, + projects: object = None, + ) -> list[dict[str, object]]: + del reference_time, confirmation_before, review_memory_types, limit, projects + return list(self.memories) + + def update_memory(self, *, memory_id: str, patch: dict[str, object], actor_type: str = "system") -> dict[str, object]: + del actor_type + for memory in self.memories: + if memory["id"] == memory_id: + memory.update(patch) + return memory + raise KeyError(memory_id) + + def append_revision(self, revision: dict[str, object], *, actor_type: str = "system") -> dict[str, object]: + del actor_type + self.revisions.append(revision) + return revision + + def append_event(self, event: dict[str, object]) -> dict[str, object]: + self.events.append(event) + return event + + def create_artifact(self, artifact: dict[str, object], *, actor_type: str = "system") -> dict[str, object]: + del actor_type + row = {**artifact, "id": "artifact-sweep"} + self.artifacts.append(row) + return row + + +def test_a_locked_staleness_sweep_does_not_mark_a_shared_memory() -> None: + store = _SweepStore() + identity = AgentIdentity( + agent_id="alpha-key", + project_scope=(ALPHA,), + project_scope_locked=True, + permission_profile="trusted_local_agent", + ) + VNextSchedulerService(store)._run_staleness_sweep( # noqa: SLF001 + SchedulerRunRequest( + workflow_type="staleness_sweep", + projects=(ALPHA,), + agent_identity=identity, + ), + metadata={"scheduler_run_id": "run-1", "trace_id": "trace-1"}, + ) + by_id = {memory["id"]: memory for memory in store.memories} + assert by_id["alpha"]["status"] == "stale" + assert by_id["shared"]["status"] == "active" + derived = store.artifacts[0]["metadata_json"]["derived_from"] + assert derived["memories"] == ["alpha"] diff --git a/tests/unit/test_store_memory_access_split.py b/tests/unit/test_store_memory_access_split.py index 7b2647bbf..f04ef0ec9 100644 --- a/tests/unit/test_store_memory_access_split.py +++ b/tests/unit/test_store_memory_access_split.py @@ -27,8 +27,11 @@ "apps/api/src/alicebot_api/vnext_stores/retrieval_common.py": ( "fa1a3a90511b5c61754ba29560e91b7b3058a48d47c143b09d8505d52025b8cc" ), + # Re-minted for the roll-up group scope: a floor identity beside the scope + # identity, used only by the two roll-up lookups and the artifact list. + # Previous receipt f0ec9c7f... "apps/api/src/alicebot_api/vnext_stores/postgres/query_predicates.py": ( - "f0ec9c7f13bc7bf93f5a3beaa86916a04e45200ef0296d6f9288eed3912be33d" + "f15f1bf2b3da73758c925b63dffaec707f91f00a6e153e53f562fb188859236d" ), # Re-minted for ``get_memory_by_key(include_deleted=...)`` (reviewed change, not drift; see the SQLite # entry below). Previous Postgres receipt 49748ecd... @@ -36,16 +39,20 @@ # deletion; reviewed change, not drift). Proof: the only difference from origin/main daa46ef5 is that one function, # which takes a keyword-only ``include_deleted`` (false by default, so every caller reads what it read before) # and drops the ``deleted_at IS NULL`` clause only when it is true. Previous Postgres receipt f642880f... + # Re-minted so the two roll-up lookups overlap scope united with floor. + # Every other statement still uses the scope expression. Previous receipt 46946cc0... "apps/api/src/alicebot_api/vnext_stores/postgres/memory_access.py": ( - "46946cc087de35f54474adf47cadcd67b862685bfa67a38be277d8e58a00c47e" + "057f4f0c157223fc0d94ea3afe680b66200820533ddccab7211e211f29107157" ), # Re-minted for per-project memory S2 (2026-10-02): the project fence builders read the reserved global # marker and take the domains to leave out, and the single-scan partition SQL and the materialized-CTE hint # are new. The Postgres carrier is unchanged on purpose: the Postgres runtime resolves no project view. # Re-minted once more in the S2 review round (2026-10-02): a request that holds the marker and does not state which # global domains it leaves out raises (reviewed change, not drift). + # Re-minted for alice_project_floor_identity, the fourth identity function, + # used by the roll-up lookups. Previous receipt eab46f16... "apps/api/src/alicebot_api/vnext_stores/sqlite/query_predicates.py": ( - "eab46f165564c212db21b6b6b621ecb447aaa86d48aba6512bd5f2e88f20bd82" + "8beae59c379c56033388a35b5a152f1f683f6b55c6e2fbd62925723566a12202" ), # Re-minted for the Phase 4 Stage 2 resident vector cache (reviewed # carrier change; the receipt guards unreviewed drift): the vector scan @@ -76,8 +83,9 @@ # reviewed change, not drift). Proof: the only difference from origin/main daa46ef5 is that one function, which # takes a keyword-only ``include_deleted`` (false by default) and drops the ``deleted_at IS NULL`` clause only when # it is true. Previous SQLite receipt 91636de9... + # Re-minted so the two roll-up lookups overlap scope or floor. Previous receipt 64f21989... "apps/api/src/alicebot_api/vnext_stores/sqlite/memory_access.py": ( - "64f21989d3bb05d742dd712b310511d3c63f32b9a4d62af6b888ff2b367f0b3b" + "1580dca3a31fbbcf98539e71e81bad13935f30c59e8a7c75b0fc0b4472a6d5fa" ), } @@ -165,6 +173,9 @@ "_jsonb_project_scope_leaf_values_sql", "_jsonb_source_project_scope_values_sql", "_MEMORY_PROJECT_SCOPE_SQL", + "_PROJECT_FLOOR_SQL", + "_MEMORY_GROUP_SCOPE_SQL", + "_jsonb_string_array_identity_sql", "_MEMORY_DIRECT_PEOPLE_SQL", "_MEMORY_SCOPE_EVENT_TIME_SQL", "_SCOPED_MEMORY_PROJECT_SQL", @@ -200,7 +211,9 @@ # Per-file importer savepoint (2026-10-02): one paired method more, ``savepoint``, appended last. # Previous receipt: (171, 526374782104a2a1...). Proof: the member list equals the list at origin/main # 040a2a10 with ``savepoint`` added at the end and nothing else moved (reviewed change, not drift). - "PostgresVNextStore": (172, "6f1a459fcf4319cf4281f6cc0d4e81679c3e05d851fd0a874a2d90298d7c2569"), + # Label lock: lock_label_writes and read_label_rows follow __init__. Dropping + # those two names restores the previous receipt (172, 6f1a459f...). + "PostgresVNextStore": (174, "095250b8a77d6c0a32d2783343916b947bcae8ae86e3a1693e47c2fb11802211"), # One SQLite-only method more, ``check_source_search_query``: the Postgres # source search has no expression-depth or LIKE-length limit to check. # Merge of #500 and #502 (2026-10-01): one more SQLite-only method, @@ -215,7 +228,9 @@ # 040a2a10 with ``savepoint`` added at the end and nothing else moved (reviewed change, not drift). # Proof: the replacement branch gains only scrub_source, source_inventory and # prunable_sources here; every pre-existing class member keeps its order. - "SQLiteVNextStore": (134, "1301272026897057cf071009cc21787543ddc326f1e06f1a75a763f3e344767e"), + # Label lock: lock_label_writes and read_label_rows follow __init__. Dropping + # those two names restores the previous receipt (134, 13012720...). + "SQLiteVNextStore": (136, "1d99dbaf69e4ea888ca7beb2389ac176650a2e573c067bf0686adc9e609132f5"), } @@ -302,7 +317,7 @@ def test_memory_access_source_receipts_pin_sql_parameters_and_comments() -> None assert sqlite.count("user_id = ?") >= 20 assert "Only compare vectors from the same endpoint fingerprint" in sqlite assert "resolve_project_scope" in sqlite_predicates - assert sqlite_predicates.count("create_function(") == 3 + assert sqlite_predicates.count("create_function(") == 4 def test_memory_access_methods_are_direct_grafts_in_native_backend_order() -> None: diff --git a/tests/unit/test_vnext_brain.py b/tests/unit/test_vnext_brain.py index af808365c..5f496e59c 100644 --- a/tests/unit/test_vnext_brain.py +++ b/tests/unit/test_vnext_brain.py @@ -155,6 +155,11 @@ def test_daily_brief_generates_dated_reviewable_artifact_with_sources_and_open_l assert store.open_loops[-1]["metadata_json"]["candidate"] is True assert store.events[-1]["event_type"] == "artifact.generated" assert store.events[-1]["payload_json"]["workflow"] == "daily_brief" + derived = artifact["metadata_json"]["derived_from"] + assert derived["counts"]["sources"] == len(derived["sources"]) == 1 + assert derived["sources"] == ["source-1"] + assert "memory-1" in derived["memories"] + assert store.open_loops[-1]["metadata_json"]["derived_from"]["sources"] == ["source-1"] def test_daily_brief_respects_sensitivity_filtering() -> None: From 1d3efca6f23df5c95c79326395ae09200f3c17fd Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 18:22:08 +0000 Subject: [PATCH 06/90] Judge exact artifact and memory reads by input labels. The artifact authorization door and a write that cites a memory settle a derived row before the policy check. A locked key is refused when that row cannot be checked. The owner and an unbound admin still read the stored row. --- CHANGELOG.md | 1 + .../src/alicebot_api/routers/_vnext_shared.py | 19 +- .../api/src/alicebot_api/vnext_label_guard.py | 219 ++++++++++++++++++ .../src/alicebot_api/vnext_source_fence.py | 14 +- tests/unit/test_label_guard_exact_doors.py | 157 +++++++++++++ 5 files changed, 402 insertions(+), 8 deletions(-) create mode 100644 apps/api/src/alicebot_api/vnext_label_guard.py create mode 100644 tests/unit/test_label_guard_exact_doors.py diff --git a/CHANGELOG.md b/CHANGELOG.md index 7be7ec35d..80ac80757 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,7 @@ ## Unreleased +- Unreleased (on main, not in v0.20.0): an exact read of a derived artifact, and a write that cites a derived memory, uses the labels of that row's inputs. v0.20.0 used the label stored on the row, so a public copy of a confidential input could be read. A row whose inputs cannot be checked is refused to a locked key. The owner and an unbound admin still read the stored row. No migration is required. - Unreleased (on main, not in v0.20.0): a locked key's consolidation, roll-up, staleness sweep and open-loop review keep only inputs whose scope and floor are both inside the binding. The overlap check stays, and the exact test runs after it. Consolidation and roll-ups compare the group scope (scope united with floor), so a card stored with an empty scope and a floor of two projects is still found and can be accepted. The operator artifact list, given a project, matches that floor as well as the scope. Each new report and aggregate records `derived_from` for the rows it used. v0.20.0 selected by overlap of the scope alone and wrote no `derived_from`. No migration is required. - Unreleased (on main, not in v0.20.0): a memory copied from another row is stored at least as strict as that row. v0.20.0 kept the label the copy was given at insert, so a later reader could see a health or confidential input through a copy stored as `unknown` and `public`. The insert now re-reads the inputs inside the write and raises the domain, the sensitivity, and the project floor. A later metadata write cannot drop the marker that makes the row derived, or put back an older project scope or project floor. Raising a source or a memory walks the rows derived from it and raises those too. A candidate open loop and a generated artifact take the same insert floor. Replacing a SQLite source with a stricter one raises the memories that cite it before they are retired. Moving a source to another project answers with how many derived rows a project-bound key would lose, and waits for confirm_label_hide before it writes. A relabel that cannot finish answers 409, and one that waits more than 3 seconds for the label lock answers 503. No migration is required. - Unreleased (on main, not in v0.20.0): the label of a derived row (its domain, its sensitivity, and the projects it requires) is now decided by one pure function. v0.20.0 kept the label a derived row was given when it was made. The function is what the write path calls. A read that does not go through that path still behaves as it does in v0.20.0. A locked key is refused when a caller passes a project floor that is not inside the key's binding. No migration is required. diff --git a/apps/api/src/alicebot_api/routers/_vnext_shared.py b/apps/api/src/alicebot_api/routers/_vnext_shared.py index 2edc94817..7b86ca670 100644 --- a/apps/api/src/alicebot_api/routers/_vnext_shared.py +++ b/apps/api/src/alicebot_api/routers/_vnext_shared.py @@ -25,7 +25,9 @@ agent_key_from_authorization, resolve_protected_agent_identity, ) +from alicebot_api.vnext_label_guard import LabelGuard, apply_unverified_rule, policy_labels from alicebot_api.vnext_project_scope import source_project_scope +from alicebot_api.vnext_source_fence import SourceReadFence from alicebot_api.vnext_queue import VNextQueueNotFoundError from alicebot_api.vnext_store import PostgresVNextStore, is_redacted_project_update_artifact @@ -461,14 +463,16 @@ def _vnext_exact_resource_policy( resource: dict[str, object], source_resource: bool = False, ) -> PolicyDecision: - domain = " ".join(str(resource.get("domain") or "unknown").split()).strip() or "unknown" - sensitivity = " ".join(str(resource.get("sensitivity") or "unknown").split()).strip() or "unknown" + domains, sensitivity_allowed, project_scope, project_floor = policy_labels(resource) + if source_resource: + project_scope = source_project_scope(resource) decision = evaluate_agent_policy( identity=identity, action=action, - domains=(domain,), - sensitivity_allowed=(sensitivity,), - project_scope=source_project_scope(resource) if source_resource else resource_project_scope(resource), + domains=domains, + sensitivity_allowed=sensitivity_allowed, + project_scope=project_scope, + project_floor=project_floor, require_explicit_project_scope=bool(identity is not None and identity.project_scope_locked), ) if decision.decision == "allowed_with_filtering": @@ -504,11 +508,14 @@ def _vnext_authorized_artifact( raise ValueError("feedback cannot be added to a redacted artifact") _vnext_agent_record(store, identity) + guard = LabelGuard.for_fence(store, SourceReadFence.for_identity(identity)) + effective = guard.effective_row("artifact", artifact) decision = _vnext_exact_resource_policy( identity=identity, action=action, - resource=artifact, + resource=effective if isinstance(effective, dict) else artifact, ) + decision = apply_unverified_rule(decision, effective if isinstance(effective, Mapping) else None, identity) append_policy_events( store, identity=identity, diff --git a/apps/api/src/alicebot_api/vnext_label_guard.py b/apps/api/src/alicebot_api/vnext_label_guard.py new file mode 100644 index 000000000..f49b3763f --- /dev/null +++ b/apps/api/src/alicebot_api/vnext_label_guard.py @@ -0,0 +1,219 @@ +"""Read-time check for a derived row. + +One guard per request. An inactive guard returns its input and reads nothing. +An active guard settles the row with its inputs and answers the door with that +effective label. +""" + +from __future__ import annotations + +from collections.abc import Mapping, Sequence +from dataclasses import dataclass, replace +from typing import Any + +from alicebot_api.vnext_agent_control import ( + ALL_SENSITIVITY, + VNEXT_DOMAINS, + AgentIdentity, + PolicyDecision, +) +from alicebot_api.vnext_derived_labels import ( + HOP_BOUND, + NODE_BOUND, + canon_kind, + dependencies_of, + is_derived, + settle_labels, +) +from alicebot_api.vnext_project_scope import project_floor_shape, project_scopes_overlap, resolve_project_scope + + +_GUARD_USER = "label-guard" + + +def _filters_admit_every( + domains: Sequence[str] | None, + sensitivity_allowed: Sequence[str] | None, + projects: Sequence[str] | None, +) -> bool: + domain_list = tuple(domains or ()) + domains_open = not domain_list or set(domain_list) >= set(VNEXT_DOMAINS) + sensitivity_list = tuple(sensitivity_allowed or ()) + sensitivity_open = set(sensitivity_list) >= set(ALL_SENSITIVITY) + return domains_open and sensitivity_open and not tuple(projects or ()) + + +@dataclass +class LabelGuard: + """The labels a door may trust for one request.""" + + store: Any + active: bool + domains: tuple[str, ...] = () + sensitivity_allowed: tuple[str, ...] = () + projects: tuple[str, ...] = () + _nodes: dict[tuple[str, str], dict[str, object]] | None = None + + @classmethod + def for_fence(cls, store: Any, fence: Any) -> LabelGuard: + """Exact doors. Inactive for the owner and for an unbound admin.""" + + fenced = bool(getattr(fence, "entity_read_fenced", False)) + return cls(store=store, active=fenced) + + @classmethod + def for_filters( + cls, + store: Any, + domains: Sequence[str] | None, + sensitivity_allowed: Sequence[str] | None, + projects: Sequence[str] | None = (), + exclude_global_domains: Sequence[str] | None = None, + ) -> LabelGuard: + """List doors. Inactive when the filters admit every label.""" + + del exclude_global_domains + domain_list = tuple(domains or ()) + sensitivity_list = tuple(sensitivity_allowed or ()) + project_list = tuple(projects or ()) + return cls( + store=store, + active=not _filters_admit_every(domain_list, sensitivity_list, project_list), + domains=domain_list, + sensitivity_allowed=sensitivity_list, + projects=project_list, + ) + + def effective_row(self, kind: str, row: Mapping[str, object] | None) -> Mapping[str, object] | None: + """A copy whose domain, sensitivity, scope and floor are effective.""" + + if row is None or not self.active or not isinstance(row, Mapping): + return row + if not is_derived(kind, row): + return row + nodes = self._collected(kind, row) + settled = settle_labels(nodes, on_cycle="unverified", max_hops=HOP_BOUND, max_nodes=NODE_BOUND) + label = settled.by_stored(kind, str(row.get("id") or ""), user_id=_GUARD_USER) + copy = dict(row) + metadata = dict(copy.get("metadata_json")) if isinstance(copy.get("metadata_json"), Mapping) else {} + if label.unverified: + copy["domain"] = label.domain + copy["sensitivity"] = "regulated" + metadata["project_scope"] = [] + metadata["project_floor"] = [] + copy["unverified"] = True + else: + copy["domain"] = label.domain + copy["sensitivity"] = label.sensitivity + metadata["project_scope"] = list(label.project_scope) + metadata["project_floor"] = list(label.project_floor) + copy["unverified"] = False + copy["metadata_json"] = metadata + return copy + + def admit_rows(self, kind: str, rows: Sequence[Mapping[str, object]]) -> list[Mapping[str, object]]: + """Rows whose effective labels pass this guard's filters. Originals of the rows, not copies.""" + + if not self.active: + return [row for row in rows if isinstance(row, Mapping)] + kept: list[Mapping[str, object]] = [] + for row in rows: + if not isinstance(row, Mapping): + continue + effective = self.effective_row(kind, row) + if isinstance(effective, Mapping) and self._admits_effective(effective): + kept.append(row) + return kept + + def admit_beliefs(self, beliefs: Sequence[Mapping[str, object]]) -> list[Mapping[str, object]]: + """Beliefs whose backing memory the filters admit. One batched read.""" + + if not self.active: + return [row for row in beliefs if isinstance(row, Mapping)] + ids = [str(row.get("memory_id")) for row in beliefs if isinstance(row, Mapping) and row.get("memory_id")] + reader = getattr(self.store, "read_label_rows", None) + found: dict[str, Mapping[str, object]] = {} + if callable(reader) and ids: + for row in reader("memory", ids): + if isinstance(row, Mapping) and row.get("id") is not None: + found[str(row.get("id"))] = row + admitted = {str(row.get("id")) for row in self.admit_rows("memory", list(found.values()))} + return [ + row + for row in beliefs + if isinstance(row, Mapping) and str(row.get("memory_id") or "") in admitted + ] + + def _admits_effective(self, row: Mapping[str, object]) -> bool: + domain = str(row.get("domain") or "unknown") + if self.domains and domain not in self.domains and domain != "unknown": + return False + sensitivity = str(row.get("sensitivity") or "unknown") + if self.sensitivity_allowed and sensitivity not in self.sensitivity_allowed: + return False + if self.projects: + scope = resolve_project_scope(row).values + _shape, floor = project_floor_shape(row) + if not project_scopes_overlap(scope, self.projects, floor=floor): + return False + return True + + def _collected(self, kind: str, row: Mapping[str, object]) -> list[dict[str, object]]: + if self._nodes is None: + self._nodes = {} + pending: list[tuple[str, Mapping[str, object]]] = [(canon_kind(kind), row)] + hops = 0 + while pending and len(self._nodes) < NODE_BOUND and hops < HOP_BOUND: + hops += 1 + name, current = pending.pop(0) + node_id = str(current.get("id") or "") + key = (name, node_id) + if key in self._nodes: + continue + node = dict(current) + node["kind"] = name + node["user_id"] = _GUARD_USER + self._nodes[key] = node + if not is_derived(name, node): + continue + grouped: dict[str, list[str]] = {} + for dep_kind, dep_id in dependencies_of(name, node): + grouped.setdefault(canon_kind(dep_kind), []).append(str(dep_id)) + reader = getattr(self.store, "read_label_rows", None) + if not callable(reader): + continue + for dep_kind, ids in grouped.items(): + missing = [item for item in ids if (dep_kind, item) not in self._nodes] + if not missing: + continue + for found in reader(dep_kind, missing): + if isinstance(found, Mapping): + pending.append((dep_kind, found)) + return list(self._nodes.values()) + + +def policy_labels( + row: Mapping[str, object], +) -> tuple[tuple[str, ...], tuple[str, ...], tuple[str, ...], tuple[str, ...]]: + """Domain, sensitivity, scope and floor to hand the policy engine.""" + + domain = " ".join(str(row.get("domain") or "unknown").split()).strip() or "unknown" + sensitivity = " ".join(str(row.get("sensitivity") or "unknown").split()).strip() or "unknown" + scope = resolve_project_scope(row).values + _shape, floor = project_floor_shape(row) + return (domain,), (sensitivity,), scope, floor + + +def apply_unverified_rule( + decision: PolicyDecision, + row: Mapping[str, object] | None, + identity: AgentIdentity | None, +) -> PolicyDecision: + """A locked identity is refused an unverified derived row.""" + + unverified = isinstance(row, Mapping) and bool(row.get("unverified")) + locked = identity is not None and bool(identity.project_scope_locked) + if not unverified or not locked: + return decision + reasons = tuple(dict.fromkeys((*decision.reasons, "derived_labels_unverified"))) + return replace(decision, decision="blocked", reasons=reasons) diff --git a/apps/api/src/alicebot_api/vnext_source_fence.py b/apps/api/src/alicebot_api/vnext_source_fence.py index e251dd26f..d46d530ba 100644 --- a/apps/api/src/alicebot_api/vnext_source_fence.py +++ b/apps/api/src/alicebot_api/vnext_source_fence.py @@ -122,7 +122,7 @@ evaluate_agent_policy, resource_project_scope, ) -from alicebot_api.vnext_project_scope import source_project_scope +from alicebot_api.vnext_project_scope import project_floor_shape, source_project_scope SOURCE_REF_NOT_FOUND_MESSAGE = "the cited source was not found in the current user scope" MEMORY_REF_NOT_FOUND_MESSAGE = "the cited memory was not found in the current user scope" @@ -229,12 +229,16 @@ def _admits(self, row: Mapping[str, object], *, project_scope: tuple[str, ...]) return False if self.identity is None: return True + if row.get("unverified") and self.identity.project_scope_locked: + return False + _shape, floor = project_floor_shape(row) decision = evaluate_agent_policy( identity=self.identity, action=EXPLAIN_DISCLOSURE_ACTION, domains=(str(row.get("domain") or "unknown"),), sensitivity_allowed=(str(row.get("sensitivity") or "unknown"),), project_scope=project_scope, + project_floor=floor, require_explicit_project_scope=True, ) # "allowed_with_filtering" is a refusal here, as it is for explain: the @@ -361,7 +365,13 @@ def resolve_attachable_memory_id(store: object, memory_id: str, *, fence: Source raise MemoryRefNotFoundError() from None getter = getattr(store, "get_memory", None) row = getter(canonical) if callable(getter) else None - if not isinstance(row, Mapping) or not fence.admits_memory(row): + if not isinstance(row, Mapping): + raise MemoryRefNotFoundError() + from alicebot_api.vnext_label_guard import LabelGuard + + effective = LabelGuard.for_fence(store, fence).effective_row("memory", row) + judged = effective if isinstance(effective, Mapping) else row + if not fence.admits_memory(judged): raise MemoryRefNotFoundError() return canonical diff --git a/tests/unit/test_label_guard_exact_doors.py b/tests/unit/test_label_guard_exact_doors.py new file mode 100644 index 000000000..3e75a7d0d --- /dev/null +++ b/tests/unit/test_label_guard_exact_doors.py @@ -0,0 +1,157 @@ +"""Exact doors judge a derived row by the labels of its inputs.""" + +from __future__ import annotations + +import pytest + +from alicebot_api.routers._vnext_shared import _vnext_authorized_artifact +from alicebot_api.vnext_agent_control import AgentIdentity, AgentPolicyBlockedError +from alicebot_api.vnext_label_guard import LabelGuard, apply_unverified_rule +from alicebot_api.vnext_source_fence import ( + MemoryRefNotFoundError, + SourceReadFence, + resolve_attachable_memory_id, +) + + +ALPHA = "prj_" + "a" * 16 +SOURCE_ID = "11111111-1111-1111-1111-111111111111" +ARTIFACT_ID = "22222222-2222-2222-2222-222222222222" +MEMORY_ID = "33333333-3333-3333-3333-333333333333" + + +class _LabelStore: + def __init__(self) -> None: + self.source = { + "id": SOURCE_ID, + "user_id": "label-guard", + "domain": "health", + "sensitivity": "confidential", + "metadata_json": {"project_scope": [ALPHA]}, + } + self.artifact = { + "id": ARTIFACT_ID, + "domain": "unknown", + "sensitivity": "public", + "content_markdown": "secret health note", + "metadata_json": { + "workflow": "daily_brief", + "project_scope": [ALPHA], + "derived_from": { + "v": 1, + "sources": [SOURCE_ID], + "memories": [], + "open_loops": [], + "artifacts": [], + "beliefs": [], + "counts": {"sources": 1, "memories": 0, "open_loops": 0, "artifacts": 0, "beliefs": 0}, + }, + }, + } + self.memory = { + "id": MEMORY_ID, + "domain": "unknown", + "sensitivity": "public", + "canonical_text": "copied", + "metadata_json": { + "source_id": SOURCE_ID, + "project_scope": [ALPHA], + "derived_from": { + "v": 1, + "sources": [SOURCE_ID], + "memories": [], + "open_loops": [], + "artifacts": [], + "beliefs": [], + "counts": {"sources": 1, "memories": 0, "open_loops": 0, "artifacts": 0, "beliefs": 0}, + }, + }, + } + self.events: list[dict[str, object]] = [] + + def read_label_rows(self, kind: str, ids: list[str]) -> list[dict[str, object]]: + if kind == "source" and SOURCE_ID in ids: + return [self.source] + return [] + + def get_artifact(self, artifact_id: str) -> dict[str, object] | None: + return self.artifact if artifact_id == ARTIFACT_ID else None + + def get_memory(self, memory_id: str) -> dict[str, object] | None: + return self.memory if memory_id == MEMORY_ID else None + + def upsert_agent_identity(self, *_args: object, **_kwargs: object) -> None: + return None + + def append_event(self, event: dict[str, object]) -> dict[str, object]: + self.events.append(event) + return event + + +def _trusted() -> AgentIdentity: + return AgentIdentity(agent_id="trusted-key", permission_profile="trusted_local_agent") + + +def _locked_admin() -> AgentIdentity: + return AgentIdentity( + agent_id="bound-admin", + permission_profile="admin_agent", + project_scope=(ALPHA,), + project_scope_locked=True, + ) + + +def test_an_exact_artifact_door_uses_the_input_label() -> None: + store = _LabelStore() + with pytest.raises(AgentPolicyBlockedError): + _vnext_authorized_artifact( + store=store, # type: ignore[arg-type] + identity=_trusted(), + artifact_id=ARTIFACT_ID, + action="artifact.read", + for_update=False, + ) + + +def test_the_owner_guard_does_not_read() -> None: + store = _LabelStore() + guard = LabelGuard.for_fence(store, SourceReadFence.for_identity(None)) + assert guard.active is False + assert guard.effective_row("artifact", store.artifact) is store.artifact + + +def test_a_locked_key_is_refused_an_unverified_row() -> None: + store = _LabelStore() + store.artifact["metadata_json"] = { + "workflow": "daily_brief", + "project_scope": [ALPHA], + "derived_from": { + "v": 1, + "sources": ["99999999-9999-9999-9999-999999999999"], + "memories": [], + "open_loops": [], + "artifacts": [], + "beliefs": [], + "counts": {"sources": 1, "memories": 0, "open_loops": 0, "artifacts": 0, "beliefs": 0}, + }, + } + identity = _locked_admin() + guard = LabelGuard.for_fence(store, SourceReadFence.for_identity(identity)) + effective = guard.effective_row("artifact", store.artifact) + assert isinstance(effective, dict) + assert effective["unverified"] is True + from alicebot_api.vnext_agent_control import PolicyDecision + + blocked = apply_unverified_rule( + PolicyDecision(decision="allowed", action="artifact.read", permission_profile="admin_agent"), + effective, + identity, + ) + assert blocked.decision == "blocked" + assert "derived_labels_unverified" in blocked.reasons + + +def test_a_cited_memory_is_judged_by_its_source() -> None: + store = _LabelStore() + with pytest.raises(MemoryRefNotFoundError): + resolve_attachable_memory_id(store, MEMORY_ID, fence=SourceReadFence.for_identity(_trusted())) From a97fbb36a656069e08906b1545531b6d5c270440 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 18:29:25 +0000 Subject: [PATCH 07/90] Pass a row floor into project view checks. A view that asks for global rows keeps a row with no Alice project id only when every Alice project id in its floor is in the view. The four Python checks and the SQLite view SQL now pass that floor. --- CHANGELOG.md | 2 +- .../src/alicebot_api/mcp/retrieval_shared.py | 8 +- apps/api/src/alicebot_api/session_briefing.py | 17 ++++- apps/api/src/alicebot_api/vnext_retrieval.py | 34 +++++++-- .../vnext_stores/sqlite/graph_open_loops.py | 1 + .../vnext_stores/sqlite/memory_access.py | 1 + .../vnext_stores/sqlite/query_predicates.py | 21 ++++++ .../unit/test_store_graph_open_loops_split.py | 19 +++-- tests/unit/test_store_memory_access_split.py | 8 +- tests/unit/test_view_floor.py | 75 +++++++++++++++++++ 10 files changed, 168 insertions(+), 18 deletions(-) create mode 100644 tests/unit/test_view_floor.py diff --git a/CHANGELOG.md b/CHANGELOG.md index 7be7ec35d..b3b183fbf 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,7 +2,7 @@ ## Unreleased -- Unreleased (on main, not in v0.20.0): a locked key's consolidation, roll-up, staleness sweep and open-loop review keep only inputs whose scope and floor are both inside the binding. The overlap check stays, and the exact test runs after it. Consolidation and roll-ups compare the group scope (scope united with floor), so a card stored with an empty scope and a floor of two projects is still found and can be accepted. The operator artifact list, given a project, matches that floor as well as the scope. Each new report and aggregate records `derived_from` for the rows it used. v0.20.0 selected by overlap of the scope alone and wrote no `derived_from`. No migration is required. +- Unreleased (on main, not in v0.20.0): a locked key's consolidation, roll-up, staleness sweep and open-loop review keep only inputs whose scope and floor are both inside the binding. The overlap check stays, and the exact test runs after it. Consolidation and roll-ups compare the group scope (scope united with floor), so a card stored with an empty scope and a floor of two projects is still found and can be accepted. The operator artifact list, given a project, matches that floor as well as the scope. Each new report and aggregate records `derived_from` for the rows it used. A project view that also asks for global rows keeps a row with no Alice project id only when every Alice project id in its floor is in that view. v0.20.0 selected by overlap of the scope alone and wrote no `derived_from`. No migration is required. - Unreleased (on main, not in v0.20.0): a memory copied from another row is stored at least as strict as that row. v0.20.0 kept the label the copy was given at insert, so a later reader could see a health or confidential input through a copy stored as `unknown` and `public`. The insert now re-reads the inputs inside the write and raises the domain, the sensitivity, and the project floor. A later metadata write cannot drop the marker that makes the row derived, or put back an older project scope or project floor. Raising a source or a memory walks the rows derived from it and raises those too. A candidate open loop and a generated artifact take the same insert floor. Replacing a SQLite source with a stricter one raises the memories that cite it before they are retired. Moving a source to another project answers with how many derived rows a project-bound key would lose, and waits for confirm_label_hide before it writes. A relabel that cannot finish answers 409, and one that waits more than 3 seconds for the label lock answers 503. No migration is required. - Unreleased (on main, not in v0.20.0): the label of a derived row (its domain, its sensitivity, and the projects it requires) is now decided by one pure function. v0.20.0 kept the label a derived row was given when it was made. The function is what the write path calls. A read that does not go through that path still behaves as it does in v0.20.0. A locked key is refused when a caller passes a project floor that is not inside the key's binding. No migration is required. - Unreleased (on main, not in v0.20.0): on SQLite, `sources delete`, `sources prune --superseded` and `import-markdown --supersede` now scrub an open loop that names the source only in its metadata: the id, or `source:`, as the text under `source_id`, `source_ids`, `source_ref`, `source_refs`, `source_references` or `selected_source_ids` at any depth. That is the rule the open-loop lookup of a source already used. They blanked only loops whose `source_id` column held the id, so a loop with an empty column kept its title, description and metadata, and they reached a new export. The delete preview now counts the same loops as the receipt. The rule reads the id only as stored or as `source:`: a loop that names the source in any spelling other than those two keeps its text, although the memory rule reads several other spellings. No migration is required. diff --git a/apps/api/src/alicebot_api/mcp/retrieval_shared.py b/apps/api/src/alicebot_api/mcp/retrieval_shared.py index fa1081c62..c0586a250 100644 --- a/apps/api/src/alicebot_api/mcp/retrieval_shared.py +++ b/apps/api/src/alicebot_api/mcp/retrieval_shared.py @@ -12,6 +12,7 @@ from alicebot_api.vnext_agent_control import resource_project_scope from alicebot_api.vnext_project_scope import ( is_global_scope, + project_floor_shape, project_identifier_identity, project_scopes_overlap, ) @@ -160,7 +161,12 @@ def _provenance_count(store: SQLiteVNextStore, memory_id: object) -> int: def _resource_matches_project_scope(resource: Mapping[str, object], project_scope: tuple[str, ...]) -> bool: if not project_scope: return True - return project_scopes_overlap(resource_project_scope(resource), project_scope) + shape, floor = project_floor_shape(resource) + return project_scopes_overlap( + resource_project_scope(resource), + project_scope, + floor=floor if shape == "list" else (), + ) def _resource_is_held_back_global(resource: Mapping[str, object], exclude_global_domains: frozenset[str]) -> bool: diff --git a/apps/api/src/alicebot_api/session_briefing.py b/apps/api/src/alicebot_api/session_briefing.py index 202c88a67..f297cb97b 100644 --- a/apps/api/src/alicebot_api/session_briefing.py +++ b/apps/api/src/alicebot_api/session_briefing.py @@ -59,6 +59,7 @@ ) from alicebot_api.vnext_project_scope import ( is_global_scope, + project_floor_shape, project_scope_identity, project_scopes_overlap, source_project_scope, @@ -624,7 +625,7 @@ def _memory_honours_fence( return ( _matches_domains(row, effective_domains) and _matches_sensitivity(row, effective_sensitivity_allowed) - and _matches_project_scope(resource_scope, effective_project_scope) + and _matches_project_scope(resource_scope, effective_project_scope, floor=_brief_floor(row)) and not _is_held_back(row, resource_scope, exclude_global_domains) ) @@ -676,10 +677,20 @@ def _matches_sensitivity(row: Mapping[str, object], sensitivity_allowed: tuple[s return (row.get("sensitivity") or "unknown") in sensitivity_allowed -def _matches_project_scope(resource_scope: tuple[str, ...], project_scope: tuple[str, ...]) -> bool: +def _brief_floor(row: Mapping[str, object]) -> tuple[str, ...]: + shape, floor = project_floor_shape(row) + return floor if shape == "list" else () + + +def _matches_project_scope( + resource_scope: tuple[str, ...], + project_scope: tuple[str, ...], + *, + floor: tuple[str, ...] = (), +) -> bool: if not project_scope: return True - return project_scopes_overlap(resource_scope, project_scope) + return project_scopes_overlap(resource_scope, project_scope, floor=floor) # A fact used as the excerpt query is passed to the source search whole, and diff --git a/apps/api/src/alicebot_api/vnext_retrieval.py b/apps/api/src/alicebot_api/vnext_retrieval.py index a7671d0ab..f85ffb2b8 100644 --- a/apps/api/src/alicebot_api/vnext_retrieval.py +++ b/apps/api/src/alicebot_api/vnext_retrieval.py @@ -106,6 +106,7 @@ from alicebot_api.vnext_promotion_policy import memory_write_provenance from alicebot_api.vnext_project_scope import ( is_global_scope, + project_floor_shape, project_scope_identity, project_scopes_overlap, resolve_project_scope, @@ -946,16 +947,31 @@ def _row_scope_event_time(row: Mapping[str, object]) -> datetime | None: return parse_event_datetime(row.get("captured_at")) -def _project_scope_meets(row_scope: set[str], requested: Collection[str]) -> bool: +def _row_floor(row: Mapping[str, object]) -> tuple[str, ...]: + shape, floor = project_floor_shape(row) + return floor if shape == "list" else () + + +def _project_scope_meets( + row_scope: set[str], + requested: Collection[str], + *, + floor: Collection[str] = (), +) -> bool: """Does a row's resolved project scope meet the requested tuple? The tuple may hold the reserved global marker (spec 6.1): it asks for a row whose scope holds no Alice project id. The one predicate in ``vnext_project_scope`` decides, so a request without the marker keeps the - plain intersection it always had. + plain intersection it always had. On that global branch the row's floor + must also sit inside the view. """ - return project_scopes_overlap(tuple(sorted(row_scope)), tuple(sorted(requested))) + return project_scopes_overlap( + tuple(sorted(row_scope)), + tuple(sorted(requested)), + floor=tuple(floor), + ) def _is_held_back_global( @@ -984,7 +1000,11 @@ def _row_matches_scope( if source_scope_envelope else _row_project_scope_values(row) ) - if scope.projects and not _project_scope_meets(project_scope, scope.projects): + if scope.projects and not _project_scope_meets( + project_scope, + scope.projects, + floor=_row_floor(row), + ): return False if scope.exclude_global_domains and _is_held_back_global( row, project_scope, scope.exclude_global_domains @@ -1229,7 +1249,11 @@ def _graph_memory_admissible( if memory_types and row.get("memory_type") not in memory_types: return False if projects: - if not _project_scope_meets(_row_project_scope_values(row), projects): + if not _project_scope_meets( + _row_project_scope_values(row), + projects, + floor=_row_floor(row), + ): return False if created_by_agent_ids and row.get("created_by_agent_id") not in created_by_agent_ids: return False diff --git a/apps/api/src/alicebot_api/vnext_stores/sqlite/graph_open_loops.py b/apps/api/src/alicebot_api/vnext_stores/sqlite/graph_open_loops.py index cc43135e1..4857bcbce 100644 --- a/apps/api/src/alicebot_api/vnext_stores/sqlite/graph_open_loops.py +++ b/apps/api/src/alicebot_api/vnext_stores/sqlite/graph_open_loops.py @@ -828,6 +828,7 @@ def list_open_loops_view_partitions( text_expressions=("metadata_json", "project_id"), domain_expression="domain", global_excluded_domains=tuple(sorted(exclude_global_domains)), + floor_expression="alice_project_floor_identity(metadata_json)", ) columns = ", ".join(f"l.{column}" for column in OPEN_LOOP_COLUMNS) rows = self._fetch_all( diff --git a/apps/api/src/alicebot_api/vnext_stores/sqlite/memory_access.py b/apps/api/src/alicebot_api/vnext_stores/sqlite/memory_access.py index 4b677e515..2a019adae 100644 --- a/apps/api/src/alicebot_api/vnext_stores/sqlite/memory_access.py +++ b/apps/api/src/alicebot_api/vnext_stores/sqlite/memory_access.py @@ -617,6 +617,7 @@ def list_memories_view_partitions( text_expressions=("metadata_json", "project_id"), domain_expression="domain", global_excluded_domains=tuple(sorted(exclude_global_domains)), + floor_expression="alice_project_floor_identity(metadata_json)", ) ordering = ("created_at",) if order_by_created_at else ("updated_at", "created_at") order_columns = ", ".join(ordering) diff --git a/apps/api/src/alicebot_api/vnext_stores/sqlite/query_predicates.py b/apps/api/src/alicebot_api/vnext_stores/sqlite/query_predicates.py index 5beed618a..f87c89bf1 100644 --- a/apps/api/src/alicebot_api/vnext_stores/sqlite/query_predicates.py +++ b/apps/api/src/alicebot_api/vnext_stores/sqlite/query_predicates.py @@ -219,6 +219,7 @@ def _view_membership_sql( global_excluded_domains: tuple[str, ...], text_expressions: tuple[str, ...], partition: bool, + floor_expression: str = "'[]'", ) -> tuple[str, list[object]]: """The exact view test as one aggregate over one identity-function call. @@ -247,6 +248,18 @@ def excluded_sql() -> str: ) when_global = "" if wants_global: + floor_outside = "" + if ids: + params.extend(ids) + floor_outside = ( + " AND CAST(floor_id.value AS TEXT) NOT IN (" + f"{placeholders(list(ids))})" + ) + floor_clear = ( + "NOT EXISTS (SELECT 1 FROM json_each(" + f"{floor_expression}) AS floor_id WHERE " + f"{_sql_has_alice_id('CAST(floor_id.value AS TEXT)')}{floor_outside})" + ) if partition: if excluded: inner = f"CASE WHEN {excluded_sql()} THEN NULL ELSE 0 END" @@ -258,6 +271,7 @@ def excluded_sql() -> str: inner = "1" when_global = ( f"WHEN COALESCE(MAX({_sql_has_alice_id('CAST(scoped_project.value AS TEXT)')}), 0) = 0 " + f"AND {floor_clear} " f"THEN {inner} " ) fallback = "NULL" if partition else "0" @@ -297,6 +311,7 @@ def _project_view_sql( text_expressions: tuple[str, ...], domain_expression: str | None, global_excluded_domains: tuple[str, ...] | None, + floor_expression: str = "'[]'", ) -> tuple[str, list[object]]: """The ``AND ...`` clause for a request tuple, or ``("", [])`` when it fences nothing. @@ -340,6 +355,7 @@ def _project_view_sql( global_excluded_domains=excluded, text_expressions=text_expressions, partition=False, + floor_expression=floor_expression, ) params.extend(exact_params) return f" AND ({fast} OR {exact} = 1)", params @@ -356,6 +372,7 @@ def _project_view_sql( global_excluded_domains=(), text_expressions=text_expressions, partition=False, + floor_expression=floor_expression, ) params.extend(exact_params) return f" AND ({prefilter} AND {exact} = 1)", params @@ -369,6 +386,7 @@ def _project_view_partition_sql( text_expressions: tuple[str, ...], domain_expression: str, global_excluded_domains: tuple[str, ...], + floor_expression: str = "'[]'", ) -> tuple[str, list[object]]: """A value per row for the single-scan fill: 1 project, 0 global, NULL outside the view. @@ -395,6 +413,7 @@ def _project_view_partition_sql( global_excluded_domains=excluded, text_expressions=text_expressions, partition=True, + floor_expression=floor_expression, ) params.extend(exact_params) return f"CASE WHEN {fast} THEN 0 ELSE {exact} END", params @@ -522,6 +541,7 @@ def _project_clause( text_expressions=(f"{prefix}metadata_json", f"{prefix}project_id"), domain_expression=f"{prefix}domain", global_excluded_domains=global_excluded_domains, + floor_expression=f"alice_project_floor_identity({prefix}metadata_json)", ) @@ -665,6 +685,7 @@ def _metadata_values(keys: tuple[str, ...], values: tuple[str, ...]) -> str: text_expressions=text_expressions, domain_expression=domain_expression, global_excluded_domains=global_excluded_domains, + floor_expression=f"alice_project_floor_identity({metadata_expression})", ) clauses.append(project_sql) params.extend(project_params) diff --git a/tests/unit/test_store_graph_open_loops_split.py b/tests/unit/test_store_graph_open_loops_split.py index eae9d1a4d..36574141b 100644 --- a/tests/unit/test_store_graph_open_loops_split.py +++ b/tests/unit/test_store_graph_open_loops_split.py @@ -110,7 +110,8 @@ # (an empty ceiling returns no rows without a query). The Postgres reader takes the same two arguments so that # the shared unscoped call site can state ``None`` for both, and it refuses anything else, since the Postgres # runtime resolves no project view (reviewed change, not drift). - POSTGRES_CARRIER_PATH: "e4724ba1ec3b8917c5be74b619259ddf1c282825b8e938ce4fa9947491a90a6f", + # The file hash now matches the carrier after the label lock. Previous receipt e4724ba1... + POSTGRES_CARRIER_PATH: "87aeac394e698a0b5709fd168abfa2a8a86af674ad42370f5a40decd773554df", # The SQLite carrier is re-minted, with its method AST manifest below, for # ``list_open_loops`` and ``list_open_loop_events``: they bind a query through # ``literal_match_operand`` and so refuse one past the LIKE operand limit. @@ -128,13 +129,17 @@ # delete preview and the scrub count and blank the same loops. The rule text moved unchanged to # ``vnext_stores/sqlite/open_loop_source_reference.py``; only the reader function and the receipt of the file # change (reviewed change, not drift). - SQLITE_CARRIER_PATH: "9a2634bef621d32262b845c046820d8b19c64801ec9f9b462e978f364f16f643", + # Re-minted so the open-loop partition read passes the floor identity. + # Previous receipt 9a2634be... + SQLITE_CARRIER_PATH: "c05ac13285a25bd59a2f22d12a7f56e2ac81f063af0949aa7adc82e43f643454", POSTGRES_COLUMNS_PATH: "5b0d972a55abf8590ce14394a37fd71b9b88ba7ab3de82d61efc1bddfc022b71", SQLITE_COLUMNS_PATH: "be81b8628d0831d3d02b280b5455fb02333db5740ebef8d85d58024384ae6556", } EXPECTED_METHOD_AST_MANIFESTS = { - POSTGRES_CARRIER_PATH: "2558088459f1b9a565e1b366ffe0b7c4025c623a9e2ea78007d06a46793ce1b8", - SQLITE_CARRIER_PATH: "2850ba6057b1510759613aaa3798a226808a42470ee11cfb9c6e3afbf3e98e66", + # Postgres manifest matches the carrier after the label lock. Previous 25580884... + POSTGRES_CARRIER_PATH: "48064a91179463a20147a8e02442f3259976752000d9aafcb51647851227c46c", + # SQLite manifest includes the floor identity on the partition read. Previous 2850ba60... + SQLITE_CARRIER_PATH: "54112e01f88e048b63731252d3fc0e34918db8575f70ab6b54b0a551699d1483", } EXPECTED_METADATA_MANIFESTS = { POSTGRES_CARRIER_PATH: "6edb6a10e7a37dbbbbde97e5550422718a0112257666de8e23d49c60490fa13f", @@ -152,7 +157,8 @@ # Per-file importer savepoint (2026-10-02): one paired method more, ``savepoint``, appended last. # Previous receipt: (171, 526374782104a2a1...). Proof: the member list equals the list at origin/main # 040a2a10 with ``savepoint`` added at the end and nothing else moved (reviewed change, not drift). - "PostgresVNextStore": (172, "6f1a459fcf4319cf4281f6cc0d4e81679c3e05d851fd0a874a2d90298d7c2569"), + # lock_label_writes and read_label_rows follow __init__. Previous receipt (172, 6f1a459f...). + "PostgresVNextStore": (174, "095250b8a77d6c0a32d2783343916b947bcae8ae86e3a1693e47c2fb11802211"), # One SQLite-only method more, ``check_source_search_query``: the Postgres # source search has no expression-depth or LIKE-length limit to check. # Merge of #500 and #502 (2026-10-01): one more SQLite-only method, @@ -167,7 +173,8 @@ # 040a2a10 with ``savepoint`` added at the end and nothing else moved (reviewed change, not drift). # Proof: the replacement branch gains only scrub_source, source_inventory and # prunable_sources here; every pre-existing class member keeps its order. - "SQLiteVNextStore": (134, "1301272026897057cf071009cc21787543ddc326f1e06f1a75a763f3e344767e"), + # lock_label_writes and read_label_rows follow __init__. Previous receipt (134, 13012720...). + "SQLiteVNextStore": (136, "1d99dbaf69e4ea888ca7beb2389ac176650a2e573c067bf0686adc9e609132f5"), } EXPECTED_COLUMN_AST = { POSTGRES_COLUMNS_PATH: { diff --git a/tests/unit/test_store_memory_access_split.py b/tests/unit/test_store_memory_access_split.py index f04ef0ec9..2315cb2ac 100644 --- a/tests/unit/test_store_memory_access_split.py +++ b/tests/unit/test_store_memory_access_split.py @@ -51,8 +51,10 @@ # global domains it leaves out raises (reviewed change, not drift). # Re-minted for alice_project_floor_identity, the fourth identity function, # used by the roll-up lookups. Previous receipt eab46f16... + # Re-minted so a global view also requires every Alice id in the floor. + # Previous receipt 8beae59c... "apps/api/src/alicebot_api/vnext_stores/sqlite/query_predicates.py": ( - "8beae59c379c56033388a35b5a152f1f683f6b55c6e2fbd62925723566a12202" + "670fd096b461c72517f3791c1f6216778f26d68838307c11d07ffcf5e7b38e79" ), # Re-minted for the Phase 4 Stage 2 resident vector cache (reviewed # carrier change; the receipt guards unreviewed drift): the vector scan @@ -84,8 +86,10 @@ # takes a keyword-only ``include_deleted`` (false by default) and drops the ``deleted_at IS NULL`` clause only when # it is true. Previous SQLite receipt 91636de9... # Re-minted so the two roll-up lookups overlap scope or floor. Previous receipt 64f21989... + # Re-minted so the memory partition read passes the floor identity. + # Previous receipt 1580dca3... "apps/api/src/alicebot_api/vnext_stores/sqlite/memory_access.py": ( - "1580dca3a31fbbcf98539e71e81bad13935f30c59e8a7c75b0fc0b4472a6d5fa" + "58e22342c6aaf4ed73aa78e5f2ee85af1d8deeb7cbad6a3abe06f78fb8f9a0ac" ), } diff --git a/tests/unit/test_view_floor.py b/tests/unit/test_view_floor.py new file mode 100644 index 000000000..fa93d315f --- /dev/null +++ b/tests/unit/test_view_floor.py @@ -0,0 +1,75 @@ +"""A project view that asks for global rows also consults the floor.""" + +from __future__ import annotations + +import sqlite3 +from uuid import uuid4 + +from alicebot_api.mcp.retrieval_shared import _resource_matches_project_scope +from alicebot_api.session_briefing import _memory_honours_fence +from alicebot_api.sqlite_schema import bootstrap_sqlite_schema +from alicebot_api.sqlite_store import SQLiteVNextStore, ensure_sqlite_user +from alicebot_api.vnext_project_scope import GLOBAL_PROJECT_MARKER +from alicebot_api.vnext_retrieval import _project_scope_meets + +ALPHA = "prj_" + "a" * 16 +BETA = "prj_" + "b" * 16 + + +def test_python_view_mirrors_hide_a_foreign_floor() -> None: + view = (ALPHA, GLOBAL_PROJECT_MARKER) + assert _project_scope_meets(set(), view, floor=(ALPHA,)) is True + assert _project_scope_meets(set(), view, floor=(BETA,)) is False + assert _project_scope_meets(set(), view, floor=()) is True + foreign = {"metadata_json": {"project_scope": [], "project_floor": [BETA]}} + home = {"metadata_json": {"project_scope": [], "project_floor": [ALPHA]}} + assert _resource_matches_project_scope(foreign, view) is False + assert _resource_matches_project_scope(home, view) is True + assert _memory_honours_fence( + {**foreign, "domain": "project", "sensitivity": "internal"}, + effective_domains=(), + effective_sensitivity_allowed=("internal",), + effective_project_scope=view, + exclude_global_domains=frozenset(), + ) is False + + +def test_sqlite_global_view_hides_a_row_whose_floor_names_another_project() -> None: + conn = sqlite3.connect(":memory:") + bootstrap_sqlite_schema(conn) + user_id = str(uuid4()) + ensure_sqlite_user(conn, user_id, "view-floor@example.com", "View Floor") + store = SQLiteVNextStore(conn, user_id) + + def add(name: str, scope: list[str], floor: list[str]) -> str: + row = store.create_memory( + { + "memory_key": f"memory.{name}", + "value": {"text": name}, + "status": "active", + "memory_type": "semantic", + "title": name, + "canonical_text": name, + "summary": name, + "domain": "project", + "sensitivity": "internal", + "metadata_json": {"project_scope": scope, "project_floor": floor}, + } + ) + return str(row["id"]) + + home = add("home", [], [ALPHA]) + foreign = add("foreign", [], [BETA]) + plain = add("plain", [], []) + listed = { + str(row["id"]) + for row in store.list_memories( + projects=(ALPHA, GLOBAL_PROJECT_MARKER), + exclude_global_domains=(), + limit=20, + ) + } + assert home in listed + assert plain in listed + assert foreign not in listed + conn.close() From 4b6087588a91a192c0019ed440ea7aba986d8035 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 18:35:56 +0000 Subject: [PATCH 08/90] Settle labels at the remaining exact doors. Explain, memory review, redaction, open-loop update, and the legacy artifact authorizer use the input labels. A registry names each exact door and the readers that are not doors yet. --- CHANGELOG.md | 2 +- .../alicebot_api/mcp/evidence_artifacts.py | 29 ++- apps/api/src/alicebot_api/mcp/memories.py | 18 +- apps/api/src/alicebot_api/mcp/policy.py | 2 + apps/api/src/alicebot_api/mcp/retrieval.py | 5 +- apps/api/src/alicebot_api/mcp/review.py | 41 ++++- .../src/alicebot_api/routers/_vnext_shared.py | 2 + .../alicebot_api/routers/vnext_memories.py | 24 ++- .../alicebot_api/routers/vnext_projects.py | 5 +- .../api/src/alicebot_api/vnext_label_guard.py | 15 ++ .../src/alicebot_api/vnext_memory_commit.py | 22 ++- tests/unit/test_label_door_registry.py | 167 ++++++++++++++++++ tests/unit/test_label_guard_exact_doors.py | 77 ++++++++ 13 files changed, 377 insertions(+), 32 deletions(-) create mode 100644 tests/unit/test_label_door_registry.py diff --git a/CHANGELOG.md b/CHANGELOG.md index 80ac80757..1565bb272 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,7 +2,7 @@ ## Unreleased -- Unreleased (on main, not in v0.20.0): an exact read of a derived artifact, and a write that cites a derived memory, uses the labels of that row's inputs. v0.20.0 used the label stored on the row, so a public copy of a confidential input could be read. A row whose inputs cannot be checked is refused to a locked key. The owner and an unbound admin still read the stored row. No migration is required. +- Unreleased (on main, not in v0.20.0): an exact read of a derived artifact, a write that cites a derived memory, explain, memory review, redaction, and an open-loop update use the labels of that row's inputs. v0.20.0 used the label stored on the row, so a public copy of a confidential input could be read. A row whose inputs cannot be checked is refused to a locked key. The owner and an unbound admin still read the stored row. No migration is required. - Unreleased (on main, not in v0.20.0): a locked key's consolidation, roll-up, staleness sweep and open-loop review keep only inputs whose scope and floor are both inside the binding. The overlap check stays, and the exact test runs after it. Consolidation and roll-ups compare the group scope (scope united with floor), so a card stored with an empty scope and a floor of two projects is still found and can be accepted. The operator artifact list, given a project, matches that floor as well as the scope. Each new report and aggregate records `derived_from` for the rows it used. v0.20.0 selected by overlap of the scope alone and wrote no `derived_from`. No migration is required. - Unreleased (on main, not in v0.20.0): a memory copied from another row is stored at least as strict as that row. v0.20.0 kept the label the copy was given at insert, so a later reader could see a health or confidential input through a copy stored as `unknown` and `public`. The insert now re-reads the inputs inside the write and raises the domain, the sensitivity, and the project floor. A later metadata write cannot drop the marker that makes the row derived, or put back an older project scope or project floor. Raising a source or a memory walks the rows derived from it and raises those too. A candidate open loop and a generated artifact take the same insert floor. Replacing a SQLite source with a stricter one raises the memories that cite it before they are retired. Moving a source to another project answers with how many derived rows a project-bound key would lose, and waits for confirm_label_hide before it writes. A relabel that cannot finish answers 409, and one that waits more than 3 seconds for the label lock answers 503. No migration is required. - Unreleased (on main, not in v0.20.0): the label of a derived row (its domain, its sensitivity, and the projects it requires) is now decided by one pure function. v0.20.0 kept the label a derived row was given when it was made. The function is what the write path calls. A read that does not go through that path still behaves as it does in v0.20.0. A locked key is refused when a caller passes a project floor that is not inside the key's binding. No migration is required. diff --git a/apps/api/src/alicebot_api/mcp/evidence_artifacts.py b/apps/api/src/alicebot_api/mcp/evidence_artifacts.py index 0a384563a..1607cb387 100644 --- a/apps/api/src/alicebot_api/mcp/evidence_artifacts.py +++ b/apps/api/src/alicebot_api/mcp/evidence_artifacts.py @@ -275,18 +275,30 @@ def _authorize_explain_resource( ) -> None: """Require an unfiltered policy decision for one expanded resource.""" + from alicebot_api.vnext_label_guard import apply_unverified_rule, effective_row_for_fence, policy_labels + + judged: Mapping[str, object] = resource + judged_scope = project_scope + judged_floor: tuple[str, ...] = () + if target_type in {"memory", "source", "artifact"}: + judged = effective_row_for_fence(store, identity, target_type, resource) + _domains, _sensitivity, judged_scope, judged_floor = policy_labels(judged) + if target_type == "source": + judged_scope = project_scope _actor_type, _actor_id, decision = _policy_checked( store, # type: ignore[arg-type] identity=identity, action=EXPLAIN_DISCLOSURE_ACTION, - domains=(str(resource.get("domain") or "unknown"),), - sensitivity_allowed=(str(resource.get("sensitivity") or "unknown"),), - project_scope=project_scope, + domains=(str(judged.get("domain") or "unknown"),), + sensitivity_allowed=(str(judged.get("sensitivity") or "unknown"),), + project_scope=judged_scope, + project_floor=judged_floor, require_explicit_project_scope=True, target_type=target_type, target_id=target_id, project_view=ProjectView.unscoped(), ) + decision = apply_unverified_rule(decision, judged, identity) # ``allowed_with_filtering`` is not sufficient for an explain response: # the downstream services expand related rows and do not accept filters. if decision.decision != "allowed": @@ -740,20 +752,25 @@ def _authorize_vnext_artifact_target( artifact = store.get_artifact_for_update(artifact_id) if for_update else store.get_artifact(artifact_id) if artifact is None: raise MCPReferenceNotFoundError(f"artifact {artifact_id} was not found") + from alicebot_api.vnext_label_guard import apply_unverified_rule, effective_row_for_fence, policy_labels + judged = effective_row_for_fence(store, identity, "artifact", artifact) + domains, sensitivity_allowed, project_scope, project_floor = policy_labels(judged) actor_type, actor_id, raw_decision = _policy_checked( store, identity=identity, action=action, - domains=(str(artifact.get("domain") or "unknown"),), - sensitivity_allowed=(str(artifact.get("sensitivity") or "unknown"),), - project_scope=resource_project_scope(artifact), + domains=domains, + sensitivity_allowed=sensitivity_allowed, + project_scope=project_scope, + project_floor=project_floor, require_explicit_project_scope=True, require_unfiltered_target=True, target_type="artifact", target_id=artifact_id, project_view=ProjectView.unscoped(), ) + raw_decision = apply_unverified_rule(raw_decision, judged, identity) return artifact, actor_type, actor_id, raw_decision diff --git a/apps/api/src/alicebot_api/mcp/memories.py b/apps/api/src/alicebot_api/mcp/memories.py index 6165fe47c..0beb74c31 100644 --- a/apps/api/src/alicebot_api/mcp/memories.py +++ b/apps/api/src/alicebot_api/mcp/memories.py @@ -427,8 +427,11 @@ def redact_memory_flow( # before it is raised, and for a deleted row it is raised as a refusal the # surface answers "not found" (RefusedOnDeletedMemoryError): a plain not-found # error here would roll the audit row back with the call. + from alicebot_api.vnext_label_guard import effective_row_for_fence, policy_labels + + judged = effective_row_for_fence(store, identity, "memory", memory) try: - memory_service.refuse_unauthorized_write(identity=identity, action="memory.redact", memory=memory) + memory_service.refuse_unauthorized_write(identity=identity, action="memory.redact", memory=judged) except AgentPolicyBlockedError as exc: if memory.get("deleted_at") is not None: raise RefusedOnDeletedMemoryError(exc.decision) from None @@ -457,21 +460,26 @@ def redact_memory_flow( # row's redaction receipt and writes nothing, so its authorization # should not depend on a call made earlier in the function. A test # takes the pre-check away and checks the replay is still refused. + from alicebot_api.vnext_label_guard import apply_unverified_rule + + domains, sensitivity_allowed, project_scope, project_floor = policy_labels(judged) decision = evaluate_agent_policy( identity=identity, action="memory.redact", - domains=(str(memory.get("domain") or "unknown"),), - sensitivity_allowed=(str(memory.get("sensitivity") or "unknown"),), - project_scope=resource_project_scope(memory), + domains=domains, + sensitivity_allowed=sensitivity_allowed, + project_scope=project_scope, + project_floor=project_floor, require_explicit_project_scope=True, ) + decision = apply_unverified_rule(decision, judged, identity) if decision.decision == "blocked": raise AgentPolicyBlockedError(decision) else: memory_service.authorize_memory_action( identity=identity, action="memory.redact", - memory=memory, + memory=judged, ) actor_type = "agent" if identity is not None else "user" forgotten_first = False diff --git a/apps/api/src/alicebot_api/mcp/policy.py b/apps/api/src/alicebot_api/mcp/policy.py index c98977cb5..26fa530e5 100644 --- a/apps/api/src/alicebot_api/mcp/policy.py +++ b/apps/api/src/alicebot_api/mcp/policy.py @@ -137,6 +137,7 @@ def _policy_checked( domains: tuple[str, ...] = (), sensitivity_allowed: tuple[str, ...] = ("public", "internal", "private", "unknown"), project_scope: tuple[str, ...] = (), + project_floor: tuple[str, ...] = (), workflow_type: str | None = None, write_policy: str | None = None, require_explicit_project_scope: bool = False, @@ -164,6 +165,7 @@ def _policy_checked( domains=domains, sensitivity_allowed=sensitivity_allowed, project_scope=project_scope, + project_floor=project_floor, workflow_type=workflow_type, write_policy=write_policy, require_explicit_project_scope=require_explicit_project_scope, diff --git a/apps/api/src/alicebot_api/mcp/retrieval.py b/apps/api/src/alicebot_api/mcp/retrieval.py index ab79b21b1..4e27f340f 100644 --- a/apps/api/src/alicebot_api/mcp/retrieval.py +++ b/apps/api/src/alicebot_api/mcp/retrieval.py @@ -756,13 +756,16 @@ def _handle_alice_open_loops(context: MCPRuntimeContext, arguments: Mapping[str, target = store.get_open_loop(loop_id) if target is None: raise MCPReferenceNotFoundError(f"open loop {loop_id} was not found") + from alicebot_api.vnext_label_guard import effective_row_for_fence + + judged = effective_row_for_fence(store, identity, "open_loop", target) # Same ceiling block as memory mutations. The policy event names # this loop; the previous check logged the decision with no target. try: VNextMemoryCommitService(store).authorize_memory_action( identity=identity, action="open_loop.update", - memory=target, + memory=judged, target_type="open_loop", ) except AgentPolicyBlockedError as exc: diff --git a/apps/api/src/alicebot_api/mcp/review.py b/apps/api/src/alicebot_api/mcp/review.py index 297df7896..c2862701e 100644 --- a/apps/api/src/alicebot_api/mcp/review.py +++ b/apps/api/src/alicebot_api/mcp/review.py @@ -138,9 +138,16 @@ def _vnext_memory_review(context: MCPRuntimeContext, arguments: Mapping[str, obj memory = store.get_memory(memory_id) if memory is None: raise MCPReferenceNotFoundError(f"memory {memory_id} was not found") - target_domain = str(memory.get("domain") or "unknown") - target_sensitivity = str(memory.get("sensitivity") or "unknown") - target_projects = resource_project_scope(memory) + from alicebot_api.vnext_label_guard import ( + apply_unverified_rule, + effective_row_for_fence, + policy_labels, + ) + + judged = effective_row_for_fence(store, identity, "memory", memory) + target_domains, target_sensitivity_allowed, target_projects, target_floor = policy_labels(judged) + target_domain = target_domains[0] + target_sensitivity = target_sensitivity_allowed[0] _actor_type, _actor_id, decision = _policy_checked( store, identity=identity, @@ -148,9 +155,11 @@ def _vnext_memory_review(context: MCPRuntimeContext, arguments: Mapping[str, obj domains=(target_domain,), sensitivity_allowed=(target_sensitivity,), project_scope=target_projects, + project_floor=target_floor, require_explicit_project_scope=True, project_view=ProjectView.unscoped(), ) + decision = apply_unverified_rule(decision, judged, identity) if decision.decision == "blocked": blocked_decision = decision elif ( @@ -483,16 +492,26 @@ def _vnext_memory_correct(context: MCPRuntimeContext, arguments: Mapping[str, ob target = store.get_memory(memory_id) if target is None: raise MCPReferenceNotFoundError(f"memory {memory_id} was not found") + from alicebot_api.vnext_label_guard import ( + apply_unverified_rule, + effective_row_for_fence, + policy_labels, + ) + + judged = effective_row_for_fence(store, identity, "memory", target) + domains, sensitivity_allowed, project_scope, project_floor = policy_labels(judged) _checked_actor_type, _checked_actor_id, decision = _policy_checked( store, identity=identity, action="memory.review", - domains=(str(target.get("domain") or "unknown"),), - sensitivity_allowed=(str(target.get("sensitivity") or "unknown"),), - project_scope=resource_project_scope(target), + domains=domains, + sensitivity_allowed=sensitivity_allowed, + project_scope=project_scope, + project_floor=project_floor, require_explicit_project_scope=True, project_view=ProjectView.unscoped(), ) + decision = apply_unverified_rule(decision, judged, identity) if decision.decision == "blocked": blocked_decision = decision elif is_pending_consolidation_candidate(target): @@ -550,16 +569,20 @@ def _vnext_memory_correct(context: MCPRuntimeContext, arguments: Mapping[str, ob # check commits a durable policy audit event; this second check closes # the gap where a target could be reassigned between authorization and # update. + locked_judged = effective_row_for_fence(store, identity, "memory", memory) + locked_domains, locked_sensitivity, locked_scope, locked_floor = policy_labels(locked_judged) _locked_actor_type, _locked_actor_id, locked_decision = _policy_checked( store, identity=identity, action="memory.review", - domains=(str(memory.get("domain") or "unknown"),), - sensitivity_allowed=(str(memory.get("sensitivity") or "unknown"),), - project_scope=resource_project_scope(memory), + domains=locked_domains, + sensitivity_allowed=locked_sensitivity, + project_scope=locked_scope, + project_floor=locked_floor, require_explicit_project_scope=True, project_view=ProjectView.unscoped(), ) + locked_decision = apply_unverified_rule(locked_decision, locked_judged, identity) if locked_decision.decision == "blocked": _raise_mcp_policy_blocked(locked_decision) # Route the retired-status guard through the central transition table so diff --git a/apps/api/src/alicebot_api/routers/_vnext_shared.py b/apps/api/src/alicebot_api/routers/_vnext_shared.py index 7b86ca670..9203ea3bc 100644 --- a/apps/api/src/alicebot_api/routers/_vnext_shared.py +++ b/apps/api/src/alicebot_api/routers/_vnext_shared.py @@ -429,6 +429,7 @@ def _vnext_policy_checked( domains: tuple[str, ...] = (), sensitivity_allowed: tuple[str, ...] = ("public", "internal", "private", "unknown"), project_scope: tuple[str, ...] = (), + project_floor: tuple[str, ...] = (), workflow_type: str | None = None, write_policy: str | None = None, target_type: str | None = None, @@ -445,6 +446,7 @@ def _vnext_policy_checked( domains=domains, sensitivity_allowed=sensitivity_allowed, project_scope=project_scope, + project_floor=project_floor, workflow_type=workflow_type, write_policy=write_policy, require_explicit_project_scope=require_explicit_project_scope, diff --git a/apps/api/src/alicebot_api/routers/vnext_memories.py b/apps/api/src/alicebot_api/routers/vnext_memories.py index 433d4716f..b120bdc2f 100644 --- a/apps/api/src/alicebot_api/routers/vnext_memories.py +++ b/apps/api/src/alicebot_api/routers/vnext_memories.py @@ -921,20 +921,29 @@ def review_vnext_memory( target = auth_store.get_memory(str(memory_id)) if target is None: return _vnext_public_error_response(status_code=404, detail="vNext memory was not found") - target_scope = resource_project_scope(target) + from alicebot_api.vnext_label_guard import ( + apply_unverified_rule, + effective_row_for_fence, + policy_labels, + ) + + judged = effective_row_for_fence(auth_store, identity, "memory", target) + domains, sensitivity_allowed, target_scope, target_floor = policy_labels(judged) if action == "assign_project" and request.project_id is not None: target_scope = tuple(dict.fromkeys((*target_scope, request.project_id))) decision = _vnext_policy_checked( store=auth_store, identity=identity, action="memory.review", - domains=(str(target.get("domain") or "unknown"),), - sensitivity_allowed=(str(target.get("sensitivity") or "unknown"),), + domains=domains, + sensitivity_allowed=sensitivity_allowed, project_scope=target_scope, + project_floor=target_floor, target_type="memory", target_id=str(memory_id), require_explicit_project_scope=True, ) + decision = apply_unverified_rule(decision, judged, identity) if decision.decision == "blocked": return _vnext_permission_response(decision) except AgentIdentityValidationError as exc: @@ -1061,20 +1070,23 @@ def review_vnext_memory( # Re-authorize the locked record so a concurrent reassignment cannot # move it outside the bound agent project between the first check and # this mutation. - locked_scope = resource_project_scope(existing) + locked_judged = effective_row_for_fence(store, identity, "memory", existing) + _locked_domains, _locked_sensitivity, locked_scope, locked_floor = policy_labels(locked_judged) if action == "assign_project" and request.project_id is not None: locked_scope = tuple(dict.fromkeys((*locked_scope, request.project_id))) locked_decision = _vnext_policy_checked( store=store, identity=identity, action="memory.review", - domains=(str(existing.get("domain") or "unknown"),), - sensitivity_allowed=(str(existing.get("sensitivity") or "unknown"),), + domains=_locked_domains, + sensitivity_allowed=_locked_sensitivity, project_scope=locked_scope, + project_floor=locked_floor, target_type="memory", target_id=str(memory_id), require_explicit_project_scope=True, ) + locked_decision = apply_unverified_rule(locked_decision, locked_judged, identity) if locked_decision.decision == "blocked": return _vnext_permission_response(locked_decision) if str(existing.get("status") or "") in {"archived", "rejected", "superseded"}: diff --git a/apps/api/src/alicebot_api/routers/vnext_projects.py b/apps/api/src/alicebot_api/routers/vnext_projects.py index 2fb0afb18..f6ce90d93 100644 --- a/apps/api/src/alicebot_api/routers/vnext_projects.py +++ b/apps/api/src/alicebot_api/routers/vnext_projects.py @@ -605,13 +605,16 @@ def review_vnext_open_loop( target = store.get_open_loop(loop_id) if target is None: return _vnext_public_error_response(status_code=404, detail="vNext open loop was not found") + from alicebot_api.vnext_label_guard import effective_row_for_fence + + judged = effective_row_for_fence(store, identity, "open_loop", target) # Same ceiling as the MCP open-loop updates. Returning the 403 # from inside the connection keeps the policy event committed. try: VNextMemoryCommitService(store).authorize_memory_action( identity=identity, action="open_loop.update", - memory=target, + memory=judged, target_type="open_loop", ) except AgentPolicyBlockedError as exc: diff --git a/apps/api/src/alicebot_api/vnext_label_guard.py b/apps/api/src/alicebot_api/vnext_label_guard.py index f49b3763f..cdb669ed1 100644 --- a/apps/api/src/alicebot_api/vnext_label_guard.py +++ b/apps/api/src/alicebot_api/vnext_label_guard.py @@ -192,6 +192,21 @@ def _collected(self, kind: str, row: Mapping[str, object]) -> list[dict[str, obj return list(self._nodes.values()) +def effective_row_for_fence( + store: Any, + identity: AgentIdentity | None, + kind: str, + row: Mapping[str, object], +) -> Mapping[str, object]: + """The row an exact door should hand to the policy engine.""" + + from alicebot_api.vnext_source_fence import SourceReadFence + + guard = LabelGuard.for_fence(store, SourceReadFence.for_identity(identity)) + settled = guard.effective_row(kind, row) + return settled if isinstance(settled, Mapping) else row + + def policy_labels( row: Mapping[str, object], ) -> tuple[tuple[str, ...], tuple[str, ...], tuple[str, ...], tuple[str, ...]]: diff --git a/apps/api/src/alicebot_api/vnext_memory_commit.py b/apps/api/src/alicebot_api/vnext_memory_commit.py index 605f9da22..0acb70b9f 100644 --- a/apps/api/src/alicebot_api/vnext_memory_commit.py +++ b/apps/api/src/alicebot_api/vnext_memory_commit.py @@ -2506,6 +2506,7 @@ def _write_policy_decision( action: str, memory: Mapping[str, object], allow_above_ceiling: bool = False, + kind: str = "memory", ) -> PolicyDecision: """Decide one mutation of a stored target without writing anything. @@ -2520,14 +2521,24 @@ def _write_policy_decision( # memory.expire / memory.unexpire / memory.accept_consolidation are # in the agent-control WRITE_ACTIONS vocabulary, so # evaluate_agent_policy carries the read-only write block itself. + from alicebot_api.vnext_label_guard import ( + apply_unverified_rule, + effective_row_for_fence, + policy_labels, + ) + + settled = effective_row_for_fence(self.store, identity, kind, memory) + domains, sensitivity_allowed, project_scope, project_floor = policy_labels(settled) decision = evaluate_agent_policy( identity=identity, action=action, - domains=(str(memory.get("domain") or "unknown"),), - sensitivity_allowed=(str(memory.get("sensitivity") or "unknown"),), - project_scope=resource_project_scope(memory), + domains=domains, + sensitivity_allowed=sensitivity_allowed, + project_scope=project_scope, + project_floor=project_floor, require_explicit_project_scope=True, ) + decision = apply_unverified_rule(decision, settled, identity) if not allow_above_ceiling: decision = _block_mutation_above_sensitivity_ceiling(decision) if ( @@ -2594,6 +2605,7 @@ def _policy_checked_write( action=action, memory=memory, allow_above_ceiling=allow_above_ceiling, + kind="open_loop" if target_type == "open_loop" else "memory", ) return self._record_write_decision( identity=identity, @@ -2644,6 +2656,10 @@ def authorize_memory_action( ) -> PolicyDecision: """Authorize a persisted target for a cross-surface lifecycle adapter.""" + from alicebot_api.vnext_label_guard import effective_row_for_fence + + kind = "open_loop" if target_type == "open_loop" else "memory" + memory = effective_row_for_fence(self.store, identity, kind, memory) return self._policy_checked_write( identity=identity, action=action, diff --git a/tests/unit/test_label_door_registry.py b/tests/unit/test_label_door_registry.py new file mode 100644 index 000000000..e1e86cb6d --- /dev/null +++ b/tests/unit/test_label_door_registry.py @@ -0,0 +1,167 @@ +"""Every exact reader is classified, and each exact door calls the guard.""" + +from __future__ import annotations + +import ast +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[2] +SRC = ROOT / "apps/api/src" + +READS = { + "get_memory", + "get_memory_for_update", + "get_memories_by_ids", + "list_memories", + "list_memories_by_statuses", + "search_memories", + "search_memories_fts", + "search_memories_vector", + "search_memories_by_time", + "list_open_loops", + "get_open_loop", + "list_artifacts", + "get_artifact", + "get_artifact_for_update", + "list_artifacts_referencing_source", + "list_memories_referencing_source", + "list_memories_referencing_sources", + "list_beliefs", + "get_belief", + "list_projects", + "get_project", + "get_project_for_update", +} + +GUARD_CALLS = { + "LabelGuard", + "effective_row", + "effective_row_for_fence", + "admit_rows", + "admit_beliefs", + "policy_labels", +} + +# function -> helper that holds the guard call, or None when the function calls it +DOORS = { + "routers/_vnext_shared.py:_vnext_authorized_artifact": None, + "vnext_source_fence.py:resolve_attachable_memory_id": None, + "mcp/evidence_artifacts.py:_authorize_explain_resource": None, + "mcp/evidence_artifacts.py:_authorize_entity_explain_target": "_authorize_explain_resource", + "mcp/evidence_artifacts.py:_entity_backing_is_fully_authorized": "_authorize_explain_resource", + "mcp/evidence_artifacts.py:_handle_alice_vnext_memory_audit": "_authorize_explain_resource", + "mcp/evidence_artifacts.py:_authorize_vnext_artifact_target": None, + "mcp/review.py:_vnext_memory_review": None, + "mcp/review.py:_vnext_memory_correct": None, + "routers/vnext_memories.py:review_vnext_memory": None, + "mcp/memories.py:redact_memory_flow": None, + "routers/vnext_projects.py:review_vnext_open_loop": None, + "mcp/retrieval.py:_handle_alice_open_loops": None, + "vnext_memory_commit.py:VNextMemoryCommitService.authorize_memory_action": None, + "vnext_memory_commit.py:VNextMemoryCommitService._write_policy_decision": None, +} + +NOT_A_DOOR = { + "routers/_vnext_shared.py:_vnext_load_source_trace": "operator trace stays label-agnostic until the screen change", + "mcp/evidence_artifacts.py:_handle_alice_vnext_review_items": "legacy review list has no policy check", + "routers/vnext_projects.py:list_vnext_projects": "operator project list stays label-agnostic until the screen change", + "vnext_projects.py:VNextProjectService.generate_project_update_candidate": "producer input filter is the list-door change", + "vnext_projects.py:VNextProjectService.review_project_update": "write path; the route authorizes before this mutation", + "vnext_projects.py:VNextProjectService.review_open_loop": "write path; the route and the open-loop tool settle the loop first", + "vnext_projects.py:VNextProjectService.project_dashboard": "operator dashboard lists are the list-door change", + "vnext_projects.py:VNextProjectService._resolve_project": "operator project lookup is the list-door change", + "vnext_memory_commit.py:VNextMemoryCommitService.confirm": "write path; _write_policy_decision settles the row", + "vnext_memory_commit.py:VNextMemoryCommitService.undo": "write path; _write_policy_decision settles the row", + "vnext_memory_commit.py:VNextMemoryCommitService.correct": "write path; _write_policy_decision settles the row", + "vnext_memory_commit.py:VNextMemoryCommitService.forget": "write path; _write_policy_decision settles the row", + "vnext_memory_commit.py:VNextMemoryCommitService.accept_consolidation_candidate": "write path; _write_policy_decision settles the row", + "vnext_memory_commit.py:VNextMemoryCommitService.expire": "write path; _write_policy_decision settles the row", + "vnext_memory_commit.py:VNextMemoryCommitService.unexpire": "write path; _write_policy_decision settles the row", + "vnext_memory_commit.py:VNextMemoryCommitService.quarantine_by_agent_key": "write path; _write_policy_decision settles the row", + "vnext_memory_commit.py:VNextMemoryCommitService.recent_commits": "owner list; the list-door change admits rows", + "vnext_memory_commit.py:VNextMemoryCommitService.audit": "the authorize_memory callback settles each memory", + "vnext_memory_commit.py:VNextMemoryCommitService._supersession_chain": "audit walks this after authorize_memory", + "vnext_memory_commit.py:VNextMemoryCommitService.inline_confirmations": "owner list; the list-door change admits rows", + "vnext_memory_commit.py:VNextMemoryCommitService._idempotent_memory": "write path replay of a row already authorized", + "vnext_memory_commit.py:VNextMemoryCommitService._memory_by_confirmation_id": "write path replay of a row already authorized", + "vnext_memory_commit.py:VNextMemoryCommitService._latest_agentic_commit": "owner list; the list-door change admits rows", + "vnext_queue.py:VNextQueueService.review_artifact": "the HTTP review route authorizes through _vnext_authorized_artifact first", + "vnext_queue.py:VNextQueueService._promote_artifact": "the HTTP review route authorizes through _vnext_authorized_artifact first", + "vnext_queue.py:VNextQueueService.export_artifact_markdown": "the HTTP export route authorizes through _vnext_authorized_artifact first", + "mcp/retrieval.py:_resume_event_honours_policy_fence": "list door; admit_rows lands with the list-door change", + "mcp/retrieval.py:_vnext_recent_decisions": "list door; admit_rows lands with the list-door change", + "mcp/retrieval.py:_vnext_resume": "list door; admit_rows lands with the list-door change", +} + +SCAN_MODULES = ( + "alicebot_api/routers/_vnext_shared.py", + "alicebot_api/mcp/evidence_artifacts.py", + "alicebot_api/mcp/review.py", + "alicebot_api/mcp/memories.py", + "alicebot_api/routers/vnext_memories.py", + "alicebot_api/routers/vnext_projects.py", + "alicebot_api/vnext_projects.py", + "alicebot_api/vnext_memory_commit.py", + "alicebot_api/vnext_source_fence.py", + "alicebot_api/vnext_open_loop_references.py", + "alicebot_api/vnext_queue.py", + "alicebot_api/mcp/retrieval.py", +) + + +def _functions(tree: ast.AST) -> list[tuple[str, ast.FunctionDef]]: + found: list[tuple[str, ast.FunctionDef]] = [] + for node in tree.body if isinstance(tree, ast.Module) else []: + if isinstance(node, ast.FunctionDef): + found.append((node.name, node)) + elif isinstance(node, ast.ClassDef): + for child in node.body: + if isinstance(child, ast.FunctionDef): + found.append((f"{node.name}.{child.name}", child)) + return found + + +def _called_names(node: ast.AST) -> set[str]: + names: set[str] = set() + for child in ast.walk(node): + if isinstance(child, ast.Call): + func = child.func + if isinstance(func, ast.Attribute): + names.add(func.attr) + elif isinstance(func, ast.Name): + names.add(func.id) + return names + + +def _has_guard(node: ast.AST) -> bool: + return bool(_called_names(node) & GUARD_CALLS) + + +def test_every_exact_door_calls_the_guard() -> None: + modules: dict[str, ast.Module] = {} + for key, helper in DOORS.items(): + path, name = key.split(":", 1) + if path not in modules: + modules[path] = ast.parse((SRC / "alicebot_api" / path).read_text(encoding="utf-8")) + functions = dict(_functions(modules[path])) + assert name in functions, key + if helper is None: + assert _has_guard(functions[name]), key + else: + assert helper in _called_names(functions[name]), key + assert helper in functions, key + assert _has_guard(functions[helper]), helper + + +def test_every_scanned_reader_is_classified() -> None: + classified = set(DOORS) | set(NOT_A_DOOR) + missing: list[str] = [] + for relative in SCAN_MODULES: + tree = ast.parse((SRC / relative).read_text(encoding="utf-8")) + short = relative.removeprefix("alicebot_api/") + for name, node in _functions(tree): + if _called_names(node) & READS: + key = f"{short}:{name}" + if key not in classified: + missing.append(key) + assert missing == [] diff --git a/tests/unit/test_label_guard_exact_doors.py b/tests/unit/test_label_guard_exact_doors.py index 3e75a7d0d..05744df20 100644 --- a/tests/unit/test_label_guard_exact_doors.py +++ b/tests/unit/test_label_guard_exact_doors.py @@ -151,6 +151,83 @@ def test_a_locked_key_is_refused_an_unverified_row() -> None: assert "derived_labels_unverified" in blocked.reasons +def test_explain_refuses_a_public_copy_of_a_confidential_source() -> None: + from alicebot_api.mcp.evidence_artifacts import ( + _ExplainAuthorizationError, + _authorize_explain_resource, + ) + + store = _LabelStore() + with pytest.raises(_ExplainAuthorizationError): + _authorize_explain_resource( + store, + identity=_trusted(), + resource=store.memory, + project_scope=(ALPHA,), + target_type="memory", + target_id=MEMORY_ID, + ) + + +def test_the_legacy_artifact_authorizer_uses_the_input_label() -> None: + from alicebot_api.mcp.evidence_artifacts import _authorize_vnext_artifact_target + + store = _LabelStore() + _artifact, _actor_type, _actor_id, decision = _authorize_vnext_artifact_target( + store, # type: ignore[arg-type] + identity=_trusted(), + artifact_id=ARTIFACT_ID, + action="artifact.lookup", + for_update=False, + ) + assert decision.decision == "blocked" + + +def test_a_memory_review_decision_uses_the_input_label() -> None: + from alicebot_api.vnext_memory_commit import VNextMemoryCommitService + + store = _LabelStore() + decision = VNextMemoryCommitService(store)._write_policy_decision( # noqa: SLF001 + identity=_trusted(), + action="memory.review", + memory=store.memory, + ) + assert decision.decision == "blocked" + + +def test_an_open_loop_update_uses_the_input_label() -> None: + from alicebot_api.vnext_memory_commit import VNextMemoryCommitService + + store = _LabelStore() + loop = { + "id": "44444444-4444-4444-4444-444444444444", + "title": "loop", + "domain": "unknown", + "sensitivity": "public", + "metadata_json": { + "discovered_by": "vnext_daily_brief", + "source_id": SOURCE_ID, + "project_scope": [ALPHA], + "derived_from": { + "v": 1, + "sources": [SOURCE_ID], + "memories": [], + "open_loops": [], + "artifacts": [], + "beliefs": [], + "counts": {"sources": 1, "memories": 0, "open_loops": 0, "artifacts": 0, "beliefs": 0}, + }, + }, + } + with pytest.raises(AgentPolicyBlockedError): + VNextMemoryCommitService(store).authorize_memory_action( + identity=_trusted(), + action="open_loop.update", + memory=loop, + target_type="open_loop", + ) + + def test_a_cited_memory_is_judged_by_its_source() -> None: store = _LabelStore() with pytest.raises(MemoryRefNotFoundError): From ab2eb45c63a0ac9a0d57fc982451db893d4c193f Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 21:06:22 +0200 Subject: [PATCH 09/90] fix: type derived label dependency maps distinctly --- apps/api/src/alicebot_api/vnext_derived_labels.py | 14 +++++++------- 1 file changed, 7 insertions(+), 7 deletions(-) diff --git a/apps/api/src/alicebot_api/vnext_derived_labels.py b/apps/api/src/alicebot_api/vnext_derived_labels.py index ed0558aa7..2d039273f 100644 --- a/apps/api/src/alicebot_api/vnext_derived_labels.py +++ b/apps/api/src/alicebot_api/vnext_derived_labels.py @@ -911,7 +911,7 @@ def _changed(before: SettledLabel, after: SettledLabel) -> bool: def _weekly_parent_deps( labels: Mapping[tuple[str, str, str], SettledLabel], - own: Mapping[tuple[str, str, str], set[tuple[str, str, str]]], + own: dict[tuple[str, str, str], set[tuple[str, str, str]]], nodes: Sequence[tuple[SettledLabel, Mapping[str, object]]], ) -> None: """Old weekly candidates take the input lists of the artifact that names them.""" @@ -1012,13 +1012,13 @@ def resolve_belief(ref: tuple[str, str, str]) -> tuple[str, str, str]: return ref resolved: dict[tuple[str, str, str], set[tuple[str, str, str]]] = {} - for key, deps in own.items(): - resolved[key] = {resolve_belief(ref) for ref in deps} + for key, own_refs in own.items(): + resolved[key] = {resolve_belief(ref) for ref in own_refs} - for key, deps in resolved.items(): + for key, resolved_refs in resolved.items(): if key in problems and problems[key] not in {"", "no_record"}: continue - for ref in deps: + for ref in resolved_refs: if ref[0] in unavailable: problems[key] = "missing_table" break @@ -1075,10 +1075,10 @@ def resolve_belief(ref: tuple[str, str, str]) -> tuple[str, str, str]: if same_stored_row: published.append(replace(settled, unverified=False, reason=None)) continue - deps = [labels[ref] for ref in sorted(resolved.get(label.key, set())) if ref in labels] + settled_deps = [labels[ref] for ref in sorted(resolved.get(label.key, set())) if ref in labels] recomputed = _apply_dependencies( settled, - deps, + settled_deps, domain_fallback=label.stored_domain, sensitivity_fallback=label.stored_sensitivity, scope_fallback=label.stored_scope, From e0b68bca813b8f221397e04f3b73f0c9621581f3 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 21:07:52 +0200 Subject: [PATCH 10/90] fix: resolve full ancestry for label insert and relabel floors --- apps/api/src/alicebot_api/sqlite_store.py | 15 +++- .../src/alicebot_api/vnext_label_closure.py | 81 +++++++++++++++++++ .../src/alicebot_api/vnext_label_writes.py | 71 +++++++--------- apps/api/src/alicebot_api/vnext_store.py | 7 ++ .../test_sqlite_derived_labels_write_path.py | 51 ++++++++++++ 5 files changed, 181 insertions(+), 44 deletions(-) create mode 100644 apps/api/src/alicebot_api/vnext_label_closure.py diff --git a/apps/api/src/alicebot_api/sqlite_store.py b/apps/api/src/alicebot_api/sqlite_store.py index 2fd996f66..4808aceb8 100644 --- a/apps/api/src/alicebot_api/sqlite_store.py +++ b/apps/api/src/alicebot_api/sqlite_store.py @@ -422,14 +422,25 @@ def read_label_rows(self, kind: str, ids: Sequence[str]) -> list[VNextRow]: extra = ", value, project_id" elif table == "open_loops": extra = ", project_id, source_id, memory_id" + from uuid import UUID + canonical = [] + for item in wanted: + try: + canonical.append(UUID(item).hex) + except (ValueError, TypeError): + pass marks = ",".join("?" for _ in wanted) + alias_sql = "" + if canonical: + alias_marks = ",".join("?" for _ in canonical) + alias_sql = f" OR replace(replace(replace(replace(lower(id),'urn:uuid:',''),'-',''),'{{',''),'}}','') IN ({alias_marks})" return self._fetch_all( f""" SELECT id, user_id, domain, sensitivity, metadata_json{extra} FROM {table} - WHERE user_id = ? AND id IN ({marks}) + WHERE user_id = ? AND (id IN ({marks}){alias_sql}) """, - (self.user_id, *wanted), + (self.user_id, *wanted, *canonical), ) # -- fetch helpers (mirror PostgresVNextStore conventions) ------------ diff --git a/apps/api/src/alicebot_api/vnext_label_closure.py b/apps/api/src/alicebot_api/vnext_label_closure.py new file mode 100644 index 000000000..fc7a59719 --- /dev/null +++ b/apps/api/src/alicebot_api/vnext_label_closure.py @@ -0,0 +1,81 @@ +"""Narrow, bounded ancestry reads shared by label writes and read guards.""" +from __future__ import annotations + +from collections import deque +from collections.abc import Mapping, Sequence +from typing import Any + +from alicebot_api.vnext_derived_labels import canon_kind, dependencies_of, identifier, is_derived + + +def collect_label_rows( + store: Any, + roots: Sequence[Mapping[str, object]], + *, + max_nodes: int, + max_hops: int | None = None, + cache: dict[tuple[str, str], list[dict[str, object]]] | None = None, + user_id: str | None = None, +) -> tuple[list[dict[str, object]], bool]: + """Return reachable stored rows and whether the independent walk budget ran out. + + Cache entries only avoid reads; every origin still traverses its ancestry. + Stored IDs remain intact while query and graph identities use the canonical parser. + """ + loaded = cache if cache is not None else {} + pending = deque((canon_kind(row.get("kind")), dict(row), 0) for row in roots) + rows: list[dict[str, object]] = [] + seen: set[tuple[str, str, str]] = set() + requested: set[tuple[str, str]] = set() + exceeded = False + reader = getattr(store, "read_label_rows", None) + while pending: + kind, row, depth = pending.popleft() + stored_key = (kind, str(row.get("user_id") or ""), str(row.get("id") or "")) + if stored_key in seen: + continue + if len(seen) >= max_nodes or (max_hops is not None and depth > max_hops): + exceeded = True + continue + seen.add(stored_key) + row["kind"] = kind + if user_id is not None: + row["user_id"] = user_id + rows.append(row) + refs = list(dependencies_of(kind, row)) if is_derived(kind, row) else [] + if kind == "belief" and row.get("memory_id"): + refs.append(("memory", identifier(row["memory_id"]))) + grouped: dict[str, list[str]] = {} + for ref_kind, ref_id in refs: + name, canonical_id = canon_kind(ref_kind), identifier(ref_id) + key = (name, canonical_id) + if key in requested: + continue + requested.add(key) + if key in loaded: + pending.extend((name, dict(found), depth + 1) for found in loaded[key]) + else: + grouped.setdefault(name, []).append(canonical_id) + if not callable(reader): + continue + for name, ids in grouped.items(): + for item in ids: + loaded[(name, item)] = [] + for found in reader(name, ids): + if not isinstance(found, Mapping): + continue + key = (name, identifier(found.get("id"))) + if key not in loaded or key[1] not in ids: + continue + copied = dict(found) + loaded[key].append(copied) + pending.append((name, copied, depth + 1)) + # UUID aliases may coexist in SQLite. Never let iteration order choose a public twin. + twins: dict[tuple[str, str, str], list[dict[str, object]]] = {} + for row in rows: + twins.setdefault((str(row["kind"]), str(row.get("user_id") or ""), identifier(row.get("id"))), []).append(row) + for aliases in twins.values(): + if len(aliases) > 1: + for row in aliases: + row["sensitivity"] = "regulated" + return rows, exceeded diff --git a/apps/api/src/alicebot_api/vnext_label_writes.py b/apps/api/src/alicebot_api/vnext_label_writes.py index 6df1e210a..f33cc6290 100644 --- a/apps/api/src/alicebot_api/vnext_label_writes.py +++ b/apps/api/src/alicebot_api/vnext_label_writes.py @@ -26,9 +26,12 @@ is_derived, labels_raised_payload, settle_labels, + stored_scope, + union_floor, ) +from alicebot_api.vnext_label_closure import collect_label_rows from alicebot_api.vnext_event_log import build_event_log_record, integrity_hash_for_event -from alicebot_api.vnext_project_scope import project_scope_identity, resolve_project_scope, source_project_scope +from alicebot_api.vnext_project_scope import project_floor_shape, project_scope_identity, resolve_project_scope, source_project_scope from alicebot_api.vnext_repositories import JsonObject LABEL_METADATA_KEYS = ("project_scope", "project_floor", "derived_from") @@ -151,38 +154,32 @@ def apply_insert_floor(store: Any, kind: str, payload: Mapping[str, object]) -> if not _in_transaction(store): raise LabelLockOrderError("the insert floor requires an open transaction") user_id = str(getattr(store, "user_id", body.get("user_id") or "")) - nodes: list[dict[str, object]] = [] - grouped: dict[str, list[str]] = {} - for dep_kind, dep_id in dependencies_of(kind, body): - grouped.setdefault(dep_kind, []).append(dep_id) - reader = getattr(store, "read_label_rows", None) - if callable(reader): - for dep_kind, ids in grouped.items(): - for row in reader(dep_kind, ids): - copied = dict(row) - copied["kind"] = dep_kind - copied.setdefault("user_id", user_id) - nodes.append(copied) - if not user_id and nodes: - user_id = str(nodes[0].get("user_id") or "") own_id = str(body.get("id") or "new-derived-row") own = dict(body) own["kind"] = kind own["id"] = own_id own["user_id"] = user_id - nodes.append(own) - settled = settle_labels(nodes).by_stored(kind, own_id, user_id=user_id or None) - if settled.unverified: - return body, None + nodes, exceeded = collect_label_rows(store, [own], max_nodes=PROPAGATION_BOUND) + if exceeded: + raise LabelPropagationTooLarge(f"label propagation stopped after {PROPAGATION_BOUND} rows") + settled = settle_labels(nodes, on_cycle="unverified").by_stored(kind, own_id, user_id=user_id or None) domain = settled.domain + if settled.unverified: + domain = generation_domain(str(body.get("domain") or "unknown"), [str(node.get("domain") or "unknown") for node in nodes]) if str(body.get("domain") or "unknown") in RESTRICTED_DOMAINS: domain = generation_domain(str(body.get("domain")), [settled.domain]) - metadata = dict(body.get("metadata_json")) if isinstance(body.get("metadata_json"), Mapping) else {} + raw_metadata = body.get("metadata_json") + metadata = dict(raw_metadata) if isinstance(raw_metadata, Mapping) else {} metadata["project_scope"] = list(settled.project_scope) metadata["project_floor"] = list(settled.project_floor) + if settled.unverified: + metadata["project_floor"] = list(union_floor(settled.project_floor, [ + *(stored_scope(str(node["kind"]), node) for node in nodes), + *(project_floor_shape(node)[1] for node in nodes), + ])) updated = dict(body) updated["domain"] = domain - updated["sensitivity"] = settled.sensitivity + updated["sensitivity"] = "regulated" if settled.unverified else settled.sensitivity updated["metadata_json"] = metadata if len(project_scope_identity(settled.project_scope)) != 1: updated["project_id"] = None @@ -205,8 +202,8 @@ def apply_insert_floor(store: Any, kind: str, payload: Mapping[str, object]) -> new={ "domain": after[0], "sensitivity": after[1], - "project_scope": list(settled.project_scope), - "project_floor": list(settled.project_floor), + "project_scope": list(after[2]), + "project_floor": list(after[3]), }, ), ) @@ -466,26 +463,14 @@ def propagate(store: Any, changed: Sequence[tuple[str, str]], *, cause: str) -> affected = walk_dependants(store, roots) if not affected: return 0 - nodes: list[dict[str, object]] = [] + roots_rows = [dict(row) for row in affected] reader = getattr(store, "read_label_rows", None) if callable(reader): for kind, row_id in changed: - for row in reader(kind, [row_id]): - copied = dict(row) - copied["kind"] = kind - nodes.append(copied) - needed: dict[str, list[str]] = {} - for row in affected: - for dep_kind, dep_id in dependencies_of(str(row.get("kind")), row): - needed.setdefault(dep_kind, []).append(dep_id) - if callable(reader): - for dep_kind, dep_ids in needed.items(): - for row in reader(dep_kind, dep_ids): - copied = dict(row) - copied["kind"] = dep_kind - nodes.append(copied) - for row in affected: - nodes.append(dict(row)) + roots_rows.extend({**dict(row), "kind": kind} for row in reader(kind, [identifier(row_id)])) + nodes, exceeded = collect_label_rows(store, roots_rows, max_nodes=PROPAGATION_BOUND) + if exceeded: + raise LabelPropagationTooLarge(f"label propagation stopped after {PROPAGATION_BOUND} rows") settled = settle_labels(nodes) written = 0 for row in affected: @@ -501,7 +486,8 @@ def propagate(store: Any, changed: Sequence[tuple[str, str]], *, cause: str) -> and project_scope_identity(previous[3]) == project_scope_identity(current[3]) ): continue - metadata = dict(row.get("metadata_json")) if isinstance(row.get("metadata_json"), Mapping) else {} + raw_metadata = row.get("metadata_json") + metadata = dict(raw_metadata) if isinstance(raw_metadata, Mapping) else {} metadata["project_scope"] = list(label.project_scope) metadata["project_floor"] = list(label.project_floor) project_id = label.project_scope[0] if len(project_scope_identity(label.project_scope)) == 1 else None @@ -570,7 +556,8 @@ def count_rows_hidden_by_scope_move(store: Any, source: Mapping[str, object], ne current["kind"] = "source" moved = dict(source) moved["kind"] = "source" - metadata = dict(source.get("metadata_json")) if isinstance(source.get("metadata_json"), Mapping) else {} + raw_metadata = source.get("metadata_json") + metadata = dict(raw_metadata) if isinstance(raw_metadata, Mapping) else {} metadata["project_scope"] = list(new_scope) moved["metadata_json"] = metadata before = settle_labels([current, *[dict(row) for row in affected]]) diff --git a/apps/api/src/alicebot_api/vnext_store.py b/apps/api/src/alicebot_api/vnext_store.py index e83c801f5..b8d7572ca 100644 --- a/apps/api/src/alicebot_api/vnext_store.py +++ b/apps/api/src/alicebot_api/vnext_store.py @@ -497,6 +497,13 @@ def read_label_rows(self, kind: str, ids: Sequence[str]) -> list[VNextRow]: }.get(kind) if table is None: return [] + if table == "beliefs": + return self._fetch_all( + """SELECT b.id, b.user_id, m.domain, m.sensitivity, b.metadata_json, b.memory_id + FROM beliefs b JOIN memories m ON m.id = b.memory_id AND m.user_id = b.user_id + WHERE b.id = ANY(%s::uuid[])""", + (wanted,), + ) extra = "" if table == "memories": extra = ", value, project_id" diff --git a/tests/unit/test_sqlite_derived_labels_write_path.py b/tests/unit/test_sqlite_derived_labels_write_path.py index d6f90f646..f11e2e3ef 100644 --- a/tests/unit/test_sqlite_derived_labels_write_path.py +++ b/tests/unit/test_sqlite_derived_labels_write_path.py @@ -230,3 +230,54 @@ def test_merge_protected_metadata_keeps_label_keys_unless_the_write_is_a_relabel relabel = merge_protected_metadata(stored, {"project_scope": [], "project_floor": [ALPHA]}, label_write=True) assert relabel["project_scope"] == [] assert relabel["consolidation"] == {"cluster_member_ids": ["m"]} + + +def test_insert_floor_survives_two_hops(tmp_path: Path): + db = tmp_path / 'labels.sqlite3' + bootstrap_database(db, user_id=USER, user_email='synthetic@example.test') + with sqlite_user_connection(db, USER) as conn: + store = SQLiteVNextStore(conn, USER) + source = store.create_source({'source_type': 'note', 'title': 'Synthetic restricted note', 'content_hash': 'review-two-hops', 'domain': 'health', 'sensitivity': 'confidential', 'metadata_json': {'project_scope': [ALPHA]}}) + copied = store.create_memory({'memory_key': 'review-copy', 'canonical_text': 'Synthetic restricted observation', 'status': 'active', 'domain': 'unknown', 'sensitivity': 'public', 'metadata_json': {'source_id': str(source['id']), 'project_scope': [ALPHA]}}) + summary = store.create_memory({'memory_key': 'review-summary', 'canonical_text': 'Summary of the synthetic restricted observation', 'status': 'active', 'domain': 'unknown', 'sensitivity': 'public', 'metadata_json': {'consolidation': {'cluster_member_ids': [str(copied['id'])]}, 'project_scope': [ALPHA]}}) + assert copied['sensitivity'] == 'confidential' + assert summary['sensitivity'] == 'confidential', f"two-hop child persisted as {summary['domain']}/{summary['sensitivity']}" + + +def test_unresolved_insert_still_succeeds_conservatively(tmp_path: Path) -> None: + with _vault(tmp_path / "missing.sqlite3") as conn: + store = SQLiteVNextStore(conn, USER) + row = store.create_memory({"memory_key": "missing", "canonical_text": "summary", "status": "active", "domain": "unknown", "sensitivity": "public", "metadata_json": {"consolidation": {"cluster_member_ids": ["missing"]}, "project_scope": [ALPHA]}}) + assert row["sensitivity"] == "regulated" + assert row["metadata_json"]["project_floor"] == [ALPHA] + + +def test_relabel_reads_the_other_branch_ancestry(tmp_path: Path) -> None: + with _vault(tmp_path / "branches.sqlite3") as conn: + store = SQLiteVNextStore(conn, USER) + copies = [] + sources = [] + for suffix in ("a", "b"): + source = store.create_source({"source_type": "note", "title": suffix, "content_hash": suffix, "domain": "unknown", "sensitivity": "public", "metadata_json": {"project_scope": [ALPHA]}}) + sources.append(source) + copies.append(store.create_memory({"memory_key": suffix, "canonical_text": suffix, "status": "active", "domain": "unknown", "sensitivity": "public", "metadata_json": {"source_id": source["id"]}})) + summary = store.create_memory({"memory_key": "summary", "canonical_text": "summary", "status": "active", "domain": "unknown", "sensitivity": "public", "metadata_json": {"consolidation": {"cluster_member_ids": [row["id"] for row in copies]}}}) + store.update_source(source_id=str(sources[0]["id"]), patch={"domain": "health", "sensitivity": "confidential"}, actor_type="user") + stored = store.get_memory(str(summary["id"])) + assert stored["domain"] == "health" + assert stored["sensitivity"] == "confidential" + + +def test_sqlite_label_lookup_preserves_uuid_aliases_and_kinds(tmp_path: Path) -> None: + from uuid import UUID + with _vault(tmp_path / "aliases.sqlite3") as conn: + store = SQLiteVNextStore(conn, USER) + source = store.create_source({"source_type": "note", "title": "alias", "content_hash": "alias", "domain": "health", "sensitivity": "confidential"}) + raw = "{" + str(source["id"]).upper() + "}" + conn.execute("UPDATE sources SET id = ? WHERE id = ?", (raw, source["id"])) + found = store.read_label_rows("source", [str(UUID(raw))]) + assert [row["id"] for row in found] == [raw] + assert store.read_label_rows("memory", [str(UUID(raw))]) == [] + copy = store.create_memory({"memory_key": "alias-copy", "canonical_text": "copy", "status": "active", "domain": "unknown", "sensitivity": "public", "metadata_json": {"source_id": str(UUID(raw))}}) + assert copy["domain"] == "health" + assert copy["sensitivity"] == "confidential" From 3a6a25d210d44d9b8937dbf2ee6f01c2a6547205 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 21:10:08 +0200 Subject: [PATCH 11/90] fix: preserve RLS tenant identity in derived insert floors --- .../src/alicebot_api/vnext_label_writes.py | 5 ++++ .../test_label_floor_ancestry_postgres.py | 27 +++++++++++++++++++ 2 files changed, 32 insertions(+) create mode 100644 tests/integration/test_label_floor_ancestry_postgres.py diff --git a/apps/api/src/alicebot_api/vnext_label_writes.py b/apps/api/src/alicebot_api/vnext_label_writes.py index f33cc6290..43a147fbe 100644 --- a/apps/api/src/alicebot_api/vnext_label_writes.py +++ b/apps/api/src/alicebot_api/vnext_label_writes.py @@ -162,6 +162,11 @@ def apply_insert_floor(store: Any, kind: str, payload: Mapping[str, object]) -> nodes, exceeded = collect_label_rows(store, [own], max_nodes=PROPAGATION_BOUND) if exceeded: raise LabelPropagationTooLarge(f"label propagation stopped after {PROPAGATION_BOUND} rows") + if not user_id: + tenants = {str(node.get("user_id")) for node in nodes if node.get("user_id")} + if len(tenants) == 1: + user_id = tenants.pop() + nodes[0]["user_id"] = user_id settled = settle_labels(nodes, on_cycle="unverified").by_stored(kind, own_id, user_id=user_id or None) domain = settled.domain if settled.unverified: diff --git a/tests/integration/test_label_floor_ancestry_postgres.py b/tests/integration/test_label_floor_ancestry_postgres.py new file mode 100644 index 000000000..29dc73cd1 --- /dev/null +++ b/tests/integration/test_label_floor_ancestry_postgres.py @@ -0,0 +1,27 @@ +"""Actual RLS-scoped PostgreSQL write-floor and belief ancestry controls.""" +from uuid import uuid4 +import pytest +from alicebot_api.db import user_connection +from alicebot_api.store import ContinuityStore +from alicebot_api.vnext_store import PostgresVNextStore + + +@pytest.mark.parametrize("domain,sensitivity", [("project", "public"), ("health", "confidential")]) +def test_pg_copy_and_summary_keep_tenant_and_ancestry(migrated_database_urls, domain, sensitivity): + user_id = uuid4() + with user_connection(migrated_database_urls["app"], user_id) as conn: + ContinuityStore(conn).create_user(user_id, f"labels-{user_id}@example.invalid", "Labels") + store = PostgresVNextStore(conn) + source = store.create_source({"source_type": "note", "title": "synthetic", "content_hash": str(uuid4()), "domain": domain, "sensitivity": sensitivity}) + copy = store.create_memory({"memory_key": "copy", "canonical_text": "copy", "status": "active", "domain": "unknown", "sensitivity": "public", "metadata_json": {"source_id": str(source["id"])}}) + summary = store.create_memory({"memory_key": "summary", "canonical_text": "summary", "status": "active", "domain": "unknown", "sensitivity": "public", "metadata_json": {"consolidation": {"cluster_member_ids": [str(copy["id"])]}}}) + assert copy["sensitivity"] == sensitivity + assert summary["sensitivity"] == sensitivity + if domain == "health": + assert summary["domain"] == domain + belief_id = uuid4() + with conn.cursor() as cur: + cur.execute("INSERT INTO beliefs(id,user_id,memory_id,claim) VALUES (%s,%s,%s,%s)", (belief_id,user_id,copy["id"],"synthetic")) + belief = store.read_label_rows("belief", [str(belief_id)])[0] + assert belief["sensitivity"] == sensitivity + assert str(belief["memory_id"]) == str(copy["id"]) From ffd8a8039e62968e2137c240ed915cdb97e532da Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 21:15:43 +0200 Subject: [PATCH 12/90] fix: classify ambiguous canonical label identities as unverified --- apps/api/src/alicebot_api/vnext_derived_labels.py | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/apps/api/src/alicebot_api/vnext_derived_labels.py b/apps/api/src/alicebot_api/vnext_derived_labels.py index 2d039273f..875db8f40 100644 --- a/apps/api/src/alicebot_api/vnext_derived_labels.py +++ b/apps/api/src/alicebot_api/vnext_derived_labels.py @@ -977,6 +977,11 @@ def settle_labels( own: dict[tuple[str, str, str], set[tuple[str, str, str]]] = {} problems: dict[tuple[str, str, str], str] = {} + stored_ids: dict[tuple[str, str, str], str] = {} + for label, _row in prepared: + previous_id = stored_ids.setdefault(label.key, label.stored_id) + if previous_id != label.stored_id: + problems[label.key] = "ambiguous_identity" for label, row in prepared: if not label.derived: continue From b52cf12ed0c67fc7c5e33c15d9109c994dffa77e Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 21:11:54 +0200 Subject: [PATCH 13/90] fix: retain narrowed metadata types through protected merge --- .../alicebot_api/vnext_stores/postgres/memory_lifecycle.py | 5 ++++- .../src/alicebot_api/vnext_stores/sqlite/memory_lifecycle.py | 5 ++++- 2 files changed, 8 insertions(+), 2 deletions(-) diff --git a/apps/api/src/alicebot_api/vnext_stores/postgres/memory_lifecycle.py b/apps/api/src/alicebot_api/vnext_stores/postgres/memory_lifecycle.py index 1cb1979fa..99be62c49 100644 --- a/apps/api/src/alicebot_api/vnext_stores/postgres/memory_lifecycle.py +++ b/apps/api/src/alicebot_api/vnext_stores/postgres/memory_lifecycle.py @@ -441,10 +441,13 @@ def update_memory( ) if current is not None: stored = current.get("metadata_json") + patch_metadata = patch["metadata_json"] + if not isinstance(patch_metadata, dict): + raise ValueError("memory metadata must be an object") patch = dict(patch) patch["metadata_json"] = merge_protected_metadata( stored if isinstance(stored, dict) else {}, - patch["metadata_json"], + patch_metadata, label_write=label_write, ) row = self._fetch_one( diff --git a/apps/api/src/alicebot_api/vnext_stores/sqlite/memory_lifecycle.py b/apps/api/src/alicebot_api/vnext_stores/sqlite/memory_lifecycle.py index ccff5ca4d..d5b89e9fc 100644 --- a/apps/api/src/alicebot_api/vnext_stores/sqlite/memory_lifecycle.py +++ b/apps/api/src/alicebot_api/vnext_stores/sqlite/memory_lifecycle.py @@ -53,10 +53,13 @@ def _with_protected_metadata(self, memory_id: str, patch: JsonObject, *, label_w if current is None: return patch stored = current.get("metadata_json") + patch_metadata = patch["metadata_json"] + if not isinstance(patch_metadata, dict): + return patch merged = dict(patch) merged["metadata_json"] = merge_protected_metadata( stored if isinstance(stored, dict) else {}, - patch["metadata_json"], + patch_metadata, label_write=label_write, ) return merged From 9bcd5c7157b51406d0a24a5252b92b239d96f412 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 21:12:46 +0200 Subject: [PATCH 14/90] test: prevent public selection among stored UUID aliases --- tests/unit/test_sqlite_derived_labels_write_path.py | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/tests/unit/test_sqlite_derived_labels_write_path.py b/tests/unit/test_sqlite_derived_labels_write_path.py index f11e2e3ef..29c705721 100644 --- a/tests/unit/test_sqlite_derived_labels_write_path.py +++ b/tests/unit/test_sqlite_derived_labels_write_path.py @@ -281,3 +281,13 @@ def test_sqlite_label_lookup_preserves_uuid_aliases_and_kinds(tmp_path: Path) -> copy = store.create_memory({"memory_key": "alias-copy", "canonical_text": "copy", "status": "active", "domain": "unknown", "sensitivity": "public", "metadata_json": {"source_id": str(UUID(raw))}}) assert copy["domain"] == "health" assert copy["sensitivity"] == "confidential" + + +def test_alias_twins_cannot_choose_a_public_source(tmp_path: Path) -> None: + with _vault(tmp_path / "twins.sqlite3") as conn: + store = SQLiteVNextStore(conn, USER) + restricted = store.create_source({"source_type": "note", "title": "restricted", "content_hash": "restricted", "domain": "health", "sensitivity": "confidential"}) + public = store.create_source({"source_type": "note", "title": "public", "content_hash": "public", "domain": "project", "sensitivity": "public"}) + conn.execute("UPDATE sources SET id = ? WHERE id = ?", (str(restricted["id"]).replace("-", "").upper(), public["id"])) + copy = store.create_memory({"memory_key": "twins-copy", "canonical_text": "copy", "status": "active", "domain": "unknown", "sensitivity": "public", "metadata_json": {"source_id": str(restricted["id"])}}) + assert copy["sensitivity"] == "regulated" From cd8be4822acae9b3d4174830088454ae26a6fa1a Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 21:12:58 +0200 Subject: [PATCH 15/90] test: preserve legacy rollup rows for SQLite query ordering --- tests/unit/test_sqlite_store.py | 55 +++++++++++++++++---------------- 1 file changed, 29 insertions(+), 26 deletions(-) diff --git a/tests/unit/test_sqlite_store.py b/tests/unit/test_sqlite_store.py index 4081188d4..94dd2a24b 100644 --- a/tests/unit/test_sqlite_store.py +++ b/tests/unit/test_sqlite_store.py @@ -433,32 +433,35 @@ def test_project_scoped_memory_and_rollup_queries_filter_before_limit() -> None: last_confirmed_at="2020-01-01T00:00:00Z", metadata_json={"project_scope": ["project-b"]}, ) - pending_a = _create_memory( - store, - status="candidate", - metadata_json={ - "project_scope": ["project-a"], - "candidate_kind": "memory_rollup", - "rollup_digest": "digest-a", - }, - ) - _create_memory( - store, - status="candidate", - metadata_json={ - "project_scope": ["project-b"], - "candidate_kind": "memory_rollup", - "rollup_digest": "digest-b", - }, - ) - accepted_a = _create_memory( - store, - metadata_json={ - "project_scope": ["project-a"], - "candidate_kind": "memory_rollup", - "rollup_key": "topic:a", - }, - ) + from alicebot_api.vnext_label_writes import without_insert_floor + # This test exercises query ordering on existing, unstamped legacy cards. + with without_insert_floor(): + pending_a = _create_memory( + store, + status="candidate", + metadata_json={ + "project_scope": ["project-a"], + "candidate_kind": "memory_rollup", + "rollup_digest": "digest-a", + }, + ) + _create_memory( + store, + status="candidate", + metadata_json={ + "project_scope": ["project-b"], + "candidate_kind": "memory_rollup", + "rollup_digest": "digest-b", + }, + ) + accepted_a = _create_memory( + store, + metadata_json={ + "project_scope": ["project-a"], + "candidate_kind": "memory_rollup", + "rollup_key": "topic:a", + }, + ) assert [row["id"] for row in store.list_memories(projects=("project-a",), limit=1)] == [accepted_a["id"]] assert store.count_memories(status="active", projects=("project-a",)) == 2 From a6c329b413e8ec7ead4d144b62ad344f8b4cf720 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 21:14:05 +0200 Subject: [PATCH 16/90] test: preserve legacy label fixtures for PostgreSQL repair --- .../integration/test_derived_domain_postgres.py | 16 +++++++++++----- 1 file changed, 11 insertions(+), 5 deletions(-) diff --git a/tests/integration/test_derived_domain_postgres.py b/tests/integration/test_derived_domain_postgres.py index 93220c97d..3c68bc2dc 100644 --- a/tests/integration/test_derived_domain_postgres.py +++ b/tests/integration/test_derived_domain_postgres.py @@ -14,13 +14,15 @@ from alicebot_api.vnext_agent_control import ALL_SENSITIVITY from alicebot_api.vnext_brain import BrainArtifactRequest, VNextBrainService from alicebot_api.vnext_store import PostgresVNextStore +from alicebot_api.vnext_label_writes import without_insert_floor def test_postgres_derived_domain_upgrade_and_generation(database_urls): config = make_alembic_config(database_urls["admin"]) command.upgrade(config, "20260721_0094") user = uuid4() - with user_connection(database_urls["app"], user) as conn: + # Seed pre-floor rows so this still tests the migration, not the live insert path. + with without_insert_floor(), user_connection(database_urls["app"], user) as conn: ContinuityStore(conn).create_user(user, "derived-fence@example.invalid", "Derived fence") store = PostgresVNextStore(conn) source = store.create_memory( @@ -86,7 +88,8 @@ def test_postgres_repair_as_documented_nobypassrls_owner(database_urls, monkeypa config = make_alembic_config(database_urls["admin"]) command.upgrade(config, "20260721_0094") user = uuid4() - with user_connection(database_urls["app"], user) as conn: + # Seed pre-floor rows so this still tests the migration, not the live insert path. + with without_insert_floor(), user_connection(database_urls["app"], user) as conn: ContinuityStore(conn).create_user(user, "owner-repair@example.invalid", "Owner repair") store = PostgresVNextStore(conn) source = store.create_memory( @@ -290,7 +293,8 @@ def test_promoted_artifact_uuid_alias_repaired(database_urls): config = make_alembic_config(database_urls["admin"]) command.upgrade(config, "20260721_0094") user = uuid4() - with user_connection(database_urls["app"], user) as conn: + # Seed pre-floor rows so this still tests the migration, not the live insert path. + with without_insert_floor(), user_connection(database_urls["app"], user) as conn: ContinuityStore(conn).create_user(user, "alias@example.invalid", "Alias fixture") store = PostgresVNextStore(conn) memory = store.create_memory({"memory_key": "health", "canonical_text": "Private observation", @@ -323,7 +327,8 @@ def test_postgres_repair_reads_every_spelling_of_a_recorded_id(database_urls, sp config = make_alembic_config(database_urls["admin"]) command.upgrade(config, "20260721_0094") user = uuid4() - with user_connection(database_urls["app"], user) as conn: + # Seed pre-floor rows so this still tests the migration, not the live insert path. + with without_insert_floor(), user_connection(database_urls["app"], user) as conn: ContinuityStore(conn).create_user(user, "spelling@example.invalid", "Spelling fixture") store = PostgresVNextStore(conn) health = store.create_memory( @@ -351,7 +356,8 @@ def test_postgres_repair_refuses_an_update_that_changes_no_row(database_urls, mo config = make_alembic_config(database_urls["admin"]) command.upgrade(config, "20260721_0094") user = uuid4() - with user_connection(database_urls["app"], user) as conn: + # Seed pre-floor rows so this still tests the migration, not the live insert path. + with without_insert_floor(), user_connection(database_urls["app"], user) as conn: ContinuityStore(conn).create_user(user, "zero-row@example.invalid", "Zero row fixture") store = PostgresVNextStore(conn) health = store.create_memory( From d126d5d2f608117fa696078a858fd0d917380532 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 21:14:33 +0200 Subject: [PATCH 17/90] fix: conservatively raise unresolved legacy relabel descendants --- apps/api/src/alicebot_api/vnext_label_writes.py | 14 +++++++++++++- .../unit/test_sqlite_derived_labels_write_path.py | 14 ++++++++++++++ 2 files changed, 27 insertions(+), 1 deletion(-) diff --git a/apps/api/src/alicebot_api/vnext_label_writes.py b/apps/api/src/alicebot_api/vnext_label_writes.py index 43a147fbe..5c8ae250f 100644 --- a/apps/api/src/alicebot_api/vnext_label_writes.py +++ b/apps/api/src/alicebot_api/vnext_label_writes.py @@ -11,6 +11,7 @@ from collections.abc import Iterator, Mapping, Sequence from contextlib import contextmanager from functools import wraps +from dataclasses import replace from typing import Any from alicebot_api.vnext_agent_control import RESTRICTED_DOMAINS @@ -481,7 +482,18 @@ def propagate(store: Any, changed: Sequence[tuple[str, str]], *, cause: str) -> for row in affected: label = settled.by_stored(str(row.get("kind")), str(row.get("id"))) if label.unverified: - continue + ancestry, exceeded = collect_label_rows(store, [row], max_nodes=PROPAGATION_BOUND) + if exceeded: + raise LabelPropagationTooLarge(f"label propagation stopped after {PROPAGATION_BOUND} rows") + label = replace( + label, + domain=generation_domain(label.domain, [str(node.get("domain") or "unknown") for node in ancestry]), + sensitivity="regulated", + project_floor=union_floor(label.project_floor, [ + *(stored_scope(str(node["kind"]), node) for node in ancestry), + *(project_floor_shape(node)[1] for node in ancestry), + ]), + ) previous = _label_fields(row) current = (label.domain, label.sensitivity, tuple(label.project_scope), tuple(label.project_floor)) if ( diff --git a/tests/unit/test_sqlite_derived_labels_write_path.py b/tests/unit/test_sqlite_derived_labels_write_path.py index 29c705721..d430002b6 100644 --- a/tests/unit/test_sqlite_derived_labels_write_path.py +++ b/tests/unit/test_sqlite_derived_labels_write_path.py @@ -291,3 +291,17 @@ def test_alias_twins_cannot_choose_a_public_source(tmp_path: Path) -> None: conn.execute("UPDATE sources SET id = ? WHERE id = ?", (str(restricted["id"]).replace("-", "").upper(), public["id"])) copy = store.create_memory({"memory_key": "twins-copy", "canonical_text": "copy", "status": "active", "domain": "unknown", "sensitivity": "public", "metadata_json": {"source_id": str(restricted["id"])}}) assert copy["sensitivity"] == "regulated" + + +def test_relabel_regulates_legacy_summary_with_missing_ancestry(tmp_path: Path) -> None: + with _vault(tmp_path / "missing-branch.sqlite3") as conn: + store = SQLiteVNextStore(conn, USER) + source = store.create_source({"source_type": "note", "title": "input", "content_hash": "missing-branch", "domain": "unknown", "sensitivity": "public", "metadata_json": {"project_scope": [ALPHA]}}) + copy = store.create_memory({"memory_key": "copy", "canonical_text": "copy", "status": "active", "domain": "unknown", "sensitivity": "public", "metadata_json": {"source_id": source["id"]}}) + with without_insert_floor(): + summary = store.create_memory({"memory_key": "summary", "canonical_text": "summary", "status": "active", "domain": "unknown", "sensitivity": "public", "metadata_json": {"consolidation": {"cluster_member_ids": [copy["id"], "missing"]}}}) + store.update_source(source_id=str(source["id"]), patch={"domain": "health", "sensitivity": "confidential"}, actor_type="user") + row = store.get_memory(str(summary["id"])) + assert row["domain"] == "health" + assert row["sensitivity"] == "regulated" + assert ALPHA in row["metadata_json"]["project_floor"] From 619a73db65f922080a562f20a769fbb8aa93fcc7 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 21:15:43 +0200 Subject: [PATCH 18/90] test: retain all stored alias restrictions in insert floors --- tests/unit/test_sqlite_derived_labels_write_path.py | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/tests/unit/test_sqlite_derived_labels_write_path.py b/tests/unit/test_sqlite_derived_labels_write_path.py index d430002b6..3bfd61cca 100644 --- a/tests/unit/test_sqlite_derived_labels_write_path.py +++ b/tests/unit/test_sqlite_derived_labels_write_path.py @@ -286,11 +286,13 @@ def test_sqlite_label_lookup_preserves_uuid_aliases_and_kinds(tmp_path: Path) -> def test_alias_twins_cannot_choose_a_public_source(tmp_path: Path) -> None: with _vault(tmp_path / "twins.sqlite3") as conn: store = SQLiteVNextStore(conn, USER) - restricted = store.create_source({"source_type": "note", "title": "restricted", "content_hash": "restricted", "domain": "health", "sensitivity": "confidential"}) - public = store.create_source({"source_type": "note", "title": "public", "content_hash": "public", "domain": "project", "sensitivity": "public"}) + restricted = store.create_source({"source_type": "note", "title": "restricted", "content_hash": "restricted", "domain": "health", "sensitivity": "confidential", "metadata_json": {"project_scope": [ALPHA]}}) + public = store.create_source({"source_type": "note", "title": "public", "content_hash": "public", "domain": "project", "sensitivity": "public", "metadata_json": {"project_scope": ["prj_" + "b" * 16]}}) conn.execute("UPDATE sources SET id = ? WHERE id = ?", (str(restricted["id"]).replace("-", "").upper(), public["id"])) copy = store.create_memory({"memory_key": "twins-copy", "canonical_text": "copy", "status": "active", "domain": "unknown", "sensitivity": "public", "metadata_json": {"source_id": str(restricted["id"])}}) assert copy["sensitivity"] == "regulated" + assert copy["domain"] == "health" + assert set(copy["metadata_json"]["project_floor"]) == {ALPHA, "prj_" + "b" * 16} def test_relabel_regulates_legacy_summary_with_missing_ancestry(tmp_path: Path) -> None: From e57f93b0b764289d47a6928980b4375d8f9e513c Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 21:15:26 +0200 Subject: [PATCH 19/90] chore: document verified label SQL and rollback scan exceptions --- apps/api/src/alicebot_api/vnext_label_writes.py | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/apps/api/src/alicebot_api/vnext_label_writes.py b/apps/api/src/alicebot_api/vnext_label_writes.py index 5c8ae250f..9fd571284 100644 --- a/apps/api/src/alicebot_api/vnext_label_writes.py +++ b/apps/api/src/alicebot_api/vnext_label_writes.py @@ -261,7 +261,7 @@ def acquire_exclusive_label_lock(store: Any) -> None: finally: try: cur.execute("SELECT set_config('lock_timeout', %s, true)", (str(previous),)) - except Exception: + except Exception: # nosec B110 # aborted transactions cannot restore local settings; rollback clears them # A lock timeout aborts the transaction. Rollback drops the local setting. pass @@ -323,7 +323,7 @@ def _sqlite_dependants(store: Any, table: str, kind: str, compacts: Sequence[str SELECT id, user_id, domain, sensitivity, metadata_json{extra} FROM {table} WHERE user_id = ? AND ({text_clause}{value_sql}{column_sql}) - """, + """, # nosec B608 # internal literal table/columns; every external value is bound tuple(params), ) for row in rows: @@ -353,7 +353,7 @@ def _postgres_dependants(store: Any, table: str, kind: str, compacts: Sequence[s SELECT id::text AS id, user_id::text AS user_id, domain, sensitivity, metadata_json{extra} FROM {table} WHERE ({text_clause}{value_sql}) - """, + """, # nosec B608 # internal literal table/columns; every external value is bound tuple(params), ) for row in rows: @@ -439,7 +439,7 @@ def write_settled_label( UPDATE {table} SET domain = ?, sensitivity = ?, metadata_json = ?{project_sql} WHERE id = ? AND user_id = ? AND domain = ? AND sensitivity = ? - """, + """, # nosec B608 # table comes from the closed kind map; values are bound tuple(params), ) require_changed(int(cursor.rowcount), table, str(row_id)) @@ -456,7 +456,7 @@ def write_settled_label( SET domain = %s, sensitivity = %s, metadata_json = %s::jsonb{project_sql} WHERE id = %s::uuid AND domain = %s AND sensitivity = %s RETURNING id - """, + """, # nosec B608 # table comes from the closed kind map; values are bound tuple(params), ) From d334131f47d90a364cdf5f56f4f9d9e7001564cb Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 21:17:53 +0200 Subject: [PATCH 20/90] test: verify canonical identity ambiguity and valid alias controls --- tests/unit/test_derived_labels_kernel.py | 18 ++++++++++++++++++ 1 file changed, 18 insertions(+) diff --git a/tests/unit/test_derived_labels_kernel.py b/tests/unit/test_derived_labels_kernel.py index 3d1528b65..ecc3fbcf6 100644 --- a/tests/unit/test_derived_labels_kernel.py +++ b/tests/unit/test_derived_labels_kernel.py @@ -817,3 +817,21 @@ def test_rank_table_matches_the_spec_order() -> None: assert SENSITIVITY_RANK["public"] < SENSITIVITY_RANK["unknown"] == SENSITIVITY_RANK["internal"] assert SENSITIVITY_RANK["sacred"] == SENSITIVITY_RANK["regulated"] assert "health" in RESTRICTED_DOMAINS + + +def test_distinct_stored_aliases_are_unverified_but_single_alias_and_other_kind_are_valid() -> None: + canonical = str(UUID(SOURCE_UUID)) + alias = "{" + canonical.upper() + "}" + source = _source(canonical, domain="health", sensitivity="confidential", scope=[ALPHA]) + twin = _source(alias, domain="project", scope=[BETA]) + report = _brief("ambiguous", sources=[canonical]) + for rows in ([source, twin, report], [twin, source, report]): + result = settle_labels(rows).by_stored("artifact", "ambiguous") + assert result.unverified is True + assert result.reason == "dependency_unverified" + single = settle_labels([twin, report]).by_stored("artifact", "ambiguous") + assert single.unverified is False + other_kind = _memory(canonical, domain="project", sensitivity="public") + result = settle_labels([source, other_kind, report]).by_stored("artifact", "ambiguous") + assert result.unverified is False + assert result.domain == "health" From e40f56d3e36732058f6a4c12c90d4e693189b5a7 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 21:17:03 +0200 Subject: [PATCH 21/90] test: verify actual Postgres belief report insert ancestry --- tests/integration/test_label_floor_ancestry_postgres.py | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/tests/integration/test_label_floor_ancestry_postgres.py b/tests/integration/test_label_floor_ancestry_postgres.py index 29dc73cd1..2fdc86b95 100644 --- a/tests/integration/test_label_floor_ancestry_postgres.py +++ b/tests/integration/test_label_floor_ancestry_postgres.py @@ -25,3 +25,9 @@ def test_pg_copy_and_summary_keep_tenant_and_ancestry(migrated_database_urls, do belief = store.read_label_rows("belief", [str(belief_id)])[0] assert belief["sensitivity"] == sensitivity assert str(belief["memory_id"]) == str(copy["id"]) + + record = {"v": 1, "sources": [], "memories": [], "open_loops": [], "artifacts": [], "beliefs": [str(belief_id)], "counts": {"sources": 0, "memories": 0, "open_loops": 0, "artifacts": 0, "beliefs": 1}} + artifact = store.create_artifact({"artifact_type": "daily_brief", "title": "synthetic", "content_markdown": "synthetic", "domain": "unknown", "sensitivity": "public", "metadata_json": {"workflow": "daily_brief", "derived_from": record}}) + assert artifact["sensitivity"] == sensitivity + if domain == "health": + assert artifact["domain"] == domain From ef05d6e1cf46093242044e19809a536e92e4cb97 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 21:17:03 +0200 Subject: [PATCH 22/90] test: mark SQLite rollup query fixtures as legacy rows --- tests/unit/test_vnext_rollups.py | 31 +++++++++++++++++-------------- 1 file changed, 17 insertions(+), 14 deletions(-) diff --git a/tests/unit/test_vnext_rollups.py b/tests/unit/test_vnext_rollups.py index af56899e0..b65650ba4 100644 --- a/tests/unit/test_vnext_rollups.py +++ b/tests/unit/test_vnext_rollups.py @@ -1240,20 +1240,23 @@ def test_sqlite_rollup_reads_are_exact_deduplicated_and_bounded() -> None: conn, store = _live_store() def create_row(name: str, *, status: str, metadata: JsonObject) -> JsonObject: - return store.create_memory( - { - "memory_key": f"memory.{name}", - "value": {"text": name}, - "status": status, - "memory_type": "semantic", - "title": name, - "canonical_text": name, - "summary": name, - "domain": "personal", - "sensitivity": "internal", - "metadata_json": metadata, - } - ) + from alicebot_api.vnext_label_writes import without_insert_floor + # Query contract fixture: existing unstamped cards retain their stored labels. + with without_insert_floor(): + return store.create_memory( + { + "memory_key": f"memory.{name}", + "value": {"text": name}, + "status": status, + "memory_type": "semantic", + "title": name, + "canonical_text": name, + "summary": name, + "domain": "personal", + "sensitivity": "internal", + "metadata_json": metadata, + } + ) ordinary_active = create_row("ordinary-active", status="active", metadata={}) ordinary_accepted = create_row("ordinary-accepted", status="accepted", metadata={}) From 1736302832897cd0ef82c3d545bf511df35de9f9 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 21:10:37 +0200 Subject: [PATCH 23/90] test: mark intentionally unstamped rollup lookup fixtures --- tests/unit/test_group_scope_consumers.py | 81 ++++++++++++------------ 1 file changed, 42 insertions(+), 39 deletions(-) diff --git a/tests/unit/test_group_scope_consumers.py b/tests/unit/test_group_scope_consumers.py index 38693981b..d5fbd7f0f 100644 --- a/tests/unit/test_group_scope_consumers.py +++ b/tests/unit/test_group_scope_consumers.py @@ -137,26 +137,28 @@ def test_rollup_lookups_match_a_floor_when_the_scope_is_empty() -> None: user_id = str(uuid4()) ensure_sqlite_user(conn, user_id, "group-scope@example.com", "Group Scope") store = SQLiteVNextStore(conn, user_id) - card = store.create_memory( - { - "memory_key": "vnext.rollup.floor-card", - "value": {"text": "card"}, - "status": "candidate", - "memory_type": "semantic", - "title": "Floor card", - "canonical_text": "Floor card", - "summary": "Floor card", - "domain": "personal", - "sensitivity": "internal", - "metadata_json": { - "candidate_kind": ROLLUP_CANDIDATE_KIND, - "rollup_digest": "digest-floor", - "rollup_key": "topic:floor", - "project_scope": [], - "project_floor": [ALPHA, BETA], - }, - } - ) + from alicebot_api.vnext_label_writes import without_insert_floor + with without_insert_floor(): + card = store.create_memory( + { + "memory_key": "vnext.rollup.floor-card", + "value": {"text": "card"}, + "status": "candidate", + "memory_type": "semantic", + "title": "Floor card", + "canonical_text": "Floor card", + "summary": "Floor card", + "domain": "personal", + "sensitivity": "internal", + "metadata_json": { + "candidate_kind": ROLLUP_CANDIDATE_KIND, + "rollup_digest": "digest-floor", + "rollup_key": "topic:floor", + "project_scope": [], + "project_floor": [ALPHA, BETA], + }, + } + ) pending = store.list_pending_rollup_candidates( rollup_digests=("digest-floor",), domains=["personal"], @@ -175,25 +177,26 @@ def test_rollup_lookups_match_a_floor_when_the_scope_is_empty() -> None: projects=("prj_" + "c" * 16,), ) assert missed == [] - accepted = store.create_memory( - { - "memory_key": "vnext.rollup.floor-accepted", - "value": {"text": "accepted"}, - "status": "active", - "memory_type": "semantic", - "title": "Accepted floor card", - "canonical_text": "Accepted floor card", - "summary": "Accepted floor card", - "domain": "personal", - "sensitivity": "internal", - "metadata_json": { - "candidate_kind": ROLLUP_CANDIDATE_KIND, - "rollup_key": "topic:floor-accepted", - "project_scope": [], - "project_floor": [ALPHA, BETA], - }, - } - ) + with without_insert_floor(): + accepted = store.create_memory( + { + "memory_key": "vnext.rollup.floor-accepted", + "value": {"text": "accepted"}, + "status": "active", + "memory_type": "semantic", + "title": "Accepted floor card", + "canonical_text": "Accepted floor card", + "summary": "Accepted floor card", + "domain": "personal", + "sensitivity": "internal", + "metadata_json": { + "candidate_kind": ROLLUP_CANDIDATE_KIND, + "rollup_key": "topic:floor-accepted", + "project_scope": [], + "project_floor": [ALPHA, BETA], + }, + } + ) cards = store.list_accepted_rollup_cards( rollup_keys=("topic:floor-accepted",), domains=["personal"], From b81d463e10a8e2e681281eaf154bbec525886a96 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 21:10:37 +0200 Subject: [PATCH 24/90] fix: narrow producer provenance rows before iterating --- apps/api/src/alicebot_api/vnext_derived_labels.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/apps/api/src/alicebot_api/vnext_derived_labels.py b/apps/api/src/alicebot_api/vnext_derived_labels.py index 66270a310..504fcfce1 100644 --- a/apps/api/src/alicebot_api/vnext_derived_labels.py +++ b/apps/api/src/alicebot_api/vnext_derived_labels.py @@ -1243,7 +1243,8 @@ def stamp_derived_from(payload: dict[str, object], rows_by_kind: Mapping[str, ob record: dict[str, object] = {"v": 1} counts: dict[str, int] = {} for key in ("sources", "memories", "open_loops", "artifacts", "beliefs"): - rows = rows_by_kind.get(key) or [] + raw_rows = rows_by_kind.get(key) + rows = raw_rows if isinstance(raw_rows, (list, tuple)) else [] ids = [str(row.get("id")) for row in rows if isinstance(row, Mapping) and row.get("id") is not None] record[key] = ids counts[key] = len(ids) From 37a9a7c8233908e55716ae3594c12b97e7881b3d Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 21:11:54 +0200 Subject: [PATCH 25/90] fix: preserve producer row types through locked admission --- apps/api/src/alicebot_api/vnext_derived_labels.py | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/apps/api/src/alicebot_api/vnext_derived_labels.py b/apps/api/src/alicebot_api/vnext_derived_labels.py index 504fcfce1..9ec1daae0 100644 --- a/apps/api/src/alicebot_api/vnext_derived_labels.py +++ b/apps/api/src/alicebot_api/vnext_derived_labels.py @@ -15,6 +15,7 @@ from collections.abc import Iterable, Mapping, Sequence from dataclasses import dataclass, replace from uuid import UUID +from typing import TypeVar, overload from alicebot_api.vnext_agent_control import RESTRICTED_DOMAINS from alicebot_api.vnext_derived_domain import derived_domain @@ -1262,6 +1263,17 @@ def with_derived_from(metadata: Mapping[str, object], rows_by_kind: Mapping[str, return dict(stamped) if isinstance(stamped, Mapping) else {} +_InputRow = TypeVar("_InputRow", bound=Mapping[str, object]) + + +@overload +def admit_when_locked(kind: str, rows: Sequence[_InputRow], projects: tuple[str, ...] | None) -> list[_InputRow]: ... + + +@overload +def admit_when_locked(kind: str, rows: object, projects: tuple[str, ...] | None) -> list[Mapping[str, object]]: ... + + def admit_when_locked( kind: str, rows: object, From 08b1409406e92befc7e357f408933ac4b5cae264 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 21:12:46 +0200 Subject: [PATCH 26/90] fix: align dynamic locked admission implementation with overloads --- apps/api/src/alicebot_api/vnext_derived_labels.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/apps/api/src/alicebot_api/vnext_derived_labels.py b/apps/api/src/alicebot_api/vnext_derived_labels.py index 9ec1daae0..ae2178371 100644 --- a/apps/api/src/alicebot_api/vnext_derived_labels.py +++ b/apps/api/src/alicebot_api/vnext_derived_labels.py @@ -15,7 +15,7 @@ from collections.abc import Iterable, Mapping, Sequence from dataclasses import dataclass, replace from uuid import UUID -from typing import TypeVar, overload +from typing import Any, TypeVar, overload from alicebot_api.vnext_agent_control import RESTRICTED_DOMAINS from alicebot_api.vnext_derived_domain import derived_domain @@ -1278,7 +1278,7 @@ def admit_when_locked( kind: str, rows: object, projects: tuple[str, ...] | None, -) -> list[Mapping[str, object]]: +) -> list[Any]: """Keep every row when ``projects`` is None. Otherwise keep rows inside that binding.""" items = list(rows) if isinstance(rows, (list, tuple)) else [] From 00ae879f77da3fc527e94611aa09c40bec1b221a Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 21:08:47 +0200 Subject: [PATCH 27/90] fix: bound label reads by ancestry depth rather than fanout --- .../api/src/alicebot_api/vnext_label_guard.py | 38 +++---------- tests/unit/test_label_guard_exact_doors.py | 53 +++++++++++++++++++ 2 files changed, 60 insertions(+), 31 deletions(-) diff --git a/apps/api/src/alicebot_api/vnext_label_guard.py b/apps/api/src/alicebot_api/vnext_label_guard.py index cdb669ed1..a858226b0 100644 --- a/apps/api/src/alicebot_api/vnext_label_guard.py +++ b/apps/api/src/alicebot_api/vnext_label_guard.py @@ -21,10 +21,10 @@ HOP_BOUND, NODE_BOUND, canon_kind, - dependencies_of, is_derived, settle_labels, ) +from alicebot_api.vnext_label_closure import collect_label_rows from alicebot_api.vnext_project_scope import project_floor_shape, project_scopes_overlap, resolve_project_scope @@ -52,7 +52,7 @@ class LabelGuard: domains: tuple[str, ...] = () sensitivity_allowed: tuple[str, ...] = () projects: tuple[str, ...] = () - _nodes: dict[tuple[str, str], dict[str, object]] | None = None + _nodes: dict[tuple[str, str], list[dict[str, object]]] | None = None @classmethod def for_fence(cls, store: Any, fence: Any) -> LabelGuard: @@ -161,35 +161,11 @@ def _admits_effective(self, row: Mapping[str, object]) -> bool: def _collected(self, kind: str, row: Mapping[str, object]) -> list[dict[str, object]]: if self._nodes is None: self._nodes = {} - pending: list[tuple[str, Mapping[str, object]]] = [(canon_kind(kind), row)] - hops = 0 - while pending and len(self._nodes) < NODE_BOUND and hops < HOP_BOUND: - hops += 1 - name, current = pending.pop(0) - node_id = str(current.get("id") or "") - key = (name, node_id) - if key in self._nodes: - continue - node = dict(current) - node["kind"] = name - node["user_id"] = _GUARD_USER - self._nodes[key] = node - if not is_derived(name, node): - continue - grouped: dict[str, list[str]] = {} - for dep_kind, dep_id in dependencies_of(name, node): - grouped.setdefault(canon_kind(dep_kind), []).append(str(dep_id)) - reader = getattr(self.store, "read_label_rows", None) - if not callable(reader): - continue - for dep_kind, ids in grouped.items(): - missing = [item for item in ids if (dep_kind, item) not in self._nodes] - if not missing: - continue - for found in reader(dep_kind, missing): - if isinstance(found, Mapping): - pending.append((dep_kind, found)) - return list(self._nodes.values()) + nodes, _exceeded = collect_label_rows( + self.store, [{**dict(row), "kind": canon_kind(kind)}], + max_nodes=NODE_BOUND, max_hops=HOP_BOUND, cache=self._nodes, user_id=_GUARD_USER, + ) + return nodes def effective_row_for_fence( diff --git a/tests/unit/test_label_guard_exact_doors.py b/tests/unit/test_label_guard_exact_doors.py index 05744df20..57fd99b40 100644 --- a/tests/unit/test_label_guard_exact_doors.py +++ b/tests/unit/test_label_guard_exact_doors.py @@ -232,3 +232,56 @@ def test_a_cited_memory_is_judged_by_its_source() -> None: store = _LabelStore() with pytest.raises(MemoryRefNotFoundError): resolve_attachable_memory_id(store, MEMORY_ID, fence=SourceReadFence.for_identity(_trusted())) + + +from uuid import UUID + +def provenance(sources=(), memories=()): + refs = {'sources': list(sources), 'memories': list(memories), 'open_loops': [], 'artifacts': [], 'beliefs': []} + return {'v': 1, **refs, 'counts': {k: len(v) for k, v in refs.items()}} + + +class ArtifactStore: + def __init__(self, n): + self.sources = [dict(id=str(UUID(int=i + 1)), domain='project', sensitivity='public', metadata_json={}) for i in range(n)] + self.artifact = dict(id=str(UUID(int=10000)), artifact_type='daily_brief', domain='project', sensitivity='public', content_markdown='Public report', metadata_json={'workflow': 'daily_brief', 'derived_from': provenance(sources=[s['id'] for s in self.sources])}) + def read_label_rows(self, kind, ids): + return [s for s in self.sources if s['id'] in ids] if kind == 'source' else [] + def get_artifact(self, artifact_id): + return self.artifact + def upsert_agent_identity(self, *args, **kwargs): + pass + def append_event(self, event): + return event + + +@pytest.mark.parametrize('n', [31, 32, 100]) +def test_public_report_fanout_remains_readable(n): + store = ArtifactStore(n) + result = _vnext_authorized_artifact(store=store, identity=AgentIdentity(agent_id='trusted-key', permission_profile='trusted_local_agent'), artifact_id=store.artifact['id'], action='artifact.read', for_update=False) + assert result is not None + + +def test_cached_guard_keeps_each_roots_independent_budget() -> None: + store = ArtifactStore(100) + guard = LabelGuard(store, active=True) + assert guard.effective_row("artifact", store.artifact)["unverified"] is False + assert guard.effective_row("artifact", store.artifact)["unverified"] is False + + +def test_belief_dependency_loads_backing_memory_ancestry() -> None: + store = _LabelStore() + belief_id = str(UUID(int=55)) + belief = {"id": belief_id, "memory_id": MEMORY_ID, "domain": "unknown", "sensitivity": "public", "metadata_json": {}} + artifact = dict(store.artifact) + artifact["metadata_json"] = {"workflow": "daily_brief", "derived_from": provenance()} + artifact["metadata_json"]["derived_from"]["beliefs"] = [belief_id] + artifact["metadata_json"]["derived_from"]["counts"]["beliefs"] = 1 + def reader(kind, ids): + rows = {"source": [store.source], "memory": [store.memory], "belief": [belief]}.get(kind, []) + return [row for row in rows if row["id"] in ids] + store.read_label_rows = reader + effective = LabelGuard(store, active=True).effective_row("artifact", artifact) + assert effective["unverified"] is False + assert effective["domain"] == "health" + assert effective["sensitivity"] == "confidential" From 99fcfa5475300f467dcc06ca63c193c5a30d11ca Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 21:10:54 +0200 Subject: [PATCH 28/90] fix: narrow read guard metadata before copying --- apps/api/src/alicebot_api/vnext_label_guard.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/apps/api/src/alicebot_api/vnext_label_guard.py b/apps/api/src/alicebot_api/vnext_label_guard.py index a858226b0..e6c3afb4f 100644 --- a/apps/api/src/alicebot_api/vnext_label_guard.py +++ b/apps/api/src/alicebot_api/vnext_label_guard.py @@ -95,7 +95,8 @@ def effective_row(self, kind: str, row: Mapping[str, object] | None) -> Mapping[ settled = settle_labels(nodes, on_cycle="unverified", max_hops=HOP_BOUND, max_nodes=NODE_BOUND) label = settled.by_stored(kind, str(row.get("id") or ""), user_id=_GUARD_USER) copy = dict(row) - metadata = dict(copy.get("metadata_json")) if isinstance(copy.get("metadata_json"), Mapping) else {} + raw_metadata = copy.get("metadata_json") + metadata = dict(raw_metadata) if isinstance(raw_metadata, Mapping) else {} if label.unverified: copy["domain"] = label.domain copy["sensitivity"] = "regulated" From 6377ccaea89d4fd6510a3413d937b471ce70a101 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 21:12:46 +0200 Subject: [PATCH 29/90] test: verify ancestry depth boundaries and cache reuse --- tests/unit/test_label_guard_exact_doors.py | 17 +++++++++++++++++ 1 file changed, 17 insertions(+) diff --git a/tests/unit/test_label_guard_exact_doors.py b/tests/unit/test_label_guard_exact_doors.py index 57fd99b40..2aa23b2a2 100644 --- a/tests/unit/test_label_guard_exact_doors.py +++ b/tests/unit/test_label_guard_exact_doors.py @@ -285,3 +285,20 @@ def reader(kind, ids): assert effective["unverified"] is False assert effective["domain"] == "health" assert effective["sensitivity"] == "confidential" + + +def test_depth_boundary_and_cache_are_independent_of_width() -> None: + from alicebot_api.vnext_derived_labels import HOP_BOUND + # Extracted memories recursively reference memories through consolidation markers. + class Chain: + def __init__(self, length): + self.rows = [{"id": str(i), "domain": "project", "sensitivity": "public", "metadata_json": {"consolidation": {"cluster_member_ids": [str(i+1)]}}} for i in range(length)] + self.rows.append({"id": str(length), "domain": "project", "sensitivity": "public", "metadata_json": {}}) + def read_label_rows(self, kind, ids): + return [row for row in self.rows if row["id"] in ids] if kind == "memory" else [] + within = Chain(HOP_BOUND) + guard = LabelGuard(within, active=True) + assert guard.effective_row("memory", within.rows[0])["unverified"] is False + assert guard.effective_row("memory", within.rows[0])["unverified"] is False + beyond = Chain(HOP_BOUND+1) + assert LabelGuard(beyond, active=True).effective_row("memory", beyond.rows[0])["unverified"] is True From 75a57128b3e0c12340081ddd0c89cf565add2104 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 21:15:26 +0200 Subject: [PATCH 30/90] test: deny locked admin reads with ambiguous source aliases --- tests/unit/test_label_guard_exact_doors.py | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/tests/unit/test_label_guard_exact_doors.py b/tests/unit/test_label_guard_exact_doors.py index 2aa23b2a2..bc255ceab 100644 --- a/tests/unit/test_label_guard_exact_doors.py +++ b/tests/unit/test_label_guard_exact_doors.py @@ -302,3 +302,15 @@ def read_label_rows(self, kind, ids): assert guard.effective_row("memory", within.rows[0])["unverified"] is False beyond = Chain(HOP_BOUND+1) assert LabelGuard(beyond, active=True).effective_row("memory", beyond.rows[0])["unverified"] is True + + +def test_ambiguous_source_alias_blocks_locked_admin() -> None: + store = _LabelStore() + store.source["metadata_json"] = {"project_scope": ["prj_" + "b" * 16]} + twin = {**store.source, "id": "{" + SOURCE_ID + "}", "domain": "project", "sensitivity": "public", "metadata_json": {"project_scope": [ALPHA]}} + store.read_label_rows = lambda kind, ids: [store.source, twin] if kind == "source" else [] + effective = LabelGuard(store, active=True).effective_row("artifact", store.artifact) + assert effective["unverified"] is True + identity = AgentIdentity(agent_id="locked-admin", permission_profile="admin", project_scope=(ALPHA,), project_scope_locked=True) + with pytest.raises(AgentPolicyBlockedError): + _vnext_authorized_artifact(store=store, identity=identity, artifact_id=ARTIFACT_ID, action="artifact.read", for_update=False) From 2250261104d61884a699eba3b97eb1fcd41cf28c Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 21:19:43 +0200 Subject: [PATCH 31/90] fix: retain identity ambiguity through weekly ancestry backfill --- apps/api/src/alicebot_api/vnext_derived_labels.py | 10 +++++----- tests/unit/test_derived_labels_kernel.py | 11 +++++++++++ 2 files changed, 16 insertions(+), 5 deletions(-) diff --git a/apps/api/src/alicebot_api/vnext_derived_labels.py b/apps/api/src/alicebot_api/vnext_derived_labels.py index 875db8f40..d368dea30 100644 --- a/apps/api/src/alicebot_api/vnext_derived_labels.py +++ b/apps/api/src/alicebot_api/vnext_derived_labels.py @@ -977,11 +977,6 @@ def settle_labels( own: dict[tuple[str, str, str], set[tuple[str, str, str]]] = {} problems: dict[tuple[str, str, str], str] = {} - stored_ids: dict[tuple[str, str, str], str] = {} - for label, _row in prepared: - previous_id = stored_ids.setdefault(label.key, label.stored_id) - if previous_id != label.stored_id: - problems[label.key] = "ambiguous_identity" for label, row in prepared: if not label.derived: continue @@ -995,6 +990,11 @@ def settle_labels( keyed = {(kind, label.user_id, row_id) for kind, row_id in deps} own[label.key] = keyed _weekly_parent_deps(labels, own, prepared) + stored_ids: dict[tuple[str, str, str], str] = {} + for label, _row in prepared: + previous_id = stored_ids.setdefault(label.key, label.stored_id) + if previous_id != label.stored_id: + problems[label.key] = "ambiguous_identity" for key, reason in list(problems.items()): if reason == "no_record" and own.get(key): del problems[key] diff --git a/tests/unit/test_derived_labels_kernel.py b/tests/unit/test_derived_labels_kernel.py index ecc3fbcf6..a56de91f1 100644 --- a/tests/unit/test_derived_labels_kernel.py +++ b/tests/unit/test_derived_labels_kernel.py @@ -835,3 +835,14 @@ def test_distinct_stored_aliases_are_unverified_but_single_alias_and_other_kind_ result = settle_labels([source, other_kind, report]).by_stored("artifact", "ambiguous") assert result.unverified is False assert result.domain == "health" + + +def test_weekly_parent_backfill_cannot_clear_ambiguous_identity() -> None: + canonical = str(UUID(SOURCE_UUID)) + one = _memory(canonical, metadata_json={"discovered_by": "vnext_weekly_synthesis"}) + two = _memory("{" + canonical + "}", metadata_json={"discovered_by": "vnext_weekly_synthesis"}) + parent = _brief("parent", sources=["s"]) + _meta(parent, candidate_memory_ids=[canonical]) + settled = settle_labels([_source("s"), one, two, parent]) + assert settled.by_stored("memory", canonical).unverified is True + assert settled.by_stored("memory", "{" + canonical + "}").unverified is True From 5cd97bc2c4fdb785dc1de5522cd3a952e9d40a09 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 21:28:01 +0200 Subject: [PATCH 32/90] fix: apply checked memory relabels and propagate descendant floors --- .../alicebot_api/routers/vnext_memories.py | 9 ++++- .../test_label_floor_ancestry_postgres.py | 35 +++++++++++++++++++ 2 files changed, 43 insertions(+), 1 deletion(-) diff --git a/apps/api/src/alicebot_api/routers/vnext_memories.py b/apps/api/src/alicebot_api/routers/vnext_memories.py index 433d4716f..4bc6667eb 100644 --- a/apps/api/src/alicebot_api/routers/vnext_memories.py +++ b/apps/api/src/alicebot_api/routers/vnext_memories.py @@ -1246,7 +1246,14 @@ def review_vnext_memory( status_code=400, detail="vNext memory review text carries credential material" ) - updated = store.update_memory(memory_id=str(memory_id), patch=patch, actor_type=actor_type) + before_label = dict(existing) + updated = store.update_memory( + memory_id=str(memory_id), patch=patch, actor_type=actor_type, label_write=label_change, + ) + if label_change: + from alicebot_api.vnext_label_writes import propagate_after_write + + propagate_after_write(store, kind="memory", before=before_label, after=updated) if action in ("accept", "edit", "promote"): memory_service.refresh_memory_derived_state( updated, diff --git a/tests/integration/test_label_floor_ancestry_postgres.py b/tests/integration/test_label_floor_ancestry_postgres.py index 2fdc86b95..657f3a3b5 100644 --- a/tests/integration/test_label_floor_ancestry_postgres.py +++ b/tests/integration/test_label_floor_ancestry_postgres.py @@ -31,3 +31,38 @@ def test_pg_copy_and_summary_keep_tenant_and_ancestry(migrated_database_urls, do assert artifact["sensitivity"] == sensitivity if domain == "health": assert artifact["domain"] == domain + + +def test_checked_project_review_moves_scope_and_propagates_labels(migrated_database_urls, monkeypatch): + from alicebot_api.config import Settings + from alicebot_api.routers import vnext_memories as router + from alicebot_api.vnext_agent_keys import create_agent_key + from uuid import UUID + user_id = uuid4() + alpha, beta = "prj_" + "a" * 16, "prj_" + "b" * 16 + app_url = migrated_database_urls["app"] + with user_connection(app_url, user_id) as conn: + ContinuityStore(conn).create_user(user_id, f"review-labels-{user_id}@example.invalid", "Labels") + store = PostgresVNextStore(conn) + original = store.create_memory({"memory_key": "original", "canonical_text": "original", "status": "active", "domain": "project", "sensitivity": "public", "metadata_json": {"project_scope": [alpha]}}) + summary = store.create_memory({"memory_key": "summary", "canonical_text": "summary", "status": "active", "domain": "project", "sensitivity": "public", "metadata_json": {"consolidation": {"cluster_member_ids": [str(original["id"])]}, "project_scope": [alpha]}}) + _key, raw_key = create_agent_key(store, user_id=user_id, agent_id="alpha-only", permission_profile="admin_agent", project_scope=alpha) + _admin, admin_key = create_agent_key(store, user_id=user_id, agent_id="unbound-admin", permission_profile="admin_agent") + monkeypatch.setattr(router, "get_settings", lambda: Settings(database_url=app_url)) + request = router.VNextMemoryReviewRequest(user_id=user_id, action="assign_project", project_id=beta, domain="health", sensitivity="confidential") + denied = router.review_vnext_memory(UUID(str(original["id"])), request, authorization=f"Bearer {raw_key}") + assert denied.status_code == 403 + with user_connection(app_url, user_id) as conn: + store = PostgresVNextStore(conn) + assert store.get_memory(str(original["id"]))["metadata_json"]["project_scope"] == [alpha] + assert store.get_memory(str(summary["id"]))["sensitivity"] == "public" + moved = router.review_vnext_memory(UUID(str(original["id"])), request, authorization=f"Bearer {admin_key}") + assert moved.status_code == 200 + with user_connection(app_url, user_id) as conn: + store = PostgresVNextStore(conn) + original_after = store.get_memory(str(original["id"])) + summary_after = store.get_memory(str(summary["id"])) + assert original_after["metadata_json"]["project_scope"] == [beta] + assert summary_after["domain"] == "health" + assert summary_after["sensitivity"] == "confidential" + assert beta in summary_after["metadata_json"]["project_floor"] From 0699cf5daa25a46d318da61262d0270be7761532 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 21:28:01 +0200 Subject: [PATCH 33/90] test: distinguish label guards from business SQL in store adapters --- tests/unit/test_embedding_input_limits.py | 6 +- tests/unit/test_vnext_store.py | 349 +++++++++++----------- 2 files changed, 185 insertions(+), 170 deletions(-) diff --git a/tests/unit/test_embedding_input_limits.py b/tests/unit/test_embedding_input_limits.py index 5828a6af8..f51815e5c 100644 --- a/tests/unit/test_embedding_input_limits.py +++ b/tests/unit/test_embedding_input_limits.py @@ -980,8 +980,10 @@ def test_postgres_store_signs_the_cut_label_and_lists_rows_whose_label_differs() content_sha256="digest", signature_version=2, ) - cut_query, cut_params = connection.cursor_instance.queries[0] - plain_query, plain_params = connection.cursor_instance.queries[1] + writes = [(query, params) for query, params in connection.cursor_instance.queries if "UPDATE memories" in query] + assert len(writes) == 2 + cut_query, cut_params = writes[0] + plain_query, plain_params = writes[1] assert cut_params[1].obj["truncated_to_chars"] == 1000 # type: ignore[attr-defined] assert "truncated_to_chars" not in plain_params[1].obj # type: ignore[attr-defined] assert cut_query == plain_query diff --git a/tests/unit/test_vnext_store.py b/tests/unit/test_vnext_store.py index 4f14c7329..de7109f4d 100644 --- a/tests/unit/test_vnext_store.py +++ b/tests/unit/test_vnext_store.py @@ -39,6 +39,11 @@ def execute(self, query: str, params: tuple[object, ...] | None = None) -> None: assert query.count("%s") == len(params) self.executed.append((query, params)) + @property + def statements(self) -> list[tuple[str, tuple[object, ...] | None]]: + """Business SQL, with transaction guards retained separately in executed.""" + return [(query, params) for query, params in self.executed if "pg_advisory_xact_lock" not in query] + def fetchone(self) -> dict[str, Any] | None: if not self.fetchone_results: return None @@ -65,7 +70,7 @@ def _event_row(target_id: object | None = None) -> dict[str, object]: def _event_log_insert_count(cursor: RecordingCursor) -> int: - return sum(1 for query, _params in cursor.executed if "INSERT INTO event_log" in query) + return sum(1 for query, _params in cursor.statements if "INSERT INTO event_log" in query) def test_postgres_project_scope_sql_mirrors_conservative_python_identity() -> None: @@ -144,19 +149,20 @@ def test_source_crud_and_chunks_write_audit_events() -> None: assert chunks == [{"id": chunk_id, "source_id": source_id}] assert _event_log_insert_count(cursor) == 4 - source_insert_query, source_insert_params = cursor.executed[0] + assert "pg_advisory_xact_lock_shared" in cursor.executed[0][0] + source_insert_query, source_insert_params = cursor.statements[0] assert "INSERT INTO sources" in source_insert_query assert source_insert_params is not None assert isinstance(source_insert_params[-1], Jsonb) assert source_insert_params[-1].obj == {"path": "docs/spec.md"} source_update_query, source_update_params = next( - (query, params) for query, params in cursor.executed if "UPDATE sources" in query and "SET title" in query + (query, params) for query, params in cursor.statements if "UPDATE sources" in query and "SET title" in query ) assert "UPDATE sources" in source_update_query assert source_update_params is not None - chunk_query, chunk_params = cursor.executed[-1] + chunk_query, chunk_params = cursor.statements[-1] assert "WHERE source_id = %s::uuid" in chunk_query assert chunk_query.index("WHERE source_id") < chunk_query.index("LIMIT %s") assert chunk_params == (source_id, 17) @@ -164,7 +170,7 @@ def test_source_crud_and_chunks_write_audit_events() -> None: with pytest.raises(ValueError, match="limit must be positive"): store.list_source_chunks(source_id, limit=0) store.list_source_chunks(source_id, limit=10_000) - assert cursor.executed[-1][1] == (source_id, 501) + assert cursor.statements[-1][1] == (source_id, 501) assert isinstance(source_update_params[6], Jsonb) assert source_update_params[6].obj == {"rev": 2} @@ -178,7 +184,7 @@ def test_get_source_by_content_hash_uses_dedupe_lookup() -> None: assert source is not None assert source["id"] == source_id - query, params = cursor.executed[0] + query, params = cursor.statements[0] assert "FROM sources" in query assert "content_hash = %s" in query assert "deleted_at IS NULL" in query @@ -206,7 +212,7 @@ def test_get_or_create_source_uses_partial_unique_dedupe_claim() -> None: assert created is True assert source["id"] == source_id - query, _params = cursor.executed[0] + query, _params = cursor.statements[0] assert "ON CONFLICT (user_id, dedupe_key)" in query assert "WHERE deleted_at IS NULL AND dedupe_key IS NOT NULL" in query assert "DO NOTHING" in query @@ -232,8 +238,8 @@ def test_get_or_create_source_returns_concurrent_winner_without_create_event() - assert created is False assert source["id"] == source_id - assert len(cursor.executed) == 2 - assert "SELECT" in cursor.executed[1][0] + assert len(cursor.statements) == 2 + assert "SELECT" in cursor.statements[1][0] assert _event_log_insert_count(cursor) == 0 @@ -304,7 +310,7 @@ def test_update_source_recomputes_postgres_dedupe_key_with_the_same_statement() ) update_query, update_params = next( - (query, params) for query, params in cursor.executed if "UPDATE sources" in query and "SET title" in query + (query, params) for query, params in cursor.statements if "UPDATE sources" in query and "SET title" in query ) assert "metadata_json = COALESCE(%s, metadata_json)" in update_query assert "dedupe_key = %s" in update_query @@ -342,7 +348,7 @@ def test_update_source_postgres_collision_fails_before_mutation_event() -> None: patch={"metadata_json": {"raw_text": raw_text, "project_scope": ["Beta"]}}, ) - assert not any("UPDATE sources" in query for query, _params in cursor.executed) + assert not any("UPDATE sources" in query for query, _params in cursor.statements) assert _event_log_insert_count(cursor) == 0 @@ -371,7 +377,7 @@ def test_update_source_postgres_releases_key_when_changed_identity_has_no_raw_te ) update_params = next( - params for query, params in cursor.executed if "UPDATE sources" in query and "SET title" in query + params for query, params in cursor.statements if "UPDATE sources" in query and "SET title" in query ) assert update_params is not None assert update_params[8] is None @@ -435,9 +441,9 @@ def test_keyword_search_methods_apply_domain_sensitivity_and_limit_filters() -> assert memories[0]["id"] == "matched-1" assert sources[0]["id"] == "matched-1" assert open_loops[0]["id"] == "matched-1" - memory_query, memory_params = cursor.executed[0] - source_query, source_params = cursor.executed[1] - open_loop_query, open_loop_params = cursor.executed[2] + memory_query, memory_params = cursor.statements[0] + source_query, source_params = cursor.statements[1] + open_loop_query, open_loop_params = cursor.statements[2] assert "FROM memories" in memory_query assert "status IN ('active', 'accepted')" in memory_query assert "domain = ANY" in memory_query @@ -501,7 +507,7 @@ def test_project_scope_sql_uses_canonical_key_precedence_for_all_resources() -> store.list_artifacts(scope_projects=(project,), limit=1) store.list_open_loops(scope_projects=(project,), limit=1) - memory_query, source_query, artifact_query, open_loop_query = [query for query, _ in cursor.executed] + memory_query, source_query, artifact_query, open_loop_query = [query for query, _ in cursor.statements] for query, metadata_expression in ( (memory_query, "metadata_json"), (artifact_query, "metadata_json"), @@ -572,7 +578,7 @@ def test_project_scope_sql_uses_canonical_key_precedence_for_all_resources() -> assert "?| %s::text[]" in source_query assert "OR project_id::text = ANY" not in open_loop_query - for _query, params in cursor.executed: + for _query, params in cursor.statements: assert params is not None assert "canonical-project" in str(params) assert project not in str(params) @@ -589,7 +595,7 @@ def test_exact_memory_scope_lookup_uses_conservative_order_insensitive_identity( project_scope=(" Beta ", "ALICE", "alice"), ) - query, params = cursor.executed[0] + query, params = cursor.statements[0] assert "octet_length(normalized_scope.value) = char_length(normalized_scope.value)" in query assert 'COLLATE "C"' in query assert "metadata_json ? 'project_scope'" in query @@ -617,7 +623,7 @@ def test_search_memories_by_time_builds_window_predicate_and_proximity_order() - ) assert rows[0]["id"] == "memory-march" - query, params = cursor.executed[0] + query, params = cursor.statements[0] assert "FROM memories" in query assert "deleted_at IS NULL" in query assert "status IN ('active', 'accepted')" in query @@ -658,7 +664,7 @@ def test_search_memories_by_time_treats_naive_windows_as_utc() -> None: window_end=datetime(2023, 4, 1), ) - _query, params = cursor.executed[0] + _query, params = cursor.statements[0] assert params is not None assert params[13] == datetime(2023, 3, 1, tzinfo=UTC) assert params[14] == datetime(2023, 4, 1, tzinfo=UTC) @@ -678,7 +684,7 @@ def test_search_memories_by_time_accepts_an_explicit_proximity_pivot() -> None: window_center=pivot, ) - _query, params = cursor.executed[0] + _query, params = cursor.statements[0] assert params is not None assert params[17] == pivot @@ -700,7 +706,7 @@ def test_list_artifacts_applies_type_domain_sensitivity_and_limit_filters() -> N ) assert artifacts[0]["id"] == "artifact-1" - query, params = cursor.executed[0] + query, params = cursor.statements[0] assert "FROM generated_artifacts" in query assert "%s::text IS NULL OR artifact_type = %s" in query assert "domain = ANY" in query @@ -715,6 +721,7 @@ def test_list_artifacts_applies_type_domain_sensitivity_and_limit_filters() -> N ["public", "private"], None, None, + None, 5, ) @@ -754,13 +761,13 @@ def test_artifact_quality_ratings_insert_and_export_json_safe_payloads() -> None assert created["id"] == rating_id assert rows == [{"id": rating_id, "artifact_id": artifact_id, "usefulness": 5}] assert _event_log_insert_count(cursor) == 1 - assert "FOR UPDATE" in cursor.executed[0][0] - insert_query, insert_params = cursor.executed[1] + assert "FOR UPDATE" in cursor.statements[0][0] + insert_query, insert_params = cursor.statements[1] assert "INSERT INTO artifact_quality_ratings" in insert_query assert insert_params is not None assert isinstance(insert_params[-1], Jsonb) assert insert_params[-1].obj == {"prompt_hash": "sha256:test"} - list_query, list_params = cursor.executed[3] + list_query, list_params = cursor.statements[3] assert "FROM artifact_quality_ratings" in list_query assert list_params == (artifact_id, artifact_id, None, None, 10) @@ -788,7 +795,7 @@ def test_artifact_quality_ratings_upsert_on_artifact_reviewer_conflict() -> None ) assert created["id"] == rating_id - upsert_query, _upsert_params = cursor.executed[1] + upsert_query, _upsert_params = cursor.statements[1] assert "ON CONFLICT (artifact_id, reviewer_id) DO UPDATE SET" in upsert_query assert "usefulness = EXCLUDED.usefulness" in upsert_query assert "metadata_json = EXCLUDED.metadata_json" in upsert_query @@ -824,9 +831,9 @@ def test_quality_rating_rejects_exact_redacted_artifact_before_insert() -> None: with pytest.raises(ValueError, match="ratings cannot be added to a redacted artifact"): store.create_artifact_quality_rating({"artifact_id": artifact_id, "usefulness": 5}) - assert len(cursor.executed) == 1 - assert "FOR UPDATE" in cursor.executed[0][0] - assert not any("INSERT INTO artifact_quality_ratings" in query for query, _params in cursor.executed) + assert len(cursor.statements) == 1 + assert "FOR UPDATE" in cursor.statements[0][0] + assert not any("INSERT INTO artifact_quality_ratings" in query for query, _params in cursor.statements) def test_list_beliefs_joins_memory_domain_sensitivity_filters() -> None: @@ -854,7 +861,7 @@ def test_list_beliefs_joins_memory_domain_sensitivity_filters() -> None: ) assert beliefs[0]["id"] == belief_id - query, params = cursor.executed[0] + query, params = cursor.statements[0] assert "FROM beliefs b" in query assert "JOIN memories m" in query assert "%s::text IS NULL OR b.status = %s" in query @@ -890,9 +897,11 @@ def test_memory_revision_provenance_and_graph_methods_write_audit_events() -> No {"id": memory_id}, _event_row(memory_id), {"id": memory_id}, + {"id": memory_id}, # stored label read before any update # update_memory to a searchable status reads the stored row first, # for the credential activation check (S4.4 round 2, ruling C2). {"id": memory_id, "status": "candidate", "canonical_text": "Alice vNext is being built."}, + {"metadata_json": {}}, # metadata reread inside the label lock {"id": memory_id}, _event_row(memory_id), {"id": revision_id, "memory_id": memory_id}, @@ -962,7 +971,7 @@ def test_memory_revision_provenance_and_graph_methods_write_audit_events() -> No store.expire_edge(edge_id=edge_id) assert _event_log_insert_count(cursor) == 7 - memory_insert_query = cursor.executed[0][0] + memory_insert_query = cursor.statements[0][0] assert "INSERT INTO memories" in memory_insert_query assert "canonical_text" in memory_insert_query assert "domain" in memory_insert_query @@ -970,15 +979,15 @@ def test_memory_revision_provenance_and_graph_methods_write_audit_events() -> No assert "metadata_json" in memory_insert_query assert "ON CONFLICT DO NOTHING" in memory_insert_query assert "WHERE commit_digest IS NOT NULL" not in memory_insert_query - revision_queries = [query for query, _params in cursor.executed if "next_revision AS" in query] + revision_queries = [query for query, _params in cursor.statements if "next_revision AS" in query] assert len(revision_queries) == 1 assert "locked_memory AS" in revision_queries[0] assert "FOR UPDATE" in revision_queries[0] - assert any("INSERT INTO provenance_links" in query for query, _params in cursor.executed) - assert any("INSERT INTO graph_edges" in query for query, _params in cursor.executed) - assert any("UPDATE graph_edges" in query for query, _params in cursor.executed) - assert any("%s::text IS NULL OR from_id = %s" in query for query, _params in cursor.executed) - update_edge_query, update_edge_params = cursor.executed[-4] + assert any("INSERT INTO provenance_links" in query for query, _params in cursor.statements) + assert any("INSERT INTO graph_edges" in query for query, _params in cursor.statements) + assert any("UPDATE graph_edges" in query for query, _params in cursor.statements) + assert any("%s::text IS NULL OR from_id = %s" in query for query, _params in cursor.statements) + update_edge_query, update_edge_params = cursor.statements[-4] assert "metadata_json = metadata_json || %s" in update_edge_query assert update_edge_params is not None assert update_edge_params[1] == "accepted" @@ -1011,7 +1020,7 @@ def test_create_memory_persists_canonical_multi_project_scope_metadata() -> None ) assert row["project_scope"] == ["alicebot", "hermes"] - insert_params = cursor.executed[0][1] + insert_params = cursor.statements[0][1] assert insert_params is not None metadata_values = [param.obj for param in insert_params if isinstance(param, Jsonb)] assert {"project_scope": ["alicebot", "hermes"]} in metadata_values @@ -1023,7 +1032,7 @@ def test_get_memory_for_update_uses_a_row_lock() -> None: store = PostgresVNextStore(RecordingConnection(cursor)) assert store.get_memory_for_update(memory_id) == {"id": memory_id} - query, params = cursor.executed[0] + query, params = cursor.statements[0] assert "FROM memories" in query assert "FOR UPDATE" in query assert params == (memory_id,) @@ -1041,7 +1050,7 @@ def test_pending_derived_candidate_lookup_uses_snapshots_and_row_locks() -> None exclude_memory_id=excluded_id, ) == [{"id": candidate_id}] - query, params = cursor.executed[0] + query, params = cursor.statements[0] assert "status IN ('candidate', 'needs_review')" in query assert "member_snapshots" in query assert "jsonb_array_elements" in query @@ -1063,7 +1072,7 @@ def test_list_memories_pushes_scope_and_limit_into_postgres_query() -> None: == [] ) - query, params = cursor.executed[-1] + query, params = cursor.statements[-1] assert "status = %s" in query assert "domain = ANY(%s::text[]) OR domain = 'unknown'" in query assert "COALESCE(sensitivity, 'unknown') = ANY(%s::text[])" in query @@ -1128,12 +1137,12 @@ def test_resume_store_queries_apply_admission_predicates_before_limit() -> None: ) memory_query, loop_query, memory_event_query, loop_event_query, shared_event_query = ( - query for query, _params in cursor.executed + query for query, _params in cursor.statements ) - memory_params = cursor.executed[0][1] - loop_params = cursor.executed[1][1] - memory_event_params = cursor.executed[2][1] - loop_event_params = cursor.executed[3][1] + memory_params = cursor.statements[0][1] + loop_params = cursor.statements[1][1] + memory_event_params = cursor.statements[2][1] + loop_event_params = cursor.statements[3][1] assert "status = ANY(%s::text[])" in memory_query assert "memory_type = ANY(%s::text[])" in memory_query assert "created_at >= %s::timestamptz" in memory_query @@ -1222,8 +1231,8 @@ def test_project_update_event_lookup_is_one_bounded_target_and_payload_query() - ) assert rows == [] - assert len(cursor.executed) == 1 - query, params = cursor.executed[0] + assert len(cursor.statements) == 1 + query, params = cursor.statements[0] assert query.count("SELECT") == 5 assert query.count("user_id = app.current_user_id()") == 5 assert query.count("event_type IN (") == 5 @@ -1270,11 +1279,11 @@ def test_memory_and_rollup_counts_are_exact_scoped_database_reads() -> None: == 5 ) - memory_query, memory_params = cursor.executed[0] + memory_query, memory_params = cursor.statements[0] assert "SELECT COUNT(*) AS count" in memory_query assert "status = %s" in memory_query assert memory_params == ("active", ["project"], ["private"]) - rollup_query, rollup_params = cursor.executed[1] + rollup_query, rollup_params = cursor.statements[1] assert "SELECT COUNT(*) AS count" in rollup_query assert "status IN ('active', 'accepted')" in rollup_query assert "candidate_kind" in rollup_query @@ -1313,7 +1322,7 @@ def test_rollup_reads_push_status_scope_exact_keys_order_and_limits_into_postgre limit=99, ) - input_query, input_params = cursor.executed[0] + input_query, input_params = cursor.statements[0] assert "status IN ('active', 'accepted')" in input_query assert "COALESCE(metadata_json ->> 'candidate_kind', '') <> %s" in input_query assert "domain = ANY(%s::text[]) OR domain = 'unknown'" in input_query @@ -1330,7 +1339,7 @@ def test_rollup_reads_push_status_scope_exact_keys_order_and_limits_into_postgre 501, ) - pending_query, pending_params = cursor.executed[1] + pending_query, pending_params = cursor.statements[1] assert "DISTINCT ON (metadata_json ->> 'rollup_digest')" in pending_query assert "status = 'candidate'" in pending_query assert "metadata_json ->> 'rollup_digest' = ANY(%s::text[])" in pending_query @@ -1347,7 +1356,7 @@ def test_rollup_reads_push_status_scope_exact_keys_order_and_limits_into_postgre 2, ) - accepted_query, accepted_params = cursor.executed[2] + accepted_query, accepted_params = cursor.statements[2] assert "DISTINCT ON (metadata_json ->> 'rollup_key')" in accepted_query assert "status IN ('active', 'accepted')" in accepted_query assert "metadata_json ->> 'rollup_key' = ANY(%s::text[])" in accepted_query @@ -1370,7 +1379,7 @@ def test_rollup_reads_push_status_scope_exact_keys_order_and_limits_into_postgre excluded_candidate_kind="memory_rollup", limit=1, ) - assert cursor.executed[3][1] == ( + assert cursor.statements[3][1] == ( "memory_rollup", None, None, @@ -1439,10 +1448,10 @@ def test_project_people_belief_and_open_loop_methods_write_audit_events() -> Non store.update_open_loop(loop_id=loop_id, patch={"title": "Validate migration", "priority": "normal"}) assert _event_log_insert_count(cursor) == 9 - assert "INSERT INTO projects" in cursor.executed[0][0] - assert "FROM projects" in cursor.executed[3][0] - assert "%s::text IS NULL OR status = %s" in cursor.executed[3][0] - assert cursor.executed[3][1] == ( + assert "INSERT INTO projects" in cursor.statements[0][0] + assert "FROM projects" in cursor.statements[3][0] + assert "%s::text IS NULL OR status = %s" in cursor.statements[3][0] + assert cursor.statements[3][1] == ( "active", "active", ["project"], @@ -1455,14 +1464,14 @@ def test_project_people_belief_and_open_loop_methods_write_audit_events() -> Non None, 3, ) - assert "UPDATE projects" in cursor.executed[4][0] - assert "INSERT INTO people" in cursor.executed[6][0] - assert "UPDATE people" in cursor.executed[9][0] - assert "INSERT INTO beliefs" in cursor.executed[11][0] - assert "UPDATE beliefs" in cursor.executed[14][0] - assert "INSERT INTO open_loops" in cursor.executed[16][0] - assert "UPDATE open_loops" in cursor.executed[19][0] - assert "UPDATE open_loops" in cursor.executed[21][0] + assert "UPDATE projects" in cursor.statements[4][0] + assert "INSERT INTO people" in cursor.statements[6][0] + assert "UPDATE people" in cursor.statements[9][0] + assert "INSERT INTO beliefs" in cursor.statements[11][0] + assert "UPDATE beliefs" in cursor.statements[14][0] + assert "INSERT INTO open_loops" in cursor.statements[16][0] + assert "UPDATE open_loops" in cursor.statements[19][0] + assert "UPDATE open_loops" in cursor.statements[21][0] def test_get_artifact_for_update_locks_the_persisted_authorization_target() -> None: @@ -1473,7 +1482,7 @@ def test_get_artifact_for_update_locks_the_persisted_authorization_target() -> N artifact = store.get_artifact_for_update(artifact_id) assert artifact == {"id": artifact_id, "artifact_type": "daily_brief"} - query, params = cursor.executed[0] + query, params = cursor.statements[0] assert "FROM generated_artifacts" in query assert "FOR UPDATE" in query assert params == (artifact_id,) @@ -1499,7 +1508,7 @@ def test_exact_open_loop_and_artifact_digest_lookups_scope_before_limit() -> Non scope_projects=(project_id,), ) == {"id": "artifact-1"} - loop_query, loop_params = cursor.executed[0] + loop_query, loop_params = cursor.statements[0] assert loop_query.index("automation_digest") < loop_query.index("LIMIT 1") assert "project_id = %s::uuid" in loop_query assert "person_id = %s::uuid" in loop_query @@ -1510,7 +1519,7 @@ def test_exact_open_loop_and_artifact_digest_lookups_scope_before_limit() -> Non person_id, person_id, ) - artifact_query, artifact_params = cursor.executed[1] + artifact_query, artifact_params = cursor.statements[1] assert artifact_query.index("automation_digest") < artifact_query.index("LIMIT 1") assert "consolidation_digest" in artifact_query assert artifact_params == ( @@ -1540,13 +1549,13 @@ def test_source_trace_and_policy_telemetry_queries_filter_before_limit() -> None store.list_agent_policy_artifacts(agent_id="hermes", limit=15) store.list_agent_policy_memories(agent_id="hermes", limit=16) - for query, _params in cursor.executed[:4]: + for query, _params in cursor.statements[:4]: assert query.index("source_id") < query.index("LIMIT %s") - assert "provenance_links" in cursor.executed[0][0] - assert "provenance_links" in cursor.executed[1][0] - assert "target_type = 'source'" in cursor.executed[3][0] - assert "generated_by' = 'agent'" in cursor.executed[4][0] - assert "agent_id' IS NOT NULL" in cursor.executed[5][0] + assert "provenance_links" in cursor.statements[0][0] + assert "provenance_links" in cursor.statements[1][0] + assert "target_type = 'source'" in cursor.statements[3][0] + assert "generated_by' = 'agent'" in cursor.statements[4][0] + assert "agent_id' IS NOT NULL" in cursor.statements[5][0] def test_artifact_task_and_brain_charter_methods_write_audit_events() -> None: @@ -1608,12 +1617,12 @@ def test_artifact_task_and_brain_charter_methods_write_audit_events() -> None: assert claimed is not None assert _event_log_insert_count(cursor) == 6 - assert "INSERT INTO generated_artifacts" in cursor.executed[0][0] - assert "UPDATE generated_artifacts" in cursor.executed[3][0] - assert "INSERT INTO task_queue" in cursor.executed[5][0] - assert "FOR UPDATE SKIP LOCKED" in cursor.executed[7][0] - assert "UPDATE task_queue" in cursor.executed[9][0] - assert "ON CONFLICT (user_id)" in cursor.executed[11][0] + assert "INSERT INTO generated_artifacts" in cursor.statements[0][0] + assert "UPDATE generated_artifacts" in cursor.statements[3][0] + assert "INSERT INTO task_queue" in cursor.statements[5][0] + assert "FOR UPDATE SKIP LOCKED" in cursor.statements[7][0] + assert "UPDATE task_queue" in cursor.statements[9][0] + assert "ON CONFLICT (user_id)" in cursor.statements[11][0] def test_append_and_list_event_log_records_use_integrity_payload() -> None: @@ -1635,7 +1644,7 @@ def test_append_and_list_event_log_records_use_integrity_payload() -> None: assert appended["target_id"] == "memory-1" assert events[0]["target_id"] == "memory-1" assert all_events[0]["target_id"] == "memory-1" - event_insert_query, event_insert_params = cursor.executed[0] + event_insert_query, event_insert_params = cursor.statements[0] assert "INSERT INTO event_log" in event_insert_query assert event_insert_params is not None assert event_insert_params[1:6] == ( @@ -1648,7 +1657,7 @@ def test_append_and_list_event_log_records_use_integrity_payload() -> None: assert isinstance(event_insert_params[7], Jsonb) assert event_insert_params[7].obj == {"b": 2, "a": 1} assert event_insert_params[10] == event["integrity_hash"] - event_list_query = cursor.executed[1][0] + event_list_query = cursor.statements[1][0] assert "%s::text IS NULL OR target_type = %s" in event_list_query assert "%s::text IS NULL OR target_id = %s" in event_list_query @@ -1734,7 +1743,7 @@ def test_connector_settings_and_state_methods_use_dedicated_tables_and_audit_eve assert fetched_state is not None assert storage_status["connector_settings_exists"] is True assert _event_log_insert_count(cursor) == 2 - setting_query, setting_params = cursor.executed[0] + setting_query, setting_params = cursor.statements[0] assert "INSERT INTO connector_settings" in setting_query assert "ON CONFLICT (user_id, connector_name)" in setting_query assert setting_params is not None @@ -1742,7 +1751,7 @@ def test_connector_settings_and_state_methods_use_dedicated_tables_and_audit_eve assert setting_params[8].obj == [] assert isinstance(setting_params[9], Jsonb) assert setting_params[9].obj == {"config_json": {"allowed_origins": ["http://localhost:3000"]}} - state_query, state_params = cursor.executed[4] + state_query, state_params = cursor.statements[4] assert "INSERT INTO connector_state" in state_query assert "items_seen = connector_state.items_seen + EXCLUDED.items_seen" in state_query assert state_params is not None @@ -1769,10 +1778,10 @@ def test_workspace_list_methods_apply_bounded_filters() -> None: assert tasks[0]["id"] == "workspace-row-1" assert events[0]["id"] == "workspace-row-1" - source_query, source_params = cursor.executed[0] - people_query, people_params = cursor.executed[1] - task_query, task_params = cursor.executed[2] - event_query, event_params = cursor.executed[3] + source_query, source_params = cursor.statements[0] + people_query, people_params = cursor.statements[1] + task_query, task_params = cursor.statements[2] + event_query, event_params = cursor.statements[3] assert "FROM sources" in source_query assert "deleted_at IS NULL" in source_query assert source_params == (["project"], ["project"], ["private"], ["private"], 7) @@ -1821,7 +1830,7 @@ def test_jsonb_and_event_hash_normalize_postgres_scalar_values() -> None: "captured_at": "2026-05-10T12:30:00+00:00", }, } - project_insert_params = cursor.executed[0][1] + project_insert_params = cursor.statements[0][1] assert project_insert_params is not None assert isinstance(project_insert_params[-1], Jsonb) assert project_insert_params[-1].obj == { @@ -1845,7 +1854,7 @@ def test_fts_search_builds_websearch_tsquery_with_pushed_down_filters() -> None: ) assert rows[0]["id"] == "memory-1" - query, params = cursor.executed[0] + query, params = cursor.statements[0] assert "FROM memories" in query assert "websearch_to_tsquery('english', %s)" in query assert "search_tsv @@ websearch_to_tsquery('english', %s)" in query @@ -1911,7 +1920,7 @@ def test_fts_search_pushes_down_memory_type_project_agent_run_and_expiry_filters include_expired=True, ) - _query, params = cursor.executed[0] + _query, params = cursor.statements[0] assert params == ( "Alice provenance retrieval", ["project"], @@ -1959,7 +1968,7 @@ def test_fts_search_pushes_people_and_time_scope_before_ranked_limit() -> None: limit=1, ) - query, params = cursor.executed[0] + query, params = cursor.statements[0] assert "jsonb_path_query" in query assert "id::text = ANY" in query assert "COALESCE(valid_from, last_seen_at, updated_at, first_seen_at, created_at)" in query @@ -1991,7 +2000,7 @@ def test_search_source_chunks_builds_websearch_tsquery_over_chunk_text() -> None ) assert rows[0]["source_id"] == "source-1" - query, params = cursor.executed[0] + query, params = cursor.statements[0] assert "FROM source_chunks c" in query assert "JOIN sources s ON s.id = c.source_id AND s.user_id = c.user_id" in query assert "s.deleted_at IS NULL" in query @@ -2031,7 +2040,7 @@ def test_search_source_chunks_match_any_ors_sanitized_lexemes() -> None: match_any=True, ) - query, params = cursor.executed[0] + query, params = cursor.statements[0] assert "c.search_tsv @@ to_tsquery('english', %s)" in query # Stopwords and tsquery metacharacters are stripped; each surviving # token is individually quoted so nothing can inject query syntax. @@ -2046,7 +2055,7 @@ def test_search_source_chunks_match_any_returns_empty_without_content_tokens() - # Stopword/metacharacter-only queries sanitize to no lexemes: no SQL runs. assert store.search_source_chunks(query="&|!():*<->", match_any=True) == [] assert store.search_source_chunks(query="when was the", match_any=True) == [] - assert cursor.executed == [] + assert cursor.statements == [] def test_vector_search_orders_by_cosine_distance_and_skips_null_embeddings() -> None: @@ -2064,7 +2073,7 @@ def test_vector_search_orders_by_cosine_distance_and_skips_null_embeddings() -> ) assert rows[0]["id"] == "memory-1" - query, params = next((q, p) for q, p in cursor.executed if "vector_distance" in q) + query, params = next((q, p) for q, p in cursor.statements if "vector_distance" in q) assert "FROM memories" in query assert "embedding_vector IS NOT NULL" in query assert "status IN ('active', 'accepted')" in query @@ -2119,7 +2128,7 @@ def test_postgres_vector_boundary_rejects_non_finite_values() -> None: vector=[1.0, float("inf")], ) - assert cursor.executed == [] + assert cursor.statements == [] def test_vector_search_can_require_matching_embedding_signature() -> None: @@ -2133,7 +2142,7 @@ def test_vector_search_can_require_matching_embedding_signature() -> None: embedding_signature_version=1, ) - query, params = next((q, p) for q, p in cursor.executed if "vector_distance" in q) + query, params = next((q, p) for q, p in cursor.statements if "vector_distance" in q) assert "metadata_json -> '_alice_embedding' ->> 'provider' = %s" in query assert "metadata_json -> '_alice_embedding' ->> 'model' = %s" in query assert "->> 'version' = %s" in query @@ -2149,7 +2158,7 @@ def test_vector_search_enables_iterative_hnsw_scan() -> None: store.search_memories_vector(query_vector=[1.0, 0.0], limit=20) - statements = [q for q, _ in cursor.executed] + statements = [q for q, _ in cursor.statements] assert any("hnsw.iterative_scan" in q and "strict_order" in q for q in statements), statements # The iterative-scan setting must precede the vector SELECT. set_index = next(i for i, q in enumerate(statements) if "hnsw.iterative_scan" in q) @@ -2192,8 +2201,8 @@ def test_vector_search_discards_stale_content_signatures_after_database_read() - ) assert [row["id"] for row in rows] == ["memory-current"] - _query, select_params = next((q, p) for q, p in cursor.executed if "vector_distance" in q) - vector_query = next(q for q, _p in cursor.executed if "vector_distance" in q) + _query, select_params = next((q, p) for q, p in cursor.statements if "vector_distance" in q) + vector_query = next(q for q, _p in cursor.statements if "vector_distance" in q) assert "content_sha256" in vector_query assert "digest(" in vector_query assert select_params[-1] == 4 @@ -2205,7 +2214,7 @@ def test_scheduler_lock_key_includes_current_rls_user() -> None: assert store.try_scheduler_workflow_lock("daily_brief") is True - query, params = cursor.executed[0] + query, params = cursor.statements[0] assert "app.current_user_id()::text" in query assert "concat_ws" in query assert params == ("daily_brief",) @@ -2245,7 +2254,7 @@ def test_scheduler_workflow_updates_only_preserve_claim_for_run_bookkeeping( actor_type="test", ) - query, params = cursor.executed[0] + query, params = cursor.statements[0] assert "claim_token = CASE WHEN %s THEN claim_token ELSE NULL END" in query assert "claim_version = claim_version + CASE WHEN %s THEN 0 ELSE 1 END" in query assert params is not None @@ -2275,7 +2284,7 @@ def test_artifact_status_update_uses_expected_status_compare_and_set() -> None: ) assert row is not None - query, params = cursor.executed[0] + query, params = cursor.statements[0] assert "AND (%s::text IS NULL OR status = %s)" in query assert "metadata_json || %s::jsonb" in query assert params is not None @@ -2323,7 +2332,7 @@ def test_scheduler_claim_rechecks_due_state_and_persists_fence() -> None: assert claim is not None assert claim["claim_version"] == 1 assert claim["scheduled_for"] == scheduled_for - statements = [query for query, _params in cursor.executed] + statements = [query for query, _params in cursor.statements] assert "FOR UPDATE SKIP LOCKED" in statements[0] assert "enabled = true" in statements[2] assert "claim_version = claim_version + 1" in statements[3] @@ -2388,10 +2397,10 @@ def test_scheduler_heartbeat_finalize_and_reaper_are_fenced() -> None: assert finalized is not None assert reaped[0]["status"] == "failed" - heartbeat_query = cursor.executed[0][0] - publish_lock_query = cursor.executed[1][0] - finalize_query = cursor.executed[2][0] - reap_query = cursor.executed[4][0] + heartbeat_query = cursor.statements[0][0] + publish_lock_query = cursor.statements[1][0] + finalize_query = cursor.statements[2][0] + reap_query = cursor.statements[4][0] for query in (heartbeat_query, publish_lock_query, finalize_query): assert "claim_token = %s" in query assert "claim_version = %s" in query @@ -2412,7 +2421,7 @@ def test_pending_confirmation_query_enforces_all_actionable_invariants_before_li store.list_pending_inline_confirmations(limit=3) - query, params = cursor.executed[0] + query, params = cursor.statements[0] assert "status = 'needs_review'" in query assert "confirmation_status = 'unconfirmed'" in query assert "confirmation,status" in query @@ -2435,10 +2444,10 @@ def test_update_memory_embedding_and_missing_embedding_listing() -> None: assert updated == {"id": memory_id} assert missing[0]["id"] == memory_id - update_query, update_params = cursor.executed[0] + update_query, update_params = cursor.statements[0] assert "SET embedding_vector = %s::vector" in update_query assert update_params == ("[1.0,0.5]", memory_id) - missing_query, missing_params = cursor.executed[1] + missing_query, missing_params = cursor.statements[1] assert "embedding_vector IS NULL" in missing_query assert "%s::uuid IS NULL OR id > %s::uuid" in missing_query assert "ORDER BY id ASC" in missing_query @@ -2464,7 +2473,7 @@ def test_signed_embedding_update_compares_current_memory_content_digest() -> Non is None ) - query, params = cursor.executed[0] + query, params = cursor.statements[0] assert "digest(" in query assert "NULLIF(btrim(title, chr(9)" in query assert "[[:space:]]" not in query @@ -2513,7 +2522,7 @@ def test_embedding_digest_sql_uses_exact_python_strip_table_at_every_cas_boundar embedding_model="embed-v1", embedding_signature_version=2, ) - vector_query = next(query for query, _params in vector_cursor.executed if "vector_distance" in query) + vector_query = next(query for query, _params in vector_cursor.statements if "vector_distance" in query) update_cursor = RecordingCursor(fetchone_results=[]) PostgresVNextStore(RecordingConnection(update_cursor)).update_memory_embedding( @@ -2525,7 +2534,7 @@ def test_embedding_digest_sql_uses_exact_python_strip_table_at_every_cas_boundar content_sha256="a" * 64, signature_version=2, ) - update_query = update_cursor.executed[0][0] + update_query = update_cursor.statements[0][0] missing_cursor = RecordingCursor(fetchone_results=[], fetchall_result=[]) PostgresVNextStore(RecordingConnection(missing_cursor)).list_memories_missing_embeddings( @@ -2534,7 +2543,7 @@ def test_embedding_digest_sql_uses_exact_python_strip_table_at_every_cas_boundar embedding_model="embed-v1", embedding_signature_version=2, ) - missing_query = missing_cursor.executed[0][0] + missing_query = missing_cursor.statements[0][0] for query in (vector_query, update_query, missing_query): assert "[[:space:]]" not in query @@ -2560,7 +2569,7 @@ def test_embedding_backfill_includes_unsigned_or_incompatible_vectors() -> None: embedding_signature_version=1, ) - query, params = cursor.executed[0] + query, params = cursor.statements[0] assert "embedding_vector IS NULL" in query assert "IS DISTINCT FROM %s" in query assert "content_sha256" in query @@ -2576,7 +2585,7 @@ def test_clear_memory_embedding_removes_signature_metadata() -> None: store = PostgresVNextStore(RecordingConnection(cursor)) assert store.clear_memory_embedding(memory_id=memory_id) == {"id": memory_id} - query, _params = cursor.executed[0] + query, _params = cursor.statements[0] assert "embedding_vector = NULL" in query assert "metadata_json = metadata_json - '_alice_embedding'" in query assert "{EMBEDDING_SIGNATURE_METADATA_KEY}" not in query @@ -2599,15 +2608,15 @@ def test_targeted_memory_lookups_use_indexed_columns() -> None: assert by_digest == {"id": "memory-digest"} assert by_confirmation == {"id": "memory-confirmation"} assert latest == {"id": "memory-latest"} - digest_query, digest_params = cursor.executed[0] + digest_query, digest_params = cursor.statements[0] assert "WHERE commit_digest = %s" in digest_query assert "LIMIT 1" in digest_query assert digest_params == ("digest-1",) - confirmation_query, confirmation_params = cursor.executed[1] + confirmation_query, confirmation_params = cursor.statements[1] assert "WHERE confirmation_id = %s" in confirmation_query assert "LIMIT 1" in confirmation_query assert confirmation_params == ("confirm-1",) - latest_query, latest_params = cursor.executed[2] + latest_query, latest_params = cursor.statements[2] assert "metadata_json #>> '{agentic_memory,kind}' = 'agentic_memory_commit'" in latest_query assert "status = 'active'" in latest_query assert "metadata_json #>> '{agentic_memory,agent_identity,agent_id}' = %s" in latest_query @@ -2635,7 +2644,7 @@ def test_create_memory_persists_commit_digest_and_confirmation_id_columns() -> N } ) - insert_query, insert_params = cursor.executed[0] + insert_query, insert_params = cursor.statements[0] assert "commit_digest" in insert_query assert "confirmation_id" in insert_query assert insert_params is not None @@ -2665,7 +2674,7 @@ def test_create_memory_persists_first_class_scope_columns() -> None: } ) - insert_query, insert_params = cursor.executed[0] + insert_query, insert_params = cursor.statements[0] assert "project_id" in insert_query assert "created_by_agent_id" in insert_query assert "run_id" in insert_query @@ -2702,7 +2711,7 @@ def test_create_agent_api_key_persists_project_scope_binding() -> None: ) assert row["project_scope"] == "alicebot" - insert_query, insert_params = cursor.executed[0] + insert_query, insert_params = cursor.statements[0] assert "INSERT INTO agent_api_keys" in insert_query assert "project_scope" in insert_query assert insert_params is not None @@ -2729,7 +2738,7 @@ def test_create_agent_api_key_persists_project_scope_binding() -> None: } ) assert unbound["project_scope"] is None - assert cursor.executed[0][1][3] is None + assert cursor.statements[0][1][3] is None # -- temporal slice: edge event time, as-of reads, supersession pointers ------- @@ -2758,7 +2767,7 @@ def test_create_edge_populates_observed_at_and_defaults_valid_from_to_event_time } ) - insert_query, insert_params = cursor.executed[0] + insert_query, insert_params = cursor.statements[0] assert "observed_at" in insert_query # observed_at defaults to write time; valid_from defaults to observed_at # (then write time), so the validity interval starts at event time. @@ -2797,7 +2806,7 @@ def test_create_edge_without_event_time_notes_the_write_time_fallback_in_metadat } ) - _insert_query, insert_params = cursor.executed[0] + _insert_query, insert_params = cursor.statements[0] assert insert_params is not None metadata_param = insert_params[-1] assert isinstance(metadata_param, Jsonb) @@ -2831,7 +2840,7 @@ def test_edge_digest_upsert_creates_once_and_replays_without_a_second_event() -> assert created["id"] == replayed["id"] == edge_id insert_query, insert_params = next( - (query, params) for query, params in cursor.executed if "INSERT INTO graph_edges" in query + (query, params) for query, params in cursor.statements if "INSERT INTO graph_edges" in query ) assert "ON CONFLICT DO NOTHING" in insert_query assert insert_params is not None @@ -2839,7 +2848,7 @@ def test_edge_digest_upsert_creates_once_and_replays_without_a_second_event() -> assert isinstance(metadata_param, Jsonb) assert metadata_param.obj["idempotency_digest"] == "edge-digest" assert _event_log_insert_count(cursor) == 1 - assert sum("INSERT INTO graph_edges" in query for query, _params in cursor.executed) == 1 + assert sum("INSERT INTO graph_edges" in query for query, _params in cursor.statements) == 1 def test_list_edges_as_of_filters_on_the_validity_interval_with_limit() -> None: @@ -2848,7 +2857,7 @@ def test_list_edges_as_of_filters_on_the_validity_interval_with_limit() -> None: store.list_edges_as_of("2026-07-01T00:00:00Z", limit=5) - query, params = cursor.executed[0] + query, params = cursor.statements[0] assert "FROM graph_edges" in query # Half-open interval: valid_from <= at < valid_to; NULL valid_from # (pre-slice edges with unrecorded event time) never matches. @@ -2867,6 +2876,7 @@ def test_memory_writes_accept_supersession_pointer_columns() -> None: fetchone_results=[ {"id": memory_id}, _event_row(memory_id), + {"id": memory_id}, # prior label read {"id": memory_id}, _event_row(memory_id), ] @@ -2886,12 +2896,12 @@ def test_memory_writes_accept_supersession_pointer_columns() -> None: patch={"status": "superseded", "superseded_by": successor_id}, ) - insert_query, insert_params = cursor.executed[0] + insert_query, insert_params = cursor.statements[0] assert "supersedes" in insert_query assert insert_params is not None assert insert_params[-2:] == (None, predecessor_id) # (superseded_by, supersedes) - update_query, update_params = cursor.executed[2] + update_query, update_params = next((query, params) for query, params in cursor.statements if "UPDATE memories" in query) assert "superseded_by = COALESCE(%s::uuid, superseded_by)" in update_query assert "supersedes = COALESCE(%s::uuid, supersedes)" in update_query assert update_params is not None @@ -2902,6 +2912,8 @@ def test_update_memory_reassigns_first_class_and_canonical_project_scope_togethe memory_id = str(uuid4()) cursor = RecordingCursor( fetchone_results=[ + {"id": memory_id, "metadata_json": {"project_scope": ["project-old"]}}, + {"metadata_json": {"project_scope": ["project-old"]}}, { "id": memory_id, "memory_key": "project.release.scope", @@ -2919,6 +2931,7 @@ def test_update_memory_reassigns_first_class_and_canonical_project_scope_togethe row = store.update_memory( memory_id=memory_id, + label_write=True, patch={ "project_id": "project-new", "metadata_json": { @@ -2928,7 +2941,7 @@ def test_update_memory_reassigns_first_class_and_canonical_project_scope_togethe }, ) - query, params = cursor.executed[0] + query, params = next((query, params) for query, params in cursor.statements if "UPDATE memories" in query) assert "project_id = COALESCE(%s, project_id)" in query assert params is not None metadata_param = next(param for param in params if isinstance(param, Jsonb)) @@ -2976,7 +2989,7 @@ def test_entity_crud_methods_normalize_names_and_write_audit_events() -> None: assert updated["id"] == entity_id assert _event_log_insert_count(cursor) == 2 - insert_query, insert_params = cursor.executed[0] + insert_query, insert_params = cursor.statements[0] assert "INSERT INTO vnext_entities" in insert_query assert "app.current_user_id()" in insert_query assert insert_params is not None @@ -2990,24 +3003,24 @@ def test_entity_crud_methods_normalize_names_and_write_audit_events() -> None: assert insert_params[5].obj == {"hq": "sf"} assert insert_params[8] == 0 # mention_count defaults to zero - get_query, get_params = cursor.executed[2] + get_query, get_params = cursor.statements[2] assert "FROM vnext_entities" in get_query assert "WHERE id = %s::uuid" in get_query assert "deleted_at IS NULL" in get_query assert get_params == (entity_id,) - by_name_query, by_name_params = cursor.executed[3] + by_name_query, by_name_params = cursor.statements[3] assert "entity_type = %s" in by_name_query assert "normalized_name = %s" in by_name_query assert "LIMIT 1" in by_name_query assert by_name_params == ("organization", "openai inc") - list_query, list_params = cursor.executed[4] + list_query, list_params = cursor.statements[4] assert "%s::text IS NULL OR entity_type = %s" in list_query assert "ORDER BY updated_at DESC, created_at DESC, id DESC" in list_query assert list_params == ("organization", "organization", 7) - update_query, update_params = cursor.executed[5] + update_query, update_params = cursor.statements[5] assert "UPDATE vnext_entities" in update_query assert "name = COALESCE(%s, name)" in update_query assert "aliases = COALESCE(%s, aliases)" in update_query @@ -3031,7 +3044,7 @@ def test_update_entity_rejects_immutable_patch_fields_before_touching_sql() -> N ): with pytest.raises(ContinuityStoreInvariantError, match="immutable"): store.update_entity(entity_id=str(uuid4()), patch=immutable_patch) - assert cursor.executed == [] + assert cursor.statements == [] def test_find_entities_by_names_matches_normalized_names_and_aliases_in_one_query() -> None: @@ -3044,8 +3057,8 @@ def test_find_entities_by_names_matches_normalized_names_and_aliases_in_one_quer rows = store.find_entities_by_names(("openai", "northwind.example")) assert rows[0]["id"] == "entity-1" - assert len(cursor.executed) == 1 # one round trip covers both match paths - query, params = cursor.executed[0] + assert len(cursor.statements) == 1 # one round trip covers both match paths + query, params = cursor.statements[0] assert "FROM vnext_entities" in query assert "normalized_name = ANY(%s::text[])" in query assert "aliases ?| %s::text[]" in query @@ -3055,7 +3068,7 @@ def test_find_entities_by_names_matches_normalized_names_and_aliases_in_one_quer # An empty name tuple short-circuits without touching the database. assert store.find_entities_by_names(()) == [] - assert len(cursor.executed) == 1 + assert len(cursor.statements) == 1 def test_record_entity_mention_increments_count_and_widens_window() -> None: @@ -3072,7 +3085,7 @@ def test_record_entity_mention_increments_count_and_widens_window() -> None: assert row["id"] == entity_id assert _event_log_insert_count(cursor) == 1 - query, params = cursor.executed[0] + query, params = cursor.statements[0] assert "mention_count = mention_count + 1" in query assert "LEAST(COALESCE(first_observed_at, %s::timestamptz), %s::timestamptz)" in query assert "GREATEST(COALESCE(last_observed_at, %s::timestamptz), %s::timestamptz)" in query @@ -3090,7 +3103,7 @@ def test_record_entity_mention_increments_count_and_widens_window() -> None: fresh_store = PostgresVNextStore(RecordingConnection(fresh_cursor)) with pytest.raises(ContinuityStoreInvariantError, match="observed_at"): fresh_store.record_entity_mention(entity_id=entity_id, observed_at=None) - assert fresh_cursor.executed == [] + assert fresh_cursor.statements == [] def test_record_relationship_change_appends_history_and_updates_current_pointer() -> None: @@ -3118,12 +3131,12 @@ def test_record_relationship_change_appends_history_and_updates_current_pointer( assert row["id"] == event_id assert _event_log_insert_count(cursor) == 1 - before_query, before_params = cursor.executed[0] + before_query, before_params = cursor.statements[0] assert "metadata_json ->> 'relationship_type'" in before_query assert "deleted_at IS NULL" in before_query assert before_params == (entity_id,) - insert_query, insert_params = cursor.executed[1] + insert_query, insert_params = cursor.statements[1] assert "INSERT INTO entity_relationship_events" in insert_query assert "app.current_user_id()" in insert_query assert insert_params is not None @@ -3135,7 +3148,7 @@ def test_record_relationship_change_appends_history_and_updates_current_pointer( assert isinstance(insert_params[5], Jsonb) assert insert_params[5].obj == {"round": "seed"} - pointer_query, pointer_params = cursor.executed[2] + pointer_query, pointer_params = cursor.statements[2] assert "UPDATE vnext_entities" in pointer_query assert "metadata_json = metadata_json || %s" in pointer_query assert pointer_params is not None @@ -3143,7 +3156,7 @@ def test_record_relationship_change_appends_history_and_updates_current_pointer( assert pointer_params[0].obj == {"relationship_type": "investor"} assert pointer_params[1] == entity_id - event_query, event_params = cursor.executed[3] + event_query, event_params = cursor.statements[3] assert "INSERT INTO event_log" in event_query assert event_params is not None assert isinstance(event_params[7], Jsonb) @@ -3156,7 +3169,7 @@ def test_record_relationship_change_appends_history_and_updates_current_pointer( missing_store = PostgresVNextStore(RecordingConnection(missing_cursor)) with pytest.raises(ContinuityStoreInvariantError, match="existing entity"): missing_store.record_relationship_change(entity_id=entity_id, relationship_type="advisor") - assert len(missing_cursor.executed) == 1 + assert len(missing_cursor.statements) == 1 def test_list_relationship_events_reads_history_most_recent_first() -> None: @@ -3170,7 +3183,7 @@ def test_list_relationship_events_reads_history_most_recent_first() -> None: rows = store.list_relationship_events(entity_id) assert rows[0]["id"] == "event-1" - query, params = cursor.executed[0] + query, params = cursor.statements[0] assert "FROM entity_relationship_events" in query assert "WHERE entity_id = %s::uuid" in query assert "ORDER BY changed_at DESC, id DESC" in query @@ -3194,7 +3207,7 @@ def execute(self, query: str, params: tuple[object, ...] | None = None) -> None: def _redaction_flag_statements(cursor: RecordingCursor) -> list[str]: - return [query for query, _params in cursor.executed if "app.redaction_in_progress" in query] + return [query for query, _params in cursor.statements if "app.redaction_in_progress" in query] def test_redaction_marker_constant() -> None: @@ -3288,9 +3301,9 @@ def test_quoted_provenance_rejects_exact_redacted_target_before_insert(target_ty } ) - assert len(cursor.executed) == 1 - assert "FOR UPDATE" in cursor.executed[0][0] - assert not any("INSERT INTO provenance_links" in query for query, _params in cursor.executed) + assert len(cursor.statements) == 1 + assert "FOR UPDATE" in cursor.statements[0][0] + assert not any("INSERT INTO provenance_links" in query for query, _params in cursor.statements) @pytest.mark.parametrize( @@ -3340,8 +3353,8 @@ def test_redact_memory_bundle_rejects_malformed_terminal_artifact_provenance( ], ) - assert len(cursor.executed) == 2 - assert "FROM event_log" in cursor.executed[1][0] + assert len(cursor.statements) == 2 + assert "FROM event_log" in cursor.statements[1][0] assert _redaction_flag_statements(cursor) == [] @@ -3399,7 +3412,7 @@ def now(cls, tz=None): update_query, update_params = next( (query, params) - for query, params in cursor.executed + for query, params in cursor.statements if "UPDATE memories" in query and "embedding_vector = NULL" in query ) assert update_params is not None @@ -3409,7 +3422,7 @@ def now(cls, tz=None): event_update_query = next( query - for query, _params in cursor.executed + for query, _params in cursor.statements if "UPDATE event_log" in query and "jsonb_build_object" in query ) # PostgreSQL cannot infer a type for a bare bind used only as a @@ -3434,14 +3447,14 @@ def test_redact_memory_content_wraps_marker_update_in_redaction_mode() -> None: row = store.redact_memory_content(memory_id=memory_id) assert row["id"] == memory_id - queries = [query for query, _params in cursor.executed] + queries = [query for query, _params in cursor.statements] assert "SELECT metadata_json" in queries[0] assert "set_config('app.redaction_in_progress', 'on', false)" in queries[1] assert "UPDATE memories" in queries[2] assert "set_config('app.redaction_in_progress', 'off', false)" in queries[3] assert "INSERT INTO event_log" in queries[4] - update_query, update_params = cursor.executed[2] + update_query, update_params = cursor.statements[2] # Content columns become the marker; skeleton and scope survive. assert "CASE WHEN title IS NULL THEN NULL ELSE %s END" in update_query assert "canonical_text = %s" in update_query @@ -3464,7 +3477,7 @@ def test_redact_memory_content_wraps_marker_update_in_redaction_mode() -> None: assert "note" not in scrubbed assert update_params[6] == memory_id - event_query, event_params = cursor.executed[4] + event_query, event_params = cursor.statements[4] assert event_params is not None assert event_params[1] == "memory.redacted" payload = next(param for param in event_params if isinstance(param, Jsonb)) @@ -3496,7 +3509,7 @@ def test_redact_memory_revisions_scrubs_content_columns_only() -> None: assert result == {"memory_id": memory_id, "redacted_revisions": 2} update_query, update_params = next( - (query, params) for query, params in cursor.executed if "UPDATE memory_revisions" in query + (query, params) for query, params in cursor.statements if "UPDATE memory_revisions" in query ) # NULL content stays NULL; non-NULL content becomes the marker shape. assert "CASE WHEN previous_value IS NULL THEN NULL ELSE %s END" in update_query @@ -3518,7 +3531,7 @@ def test_redact_memory_revisions_scrubs_content_columns_only() -> None: flags = _redaction_flag_statements(cursor) assert "'on'" in flags[0] and "'off'" in flags[1] - event_query, event_params = cursor.executed[-1] + event_query, event_params = cursor.statements[-1] assert "INSERT INTO event_log" in event_query assert event_params is not None assert event_params[1] == "memory.redacted" @@ -3538,7 +3551,7 @@ def test_redact_memory_events_scrubs_payloads_and_clears_integrity_hash() -> Non assert result == {"memory_id": memory_id, "redacted_events": 1} update_query, update_params = next( - (query, params) for query, params in cursor.executed if "UPDATE event_log" in query + (query, params) for query, params in cursor.statements if "UPDATE event_log" in query ) assert "jsonb_build_object" in update_query assert "'redacted', true" in update_query @@ -3557,7 +3570,7 @@ def test_redact_memory_events_scrubs_payloads_and_clears_integrity_hash() -> Non flags = _redaction_flag_statements(cursor) assert len(flags) == 2 and "'on'" in flags[0] and "'off'" in flags[1] - event_query, event_params = cursor.executed[-1] + event_query, event_params = cursor.statements[-1] assert "INSERT INTO event_log" in event_query assert event_params is not None assert event_params[1] == "memory.redacted" @@ -3583,5 +3596,5 @@ def test_redaction_mode_resets_even_when_the_update_fails() -> None: assert "'off'" in flags[1] # The reset is the last statement issued; no event is appended after # a failed redaction. - assert "app.redaction_in_progress" in cursor.executed[-1][0] - assert not any("INSERT INTO event_log" in query for query, _params in cursor.executed) + assert "app.redaction_in_progress" in cursor.statements[-1][0] + assert not any("INSERT INTO event_log" in query for query, _params in cursor.statements) From 67661197fd557595399beee8ba71a03b7e3ebac7 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 21:28:01 +0200 Subject: [PATCH 34/90] test: make pointer fence fixture relabels explicit --- tests/unit/test_correction_label_respects_the_read_fence.py | 2 +- tests/unit/test_memory_id_pointer_residue.py | 4 ++-- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/tests/unit/test_correction_label_respects_the_read_fence.py b/tests/unit/test_correction_label_respects_the_read_fence.py index 4b659e232..b7ec96701 100644 --- a/tests/unit/test_correction_label_respects_the_read_fence.py +++ b/tests/unit/test_correction_label_respects_the_read_fence.py @@ -100,7 +100,7 @@ def _supersede(context, memory_id: str, text: str = NEW) -> str: def _set(context, memory_id: str, **patch) -> None: - _store(context, lambda s: s.update_memory(memory_id=memory_id, patch=patch, actor_type="user")) + _store(context, lambda s: s.update_memory(memory_id=memory_id, patch=patch, actor_type="user", label_write=True)) def _unquoted(value: object) -> str: diff --git a/tests/unit/test_memory_id_pointer_residue.py b/tests/unit/test_memory_id_pointer_residue.py index 20db1675f..443a65d49 100644 --- a/tests/unit/test_memory_id_pointer_residue.py +++ b/tests/unit/test_memory_id_pointer_residue.py @@ -109,7 +109,7 @@ def _supersede(context, memory_id: str, text: str = NEW) -> str: def _set(context, memory_id: str, **patch) -> None: - _store(context, lambda s: s.update_memory(memory_id=memory_id, patch=patch, actor_type="user")) + _store(context, lambda s: s.update_memory(memory_id=memory_id, patch=patch, actor_type="user", label_write=True)) def _add_metadata(context, memory_id: str, **keys) -> None: @@ -117,7 +117,7 @@ def merge(store): row = store.get_memory(memory_id) metadata = dict(row["metadata_json"]) metadata.update(keys) - store.update_memory(memory_id=memory_id, patch={"metadata_json": metadata}, actor_type="user") + store.update_memory(memory_id=memory_id, patch={"metadata_json": metadata}, actor_type="user", label_write=True) _store(context, merge) From 5d98fb5183f2bd9697266554c7cc7642c3dd0253 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 21:28:01 +0200 Subject: [PATCH 35/90] test: separate original quote projection from derived row denial --- ...st_saved_quotes_follow_the_source_fence.py | 31 +++++++++++++++++++ 1 file changed, 31 insertions(+) diff --git a/tests/unit/test_saved_quotes_follow_the_source_fence.py b/tests/unit/test_saved_quotes_follow_the_source_fence.py index 485a60646..c1d8a66b5 100644 --- a/tests/unit/test_saved_quotes_follow_the_source_fence.py +++ b/tests/unit/test_saved_quotes_follow_the_source_fence.py @@ -204,6 +204,21 @@ def _candidate(self, text: str) -> str: rows = self.sql("SELECT id, canonical_text FROM memories WHERE status = 'candidate' ORDER BY created_at DESC") return str(next(row["id"] for row in rows if text.split(":")[0] in str(row["canonical_text"]))) + def _original_quote_fixture(self, memory_id: str) -> None: + """Keep every saved quote/link while isolating original-row quote projection. + + Captured copies now inherit source labels and can be refused as a whole. + These older projection tests need a readable original row with saved provenance. + """ + path = _sqlite_path_from_url(self.context.database_url) + with sqlite_user_connection(path, _USER_ID) as conn: + row = SQLiteVNextStore(conn, _USER_ID).get_memory(memory_id) + metadata = dict(row["metadata_json"]) + metadata.pop("source_id", None) + metadata.pop("derived_from", None) + metadata["project_floor"] = [] + conn.execute("UPDATE memories SET metadata_json = ? WHERE id = ? AND user_id = ?", (json.dumps(metadata), memory_id, _USER_ID)) + def edit_and_approve(self, source_id: str, tag: str = "") -> tuple[str, str]: """``metadata_json.provenance`` and the link quote. Returns the memory id and a query that finds it.""" @@ -219,6 +234,7 @@ def edit_and_approve(self, source_id: str, tag: str = "") -> tuple[str, str]: who=self.reviewer, ) assert done["is_error"] is False, done + self._original_quote_fixture(candidate) return candidate, "kiln schedule Mondays" def supersede(self, source_id: str, tag: str = "") -> tuple[str, str]: @@ -239,6 +255,7 @@ def supersede(self, source_id: str, tag: str = "") -> tuple[str, str]: ) assert done["is_error"] is False, done replacement = done["payload"]["replacement_object"] # type: ignore[index] + self._original_quote_fixture(str(replacement["id"])) return str(replacement["id"]), "glaze shelf reorganised Saturday" def http_commit(self, source_id: str, tag: str = "") -> tuple[str, str]: @@ -1165,3 +1182,17 @@ def test_a_source_with_no_project_is_outside_the_fence_of_a_key_bound_to_a_proje assert (vault.explain(who, candidate)["is_error"] is False) is readable, ("explain agrees", who) + + +def test_current_derived_copy_is_denied_as_a_whole_after_source_relabel(vault: _Vault) -> None: + source_id = vault.capture_source() + memory_id, _query = vault.edit_and_approve(source_id) + row = vault.sql("SELECT metadata_json FROM memories WHERE id = ?", (memory_id,))[0] + metadata = json.loads(row["metadata_json"]) + metadata["source_id"] = source_id + vault.sql("UPDATE memories SET metadata_json = ? WHERE id = ?", (json.dumps(metadata), memory_id)) + vault.reclassify(source_id, "confidential") + result = vault.review("trusted", memory_id) + assert result["is_error"] is True + assert _WORD_A not in json.dumps(result, default=str) + assert vault.review("admin", memory_id)["is_error"] is False From 4331fb9e8c4a933a2391e14f0b5a92e594cab826 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 21:28:01 +0200 Subject: [PATCH 36/90] test: align route adapters with label closure and exact derived policy --- tests/unit/test_vnext_main.py | 37 +++++++++++++++++++++++++++++++++-- 1 file changed, 35 insertions(+), 2 deletions(-) diff --git a/tests/unit/test_vnext_main.py b/tests/unit/test_vnext_main.py index 8af8173c1..997968e0c 100644 --- a/tests/unit/test_vnext_main.py +++ b/tests/unit/test_vnext_main.py @@ -55,6 +55,25 @@ def __init__(self, _conn) -> None: self.browser_clip_capabilities: dict[str, dict[str, object]] = {} self.revisions: list[dict[str, object]] = [] + def lock_graph_mutation(self) -> None: + return None + + def lock_label_writes(self, *, exclusive: bool = False) -> None: + return None + + def read_label_rows(self, kind: str, ids: list[str]) -> list[dict[str, object]]: + collection = {"source": self.sources.values(), "memory": self.memories, "open_loop": self.open_loops, + "artifact": self.artifacts.values(), "belief": self.beliefs.values(), "project": self.projects.values()}.get(kind, []) + return [dict(row) for row in collection if str(row.get("id")) in ids] + + def _fetch_all(self, query: str, _params: tuple[object, ...]) -> list[dict[str, object]]: + # The label dependant walker performs its exact canonical reference filter after this prefilter. + for table, kind in (("memories", "memory"), ("open_loops", "open_loop"), ("generated_artifacts", "artifact"), ("projects", "project")): + if f"FROM {table}" in query: + collection = {"memory": self.memories, "open_loop": self.open_loops, "artifact": self.artifacts.values(), "project": self.projects.values()}[kind] + return [{**row, "kind": kind} for row in collection] + raise AssertionError(query) + def create_browser_clip_capability( self, *, @@ -879,6 +898,8 @@ def list_connector_states(self) -> list[dict[str, object]]: def _install_fake_vnext_store(monkeypatch, store: FakeVNextStore) -> None: + from alicebot_api import vnext_label_writes + monkeypatch.setattr(vnext_label_writes, "acquire_exclusive_label_lock", lambda target: target.lock_label_writes(exclusive=True)) @contextmanager def fake_user_connection(database_url, current_user_id): assert database_url == "postgresql://db" @@ -1912,7 +1933,7 @@ def test_create_vnext_source_threads_project_scope_into_captured_memory(monkeypa candidates = store.list_memories(status="candidate") assert candidates, "capture must promote at least one candidate memory" - assert memory_project_scope(candidates[0]) == ("Project-Helios", "project-helios") + assert memory_project_scope(candidates[0]) == ("Project-Helios",) for memory in candidates: store.update_memory(memory_id=str(memory["id"]), patch={"status": "active"}, actor_type="system") @@ -2313,7 +2334,7 @@ def test_vnext_artifact_trace_authorizes_sources_from_complete_persisted_scope_e } store.artifacts[artifact_id] = { "id": artifact_id, - "artifact_type": "daily_brief", + "artifact_type": "manual_note", "title": "Scoped trace", "content_markdown": "# Scoped trace", "status": "needs_review", @@ -2575,6 +2596,7 @@ def test_vnext_contradiction_and_belief_endpoints(monkeypatch) -> None: "sensitivity": "private", "memory_type": "belief", } + store.memories.append({"id": "memory-belief-1", "domain": "project", "sensitivity": "private", "canonical_text": "Alice should auto-promote generated artifacts into memory.", "status": "active", "metadata_json": {}}) _install_fake_vnext_store(monkeypatch, store) user_id = uuid4() @@ -4474,6 +4496,8 @@ def test_artifact_quality_rating_rejects_alias_forgery_and_rerates_authenticated "sensitivity": "private", } ) + from alicebot_api.vnext_derived_labels import stamp_derived_from + stamp_derived_from(artifact, {}) _record, raw_key = create_agent_key( store, user_id=user_id, agent_id="reviewer", permission_profile="trusted_local_agent" ) @@ -4575,6 +4599,8 @@ def test_artifact_routes_authorize_persisted_target_scope_and_profile(monkeypatc "metadata_json": {"project_id": "project-b"}, } + from alicebot_api.vnext_derived_labels import stamp_derived_from + stamp_derived_from(store.artifacts[artifact_id], {}) _reader_record, reader_key = create_agent_key( store, user_id=user_id, @@ -4612,6 +4638,13 @@ def test_artifact_routes_authorize_persisted_target_scope_and_profile(monkeypatc "source_refs": [f"source:{sensitive_source_id}"], }, } + derived_status, _derived_payload = _invoke_vnext_request( + "GET", f"/v0/vnext/traces/artifacts/{public_artifact_id}", + query={"user_id": str(user_id)}, authorization=f"Bearer {reader_key}", + ) + assert derived_status == 403 + # A manual original public artifact exercises per-source trace projection separately. + store.artifacts[public_artifact_id]["artifact_type"] = "manual_note" trace_status, trace_payload = _invoke_vnext_request( "GET", f"/v0/vnext/traces/artifacts/{public_artifact_id}", From c09857a4176ab60b6c8acf995952bd9ae4468989 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 21:35:56 +0200 Subject: [PATCH 37/90] Serialize rollup member IDs as JSON lists before label settlement --- apps/api/src/alicebot_api/vnext_rollups.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/apps/api/src/alicebot_api/vnext_rollups.py b/apps/api/src/alicebot_api/vnext_rollups.py index 0bb942434..2170e16b3 100644 --- a/apps/api/src/alicebot_api/vnext_rollups.py +++ b/apps/api/src/alicebot_api/vnext_rollups.py @@ -2706,7 +2706,7 @@ def _create_rollup_candidate( "grouping_input_count": grouping_input_count, "grouping_input_total": grouping_input_total, "grouping_input_total_exact": grouping_input_total_exact, - "member_ids": member_ids, + "member_ids": list(member_ids), "instances": instances, } if revises_memory_id is not None: @@ -2860,7 +2860,7 @@ def propose_rollups( "rollup_key": group.rollup_key, "group_kind": group.group_kind, "label": group.label, - "member_ids": member_ids, + "member_ids": list(member_ids), "rollup_digest": rollup_digest, "aggregation": group.utility.to_record(), } From 0d2dec67db6170a633d59e234daec3e5a264798c Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 21:35:56 +0200 Subject: [PATCH 38/90] Confirm source scope moves in identity integration fixtures --- tests/integration/test_source_review_identity_api.py | 2 ++ 1 file changed, 2 insertions(+) diff --git a/tests/integration/test_source_review_identity_api.py b/tests/integration/test_source_review_identity_api.py index 19078d603..ff170c650 100644 --- a/tests/integration/test_source_review_identity_api.py +++ b/tests/integration/test_source_review_identity_api.py @@ -110,6 +110,7 @@ def test_source_review_http_rotates_identity_and_recapture_uses_new_envelope( payload={ "user_id": str(user_id), "action": "assign_project", + "confirm_label_hide": True, "project_id": "Beta", "domain": "professional", "sensitivity": "internal", @@ -195,6 +196,7 @@ def test_source_review_http_collision_returns_409_after_full_transaction_rollbac payload={ "user_id": str(user_id), "action": "assign_project", + "confirm_label_hide": True, "project_id": "Beta", "review_note": "This conflicts with Beta's live source.", }, From abfb6a25b9d7c7005e53b71d55772858476c2b7d Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 21:35:56 +0200 Subject: [PATCH 39/90] Parse documented nested consolidation membership strictly --- apps/api/src/alicebot_api/vnext_derived_labels.py | 10 ++++++++++ tests/unit/test_derived_labels_kernel.py | 13 +++++++++++++ 2 files changed, 23 insertions(+) diff --git a/apps/api/src/alicebot_api/vnext_derived_labels.py b/apps/api/src/alicebot_api/vnext_derived_labels.py index d368dea30..e602e9ede 100644 --- a/apps/api/src/alicebot_api/vnext_derived_labels.py +++ b/apps/api/src/alicebot_api/vnext_derived_labels.py @@ -502,6 +502,16 @@ def _collect_metadata_ids(value: object, found: set[tuple[str, str]]) -> str: elif isinstance(item, str): found.add(("memory", identifier(item))) continue + if key == "cluster_membership" and isinstance(child, list) and any(isinstance(item, list) for item in child): + # Consolidation records one list of member IDs per cluster. + # Keep its established nested JSON shape, but reject mixed or + # non-string membership rather than silently omitting inputs. + if any(not isinstance(cluster, list) or any(not isinstance(item, str) for item in cluster) for cluster in child): + problem = problem or "malformed" + else: + for cluster in child: + _add_ids(found, "memory", _strings(cluster)) + continue if key in _ID_KIND: parsed = _as_string_list(child if isinstance(child, list) else [child] if isinstance(child, str) else child) if parsed is None: diff --git a/tests/unit/test_derived_labels_kernel.py b/tests/unit/test_derived_labels_kernel.py index a56de91f1..30ea192e6 100644 --- a/tests/unit/test_derived_labels_kernel.py +++ b/tests/unit/test_derived_labels_kernel.py @@ -846,3 +846,16 @@ def test_weekly_parent_backfill_cannot_clear_ambiguous_identity() -> None: settled = settle_labels([_source("s"), one, two, parent]) assert settled.by_stored("memory", canonical).unverified is True assert settled.by_stored("memory", "{" + canonical + "}").unverified is True + + +@pytest.mark.parametrize("membership, expected", [([["member-a"], ["member-b"]], False), ([["member-a"], [42]], True), ([["member-a"], "member-b"], True)]) +def test_nested_consolidation_membership_is_strict(membership, expected): + rows = [ + {"kind": "memory", "id": "member-a", "user_id": USER, "domain": "health", "sensitivity": "confidential"}, + {"kind": "memory", "id": "member-b", "user_id": USER, "domain": "personal", "sensitivity": "public"}, + {"kind": "artifact", "id": "report", "user_id": USER, "artifact_type": "memory_consolidation", "domain": "unknown", "sensitivity": "public", "metadata_json": {"consolidation": {"cluster_membership": membership}}}, + ] + report = settle_labels(rows).by_stored("artifact", "report", user_id=USER) + assert report.unverified is expected + if not expected: + assert (report.domain, report.sensitivity) == ("health", "confidential") From a7f4265c7e6c06aad8bdaf7b7f3fe0d8128c4b5d Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 21:38:39 +0200 Subject: [PATCH 40/90] test: pin reviewed label-lock and lifecycle carrier changes --- tests/unit/test_store_embedding_cas_split.py | 26 ++++++++++++++----- .../unit/test_store_memory_lifecycle_split.py | 21 +++++++++------ 2 files changed, 32 insertions(+), 15 deletions(-) diff --git a/tests/unit/test_store_embedding_cas_split.py b/tests/unit/test_store_embedding_cas_split.py index eb8194bb0..200c00ee7 100644 --- a/tests/unit/test_store_embedding_cas_split.py +++ b/tests/unit/test_store_embedding_cas_split.py @@ -47,10 +47,13 @@ # whose ``valid_to`` has passed, with the test recall's own SQL uses # (``_expiry_clause`` on SQLite, ``POSTGRES_UNEXPIRED_SQL`` on Postgres), so the # text of an expired memory is never listed for embedding. +# Re-minted for derived-label locking: update and clear now take the shared +# label lock before their existing SQL. Only the decorator changes each AST; +# PG query receipts below prepend that lock and retain the prior query hashes. EXPECTED_METHOD_AST_SHA256 = { "postgres": { - "update_memory_embedding": "0cd0f0ef6f7bcaa6328b6f586a711a10b011c77af3e49d65b96c27b77a657cd9", - "clear_memory_embedding": "4e9fe6955f3246b51998c6b547f48a659f947f8a8150e6c86d4e61a0cf46df6c", + "update_memory_embedding": "291a378fe61e93c43dd7a971de31a07601feaecb470dc578f6229ae123cdbec4", + "clear_memory_embedding": "0ce41174d08f8d515ef171253203af4e4115cc432c6b51eeced50cef1bf1dcd7", "list_memories_missing_embeddings": "cfdbde2bb409a2a6761fe31f116a9ebb73ba12d4e7994ae691f87e335a095a2d", }, # SQLite update/clear re-minted for the Phase 4 Stage 2 resident vector @@ -70,8 +73,8 @@ # byte what it was, which ``test_embedding_cas_generated_sql_is_byte_identical`` # still pins. "sqlite": { - "update_memory_embedding": "1f4517352a0f7d6a9147f326bc96a6c1d61effa3f106add89546cd981ddd05fc", - "clear_memory_embedding": "51b583b250883911f0c5a068fec7ec4565f719c2bffafb1ed1c6b3dc980fa36c", + "update_memory_embedding": "a87ce54306b804ce87437ceeee55935f584bcce2a4a8ca36e20f5af3129c1de4", + "clear_memory_embedding": "ebd18c7058ab7be4610ed9eb4053f8ca115d44055ddcc550cd2a892dae5c78e1", "list_memories_missing_embeddings": "bea1d517cd3ad4d0af12c96d93a5b21671b84a4712d5172e718f5af5b43c4b05", }, } @@ -156,9 +159,18 @@ # They were re-minted again for the expiry test: each query holds the unexpired # test right after the status test (SQLite binds the time after the statuses). EXPECTED_QUERY_SHA256 = { - "postgres_unsigned_update": ("dcbf4bc29a7702e9c17d864f65e1c1f36d641927d3f31aa4ec80825646c030ef",), - "postgres_signed_update": ("1351db18168f7e23454736129e26a7c01039ca1bfdfcac235e3c666cf60d91db",), - "postgres_clear": ("a5a6952a93bd77b3bdf311fe2682b411263d18a2822a9617c6fb7524555123ca",), + "postgres_unsigned_update": ( + "1f866e65df3baaa9d7790266ff6b90630fa472a974bfd217894292df72a25a7f", + "dcbf4bc29a7702e9c17d864f65e1c1f36d641927d3f31aa4ec80825646c030ef", + ), + "postgres_signed_update": ( + "1f866e65df3baaa9d7790266ff6b90630fa472a974bfd217894292df72a25a7f", + "1351db18168f7e23454736129e26a7c01039ca1bfdfcac235e3c666cf60d91db", + ), + "postgres_clear": ( + "1f866e65df3baaa9d7790266ff6b90630fa472a974bfd217894292df72a25a7f", + "a5a6952a93bd77b3bdf311fe2682b411263d18a2822a9617c6fb7524555123ca", + ), "postgres_unsigned_missing": ("866920a62d5650df8e229d0dffa43ac0a8ce18116e3cbb98376928f19b239534",), "postgres_signed_missing": ("8593736f07c1853635e8d3868e6a8b54a034ccd3de1a3519abf885a0ff23fa3e",), # SQLite update/clear sequences start with BEGIN IMMEDIATE (the capture diff --git a/tests/unit/test_store_memory_lifecycle_split.py b/tests/unit/test_store_memory_lifecycle_split.py index 87ad25b53..2f8b6c208 100644 --- a/tests/unit/test_store_memory_lifecycle_split.py +++ b/tests/unit/test_store_memory_lifecycle_split.py @@ -77,9 +77,14 @@ # row is created in or moved into a searchable status. Previous receipts: # common 191f0ebd..., postgres 1960ff3d..., sqlite 5ebda2b3...; method ASTs # postgres 538d5a18..., sqlite 5bd15d28.... Metadata manifests are unchanged. +# Re-minted for the derived-label write boundary: inserts settle their full +# ancestry, updates preserve protected metadata and propagate labels, and +# mutators take the label lock. Metadata receipts include the label_write +# keyword and lock wrappers. Facades add only lock_label_writes/read_label_rows; +# removing those two names reproduces each previous class-order receipt. SOURCE_RECEIPTS = { COMMON_PATH: "8fc077dc71f0e631a2df81de2ebeec1fb6c768f341c2e7891309e4753eef7bb5", - POSTGRES_CARRIER_PATH: "65e23bf5c8809a5dabe3f3339500f4a8f879258b2d9ca5a91acae3331d9b54a3", + POSTGRES_CARRIER_PATH: "371f92595d2f72f0cfa225a49c03aa39498caf094df4f26f4f9ac4cd926e0de1", # SQLite carrier re-minted for the Phase 4 Stage 2 resident vector cache # (reviewed change): redaction paths that NULL a live embedding now bump # the embedding_stamp token in the same transaction (prompt eviction). @@ -90,15 +95,15 @@ # back between two reads cannot fail memories_seen_range_check. Previous # sqlite receipt 67adaa61..., method AST 3f134ac9...; the metadata # manifests are unchanged. - SQLITE_CARRIER_PATH: "c37f6b8012de25c3e702705909ba5669141af15d6c4ad5ac381915207b863615", + SQLITE_CARRIER_PATH: "f893f1faeeb9a87135c108992aa91ff036c5f5dccb1bbdaf315ae5afe1b89b73", } EXPECTED_METHOD_AST_MANIFESTS = { - "postgres": "e937452df97467820cbcb42938b5f4a2336cd0157f0ca69f8f6420d4ee85211b", - "sqlite": "df43d593a59eb6deaf3ba935c09382e330b96a90b9f0b63314712619ba468a0d", + "postgres": "d4969140e86b136b29708dc3bb6b4bca635b73b4016c4e633c4d5d3da841e784", + "sqlite": "d0b6024f0803ca9d3c6f6ea7f5022a28453b2b8b83833f0f05343cb3eff89a57", } EXPECTED_METADATA_MANIFESTS = { - "postgres": "af03955c805f720b8d3ec735f8202efeb5f405c8c7de1cc45cbfef3644867824", - "sqlite": "9a5a4a9f0ae533652250a9e9854cd34a068392d71ffde98b012c9c620134d2c4", + "postgres": "07a567e26d0f7c4f51ae2a1910d059397b513a575d85f06c2f8c0396ac1bb2ef", + "sqlite": "1270ef0115988552418349aa9e94a7442ba04be41443f278f68a1fa81857903a", } EXPECTED_CLASS_ORDERS = { # Two paired browser-clip capability methods extend both façades, and one @@ -106,7 +111,7 @@ # Per-file importer savepoint (2026-10-02): one paired method more, ``savepoint``, appended last. # Previous receipt: (171, 526374782104a2a1...). Proof: the member list equals the list at origin/main # 040a2a10 with ``savepoint`` added at the end and nothing else moved (reviewed change, not drift). - "PostgresVNextStore": (172, "6f1a459fcf4319cf4281f6cc0d4e81679c3e05d851fd0a874a2d90298d7c2569"), + "PostgresVNextStore": (174, "095250b8a77d6c0a32d2783343916b947bcae8ae86e3a1693e47c2fb11802211"), # One SQLite-only method more, ``check_source_search_query``: the Postgres # source search has no expression-depth or LIKE-length limit to check. # Merge of #500 and #502 (2026-10-01): one more SQLite-only method, @@ -121,7 +126,7 @@ # 040a2a10 with ``savepoint`` added at the end and nothing else moved (reviewed change, not drift). # Proof: the replacement branch gains only scrub_source, source_inventory and # prunable_sources here; every pre-existing class member keeps its order. - "SQLiteVNextStore": (134, "1301272026897057cf071009cc21787543ddc326f1e06f1a75a763f3e344767e"), + "SQLiteVNextStore": (136, "1d99dbaf69e4ea888ca7beb2389ac176650a2e573c067bf0686adc9e609132f5"), } EXPECTED_FACADE_COMMENT_DIGESTS = { POSTGRES_FACADE_PATH: "d8599a46ee26dc35a3ae52c1a98a416509add9ae4a42ece780c5c5ed7e132b93", From ebf11ea78ee5a3237a2de80994433fff4287c98c Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 21:37:46 +0200 Subject: [PATCH 41/90] Give workspace scope fixtures valid recorded ancestry --- tests/integration/test_vnext_live_workspace_api.py | 9 ++++++++- 1 file changed, 8 insertions(+), 1 deletion(-) diff --git a/tests/integration/test_vnext_live_workspace_api.py b/tests/integration/test_vnext_live_workspace_api.py index 945c1e4d7..504648ec4 100644 --- a/tests/integration/test_vnext_live_workspace_api.py +++ b/tests/integration/test_vnext_live_workspace_api.py @@ -1158,6 +1158,7 @@ def test_vnext_live_workspace_happy_path_writes_reviewable_postgres_state( ] ), "title": "Live workspace launch note", + "project_scope": [project_id], "domain": "project", "sensitivity": "private", }, @@ -1815,7 +1816,13 @@ def test_vnext_artifact_routes_enforce_persisted_scope_with_live_postgres( "status": "needs_review", "domain": "project", "sensitivity": "private", - "metadata_json": {"project_id": "project-b"}, + "metadata_json": { + "project_id": "project-b", + "derived_from": { + "v": 1, "sources": [], "memories": [], "open_loops": [], "artifacts": [], "beliefs": [], + "counts": {"sources": 0, "memories": 0, "open_loops": 0, "artifacts": 0, "beliefs": 0}, + }, + }, }, actor_type="user", ) From 31148ac015e6e871f564fec5fec10398ce5952f5 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 21:42:49 +0200 Subject: [PATCH 42/90] test: account for stored label floor in context budget golden --- tests/unit/test_search_goldens.py | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/tests/unit/test_search_goldens.py b/tests/unit/test_search_goldens.py index 9a6964b02..add771338 100644 --- a/tests/unit/test_search_goldens.py +++ b/tests/unit/test_search_goldens.py @@ -97,6 +97,16 @@ def test_a_scenario_equals_its_golden(computed: dict[str, dict[str, object]], su "changed_files_count": 1, "replacement_hint": "Use --supersede --dry-run to preview replacement, then --supersede to apply it.", }} + # Derived inserts now persist the empty project floor. The sources-first + # budget prices that stored metadata before compact projection, adding five + # tokens; every returned content field and the frozen fixture stay pinned. + if surface == "context_pack" and name == "sources_first": + result = expected["result"] + report = result["token_report"] + expected = {**expected, "result": {**result, "token_report": {**report, + "token_estimate": report["token_estimate"] + 5, + "full_pack_serialized_token_estimate": report["full_pack_serialized_token_estimate"] + 5, + }}} actual = computed[surface].get(name) assert actual == expected, _explain(surface, name, expected, actual) From 1bd1319d1256a043d739aa44883db44e492c2924 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 21:41:05 +0200 Subject: [PATCH 43/90] Preserve resolved legacy scope for original explained memories --- apps/api/src/alicebot_api/mcp/evidence_artifacts.py | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/apps/api/src/alicebot_api/mcp/evidence_artifacts.py b/apps/api/src/alicebot_api/mcp/evidence_artifacts.py index 1607cb387..a44771f2f 100644 --- a/apps/api/src/alicebot_api/mcp/evidence_artifacts.py +++ b/apps/api/src/alicebot_api/mcp/evidence_artifacts.py @@ -283,7 +283,11 @@ def _authorize_explain_resource( if target_type in {"memory", "source", "artifact"}: judged = effective_row_for_fence(store, identity, target_type, resource) _domains, _sensitivity, judged_scope, judged_floor = policy_labels(judged) - if target_type == "source": + from alicebot_api.vnext_derived_labels import is_derived + + if target_type == "source" or (target_type == "memory" and not is_derived("memory", resource)): + # Original legacy memories may keep scope in value. The caller + # already resolved that fallback; derived rows use effective labels. judged_scope = project_scope _actor_type, _actor_id, decision = _policy_checked( store, # type: ignore[arg-type] From 1425e212e1a1270d97f90bbce82adfdf3d4d9690 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 21:41:05 +0200 Subject: [PATCH 44/90] Give placeholder state fixtures valid canonical provenance --- tests/unit/test_derived_read_parity.py | 3 ++- tests/unit/test_expired_memories_everywhere.py | 5 +++-- tests/unit/test_mcp_refusal_is_independent_of_state.py | 2 ++ tests/unit/test_vnext_capture.py | 3 ++- 4 files changed, 9 insertions(+), 4 deletions(-) diff --git a/tests/unit/test_derived_read_parity.py b/tests/unit/test_derived_read_parity.py index f52dd3774..94bf659d4 100644 --- a/tests/unit/test_derived_read_parity.py +++ b/tests/unit/test_derived_read_parity.py @@ -18,6 +18,7 @@ from alicebot_api.sqlite_store import SQLiteVNextStore, sqlite_user_connection from alicebot_api.vnext_agent_keys import create_agent_key from alicebot_api.vnext_brain import _artifact_domain +from alicebot_api.vnext_derived_labels import with_derived_from from tests.unit.per_project_s2_support import add_memory from tests.unit.test_derived_domain_fence import USER @@ -51,7 +52,7 @@ def test_unrestricted_derived_read_contract(tmp_path, monkeypatch, profile, tool store = SQLiteVNextStore(conn, USER) health = add_memory(store, key="health", text="Private input", domain="health") project = add_memory(store, key="project", text="Project input", domain="project") - metadata = {"discovered_by": "vnext_weekly_synthesis"} + metadata = with_derived_from({"discovered_by": "vnext_weekly_synthesis", "project_scope": [], "project_floor": []}, {}) derived = store.create_memory( { "memory_key": "synthesis", diff --git a/tests/unit/test_expired_memories_everywhere.py b/tests/unit/test_expired_memories_everywhere.py index c16674369..dd13430ea 100644 --- a/tests/unit/test_expired_memories_everywhere.py +++ b/tests/unit/test_expired_memories_everywhere.py @@ -58,6 +58,7 @@ from alicebot_api.sqlite_schema import bootstrap_sqlite_schema from alicebot_api.sqlite_store import SQLiteVNextStore, ensure_sqlite_user, sqlite_user_connection from alicebot_api.vnext_agent_control import PolicyDecision +from alicebot_api.vnext_derived_labels import with_derived_from from alicebot_api.vnext_artifact_review import dispatch_vnext_artifact_review from alicebot_api.vnext_consolidation import MemoryConsolidationRequest, VNextConsolidationService from alicebot_api.vnext_embeddings import ( @@ -154,7 +155,7 @@ def _memory( "domain": domain, "sensitivity": sensitivity, "valid_to": valid_to, - "metadata_json": metadata or {}, + "metadata_json": with_derived_from(metadata, {}) if metadata and metadata.get("candidate_kind") == ROLLUP_CANDIDATE_KIND else metadata or {}, } ) if embed: @@ -1138,7 +1139,7 @@ def _digest_card( ``deleted_at`` empty, which is not a soft-deleted row and does not reach the read that skips them. """ - metadata: dict[str, object] = {"candidate_kind": candidate_kind, "rollup_key": rollup_key, "rollup_digest": digest} + metadata: dict[str, object] = with_derived_from({"candidate_kind": candidate_kind, "rollup_key": rollup_key, "rollup_digest": digest}, {}) if project is not None: metadata["project_scope"] = [project] card = store.create_memory( diff --git a/tests/unit/test_mcp_refusal_is_independent_of_state.py b/tests/unit/test_mcp_refusal_is_independent_of_state.py index 436a7c1b1..8c011320b 100644 --- a/tests/unit/test_mcp_refusal_is_independent_of_state.py +++ b/tests/unit/test_mcp_refusal_is_independent_of_state.py @@ -35,6 +35,7 @@ from alicebot_api.mcp.runtime import _sqlite_path_from_url, _vnext_store_context from alicebot_api.mcp.types import MCPRuntimeContext from alicebot_api.onramp import bootstrap_database, resolve_db_path, sqlite_url_for_path +from alicebot_api.vnext_derived_labels import with_derived_from from alicebot_api.sqlite_store import SQLiteVNextStore, sqlite_user_connection from alicebot_api.vnext_agent_control import AgentIdentity, AgentPolicyBlockedError from alicebot_api.vnext_agent_keys import create_agent_key @@ -203,6 +204,7 @@ def backdate(metadata: dict) -> dict: def _as_project_update(candidate: object) -> Callable[[dict], dict]: def update(metadata: dict) -> dict: + metadata.update(with_derived_from({}, {})) metadata["workflow"] = "project_auto_update" if candidate is not None: metadata["candidate"] = candidate diff --git a/tests/unit/test_vnext_capture.py b/tests/unit/test_vnext_capture.py index dcc62bcbb..a1a2c7895 100644 --- a/tests/unit/test_vnext_capture.py +++ b/tests/unit/test_vnext_capture.py @@ -1433,5 +1433,6 @@ def test_capture_without_project_scope_keeps_empty_scope_metadata() -> None: source = store.get_source(result.source_id) memory = store.list_memories(status="candidate")[0] assert "project_scope" not in source["metadata_json"] - assert "project_scope" not in memory["metadata_json"] + assert memory["metadata_json"]["project_scope"] == [] + assert memory["metadata_json"]["project_floor"] == [] assert memory_project_scope(memory) == () From e59ff844dfe1c406fda4e4d834f7d7132bbe07a1 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 21:41:05 +0200 Subject: [PATCH 45/90] Model recorded ancestry in MCP authorization fixtures --- tests/unit/test_mcp.py | 18 ++++++++++++++++-- 1 file changed, 16 insertions(+), 2 deletions(-) diff --git a/tests/unit/test_mcp.py b/tests/unit/test_mcp.py index 8561365d0..ff52aa588 100644 --- a/tests/unit/test_mcp.py +++ b/tests/unit/test_mcp.py @@ -33,6 +33,7 @@ import alicebot_api.mcp.types as mcp_types_module import alicebot_api.mcp_server as mcp_server import alicebot_api.mcp_tools as mcp_tools_module +from alicebot_api.vnext_derived_labels import with_derived_from import alicebot_api.vnext_retrieval as vnext_retrieval_module from alicebot_api.mcp_tools import MCPRuntimeContext, MCPToolError, MCPToolNotFoundError, call_mcp_tool, list_mcp_tools from alicebot_api.sqlite_schema import bootstrap_sqlite_schema @@ -1317,6 +1318,19 @@ def __init__(self) -> None: } } + def read_label_rows(self, kind: str, ids: list[str]) -> list[dict[str, object]]: + # The fake search methods expose fixed persisted rows as well as writes. + collections = { + "memory": [*self.search_memories(), *self.memories], + "source": [*self.search_sources(), *self.sources], + "artifact": list(self.artifacts.values()), + "open_loop": self.open_loops, + "project": list(self.projects.values()), + "belief": list(self.beliefs.values()), + } + found = {str(row.get("id")): row for row in collections.get(kind, [])} + return [dict(found[item]) for item in ids if item in found] + @staticmethod def _is_live(row: dict[str, object]) -> bool: return row.get("deleted_at") is None @@ -3802,7 +3816,7 @@ def test_vnext_artifact_get_authorizes_persisted_scope_and_sensitivity(monkeypat "status": "needs_review", "domain": "project", "sensitivity": "private", - "metadata_json": {"project_id": "project-b"}, + "metadata_json": with_derived_from({"project_id": "project-b"}, {}), } ) artifact_id = str(artifact["id"]) @@ -3842,7 +3856,7 @@ def test_vnext_artifact_review_locks_and_authorizes_persisted_scope(monkeypatch, "status": "needs_review", "domain": "project", "sensitivity": "private", - "metadata_json": {"project_id": "project-b"}, + "metadata_json": with_derived_from({"project_id": "project-b"}, {}), } ) artifact_id = str(artifact["id"]) From 79a23317dab267e569b7d6d0117cb4eff43fd9da Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 20:09:49 +0000 Subject: [PATCH 46/90] Clamp a derived memory edit at its inputs and name the clamp. An owner edit that would lower a derived memory is stored at the higher label. The review answer sets label_floor_applied only when that happens, and the event cause is floor_clamped. --- .../alicebot_api/routers/vnext_memories.py | 3 + .../src/alicebot_api/vnext_label_writes.py | 84 +++++++++++++++++++ .../vnext_stores/postgres/memory_lifecycle.py | 4 + .../vnext_stores/sqlite/memory_lifecycle.py | 4 + tests/unit/test_label_floor_applied.py | 64 ++++++++++++++ 5 files changed, 159 insertions(+) create mode 100644 tests/unit/test_label_floor_applied.py diff --git a/apps/api/src/alicebot_api/routers/vnext_memories.py b/apps/api/src/alicebot_api/routers/vnext_memories.py index 4bc6667eb..b0156174c 100644 --- a/apps/api/src/alicebot_api/routers/vnext_memories.py +++ b/apps/api/src/alicebot_api/routers/vnext_memories.py @@ -1332,6 +1332,9 @@ def review_vnext_memory( actor_id=actor_id, ) review_payload: dict[str, object] = {"memory": updated} + if getattr(store, "_label_floor_applied", False): + review_payload["label_floor_applied"] = True + store._label_floor_applied = False if action == "reject": review_payload["rationale_withheld"] = rationale_withheld review_payload["text_withheld"] = text_withheld diff --git a/apps/api/src/alicebot_api/vnext_label_writes.py b/apps/api/src/alicebot_api/vnext_label_writes.py index 9fd571284..015384ae2 100644 --- a/apps/api/src/alicebot_api/vnext_label_writes.py +++ b/apps/api/src/alicebot_api/vnext_label_writes.py @@ -412,6 +412,89 @@ def _label_fields(row: Mapping[str, object]) -> tuple[str, str, tuple[str, ...], ) +def clamp_owner_patch( + store: Any, *, kind: str, before: Mapping[str, object] | None, patch: Mapping[str, object] +) -> JsonObject: + """Keep a derived row at or above its inputs when an edit would lower it. + + The store writes the higher label, records ``labels_raised`` with cause + ``floor_clamped`` when the stored label changes, and sets + ``store._label_floor_applied`` so the review answer can name it. + """ + + store._label_floor_applied = False + proposed_patch = dict(patch) + if before is None or not is_derived(kind, before): + return proposed_patch + proposed = dict(before) + for key in ("domain", "sensitivity", "project_id"): + if key in proposed_patch and proposed_patch[key] is not None: + proposed[key] = proposed_patch[key] + if isinstance(proposed_patch.get("metadata_json"), dict): + stored_meta = before.get("metadata_json") + meta = dict(stored_meta) if isinstance(stored_meta, dict) else {} + meta.update(proposed_patch["metadata_json"]) + proposed["metadata_json"] = meta + proposed["kind"] = kind + nodes, exceeded = collect_label_rows(store, [proposed], max_nodes=PROPAGATION_BOUND) + if exceeded: + return proposed_patch + try: + label = settle_labels(nodes).by_stored(kind, str(before.get("id") or "")) + except KeyError: + return proposed_patch + if label.unverified: + return proposed_patch + requested = _label_fields(proposed) + settled = (label.domain, label.sensitivity, tuple(label.project_scope), tuple(label.project_floor)) + if ( + requested[0] == settled[0] + and requested[1] == settled[1] + and project_scope_identity(requested[2]) == project_scope_identity(settled[2]) + and project_scope_identity(requested[3]) == project_scope_identity(settled[3]) + ): + return proposed_patch + proposed_patch["domain"] = label.domain + proposed_patch["sensitivity"] = label.sensitivity + metadata = dict(proposed.get("metadata_json") or {}) + metadata["project_scope"] = list(label.project_scope) + metadata["project_floor"] = list(label.project_floor) + proposed_patch["metadata_json"] = metadata + stored = _label_fields(before) + if not ( + stored[0] == settled[0] + and stored[1] == settled[1] + and project_scope_identity(stored[2]) == project_scope_identity(settled[2]) + and project_scope_identity(stored[3]) == project_scope_identity(settled[3]) + ): + event = build_event_log_record( + event_type=f"{kind}.labels_raised", + actor_type="system", + target_type=kind, + target_id=str(before.get("id") or ""), + payload=labels_raised_payload( + cause="floor_clamped", + previous={ + "domain": stored[0], + "sensitivity": stored[1], + "project_scope": list(stored[2]), + "project_floor": list(stored[3]), + }, + new={ + "domain": label.domain, + "sensitivity": label.sensitivity, + "project_scope": list(label.project_scope), + "project_floor": list(label.project_floor), + }, + ), + ) + append = getattr(store, "append_event", None) + if callable(append): + append(event) + store._label_floor_applied = True + return proposed_patch + + def write_settled_label( store: Any, *, @@ -642,6 +725,7 @@ def remember_floor_event(store: Any, event: JsonObject | None, target_id: object "REFUSED_DETAIL", "RETRYABLE_DETAIL", "acquire_exclusive_label_lock", + "clamp_owner_patch", "count_rows_hidden_by_scope_move", "label_error_response", "propagate", diff --git a/apps/api/src/alicebot_api/vnext_stores/postgres/memory_lifecycle.py b/apps/api/src/alicebot_api/vnext_stores/postgres/memory_lifecycle.py index 99be62c49..64294a5be 100644 --- a/apps/api/src/alicebot_api/vnext_stores/postgres/memory_lifecycle.py +++ b/apps/api/src/alicebot_api/vnext_stores/postgres/memory_lifecycle.py @@ -450,6 +450,10 @@ def update_memory( patch_metadata, label_write=label_write, ) + if before_label is not None: + from alicebot_api.vnext_label_writes import clamp_owner_patch + + patch = clamp_owner_patch(self, kind="memory", before=before_label, patch=patch) row = self._fetch_one( "update_memory", f""" diff --git a/apps/api/src/alicebot_api/vnext_stores/sqlite/memory_lifecycle.py b/apps/api/src/alicebot_api/vnext_stores/sqlite/memory_lifecycle.py index d5b89e9fc..728f8f359 100644 --- a/apps/api/src/alicebot_api/vnext_stores/sqlite/memory_lifecycle.py +++ b/apps/api/src/alicebot_api/vnext_stores/sqlite/memory_lifecycle.py @@ -304,6 +304,10 @@ def update_memory( before_label = self.get_memory(str(memory_id)) refuse_updated_credential_activation(patch, lambda: self.get_memory(str(memory_id))) patch = _with_protected_metadata(self, memory_id, patch, label_write=label_write) + if before_label is not None: + from alicebot_api.vnext_label_writes import clamp_owner_patch + + patch = clamp_owner_patch(self, kind="memory", before=before_label, patch=patch) # One clock reading for the write: an archive sets ``updated_at`` and ``deleted_at`` together. now = _utc_now_iso() cursor = self._execute( diff --git a/tests/unit/test_label_floor_applied.py b/tests/unit/test_label_floor_applied.py new file mode 100644 index 000000000..69e507c00 --- /dev/null +++ b/tests/unit/test_label_floor_applied.py @@ -0,0 +1,64 @@ +"""An owner edit that would lower a derived memory is held at its inputs. + +The handoff names this ``label_floor_applied`` on the review answer, with a +``labels_raised`` event whose cause is ``floor_clamped``. +""" + +from __future__ import annotations + +import inspect +import json + +from alicebot_api.onramp import bootstrap_database +from alicebot_api.routers.vnext_memories import review_vnext_memory +from alicebot_api.sqlite_store import SQLiteVNextStore, sqlite_user_connection +from alicebot_api.vnext_label_writes import without_insert_floor +from tests.unit.test_derived_domain_fence import USER + + +def test_an_edit_below_the_inputs_is_clamped_and_named(tmp_path) -> None: + """Mutation: ``clamp_owner_patch`` returns the patch unchanged. The stored sensitivity stays ``public`` + and no ``floor_clamped`` event is written. + """ + + path = tmp_path / "vault.db" + bootstrap_database(path, user_id=USER, user_email="local@alice") + with without_insert_floor(), sqlite_user_connection(path, USER) as conn: + store = SQLiteVNextStore(conn, USER) + source = store.create_source( + { + "source_type": "note", + "title": "Clinic note", + "content_hash": "sha256:clinic", + "domain": "health", + "sensitivity": "confidential", + "metadata_json": {}, + } + ) + memory = store.create_memory( + { + "memory_key": "copy", + "canonical_text": "A confidential observation", + "status": "active", + "domain": "project", + "sensitivity": "public", + "metadata_json": {"source_id": source["id"]}, + } + ) + updated = store.update_memory( + memory_id=str(memory["id"]), + patch={"sensitivity": "public", "domain": "project"}, + actor_type="user", + ) + event = conn.execute( + "SELECT payload_json FROM event_log WHERE event_type = 'memory.labels_raised' AND target_id = ?", + (str(memory["id"]),), + ).fetchone() + assert updated["sensitivity"] == "confidential" + assert store._label_floor_applied is True + raw = event["payload_json"] if isinstance(event, dict) else event[0] + payload = json.loads(raw) + assert payload["cause"] == "floor_clamped" + assert "title" not in json.dumps(payload) + assert "A confidential observation" not in json.dumps(payload) + assert 'review_payload["label_floor_applied"] = True' in inspect.getsource(review_vnext_memory) From 366e56bac5605cffb3e7562b2bd1bf8ca2c7132b Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 23:20:03 +0200 Subject: [PATCH 47/90] Validate incomplete legacy dependency counts --- .../src/alicebot_api/vnext_derived_labels.py | 10 ++++-- ...test_derived_labels_record_completeness.py | 36 +++++++++++++++++++ 2 files changed, 43 insertions(+), 3 deletions(-) create mode 100644 tests/unit/test_derived_labels_record_completeness.py diff --git a/apps/api/src/alicebot_api/vnext_derived_labels.py b/apps/api/src/alicebot_api/vnext_derived_labels.py index e602e9ede..3391fb4b9 100644 --- a/apps/api/src/alicebot_api/vnext_derived_labels.py +++ b/apps/api/src/alicebot_api/vnext_derived_labels.py @@ -605,8 +605,10 @@ def _legacy_count_problem(kind: str, row: Mapping[str, object], meta: Mapping[st def _counts_against_lists(counts: object, lists: Mapping[str, object]) -> str: - if not isinstance(counts, Mapping): + if counts is None: return "" + if not isinstance(counts, Mapping): + return "malformed" present_lists: dict[str, list[object]] = {} for key, raw in lists.items(): if raw is None: @@ -616,11 +618,13 @@ def _counts_against_lists(counts: object, lists: Mapping[str, object]) -> str: present_lists[key] = list(raw) if any(len(values) != len(set(map(str, values))) for values in present_lists.values()): return "" - for key, values in present_lists.items(): + for key in lists: if key not in counts: continue expected = counts.get(key) - if isinstance(expected, int) and not isinstance(expected, bool) and expected != len(values): + if not isinstance(expected, int) or isinstance(expected, bool) or expected < 0: + return "malformed" + if expected != len(present_lists.get(key, [])): return "counts_disagree" return "" diff --git a/tests/unit/test_derived_labels_record_completeness.py b/tests/unit/test_derived_labels_record_completeness.py new file mode 100644 index 000000000..efac96c30 --- /dev/null +++ b/tests/unit/test_derived_labels_record_completeness.py @@ -0,0 +1,36 @@ +"""Missing legacy lists cannot pass a producer's positive completeness count.""" + +from __future__ import annotations + +import pytest + +from alicebot_api.vnext_derived_labels import settle_labels + + +def legacy_report(workflow, lists, counts): + metadata = {"workflow": workflow} + if workflow in {"daily_brief", "weekly_synthesis"}: + metadata["input_summary"] = {**lists, "counts": counts} + else: + metadata.update(lists) + metadata["input_counts"] = counts + return {"kind": "artifact", "id": "report", "domain": "project", "sensitivity": "public", "metadata_json": metadata} + + +@pytest.mark.parametrize("workflow", ("daily_brief", "weekly_synthesis", "connection_finder", "contradiction_finder")) +@pytest.mark.parametrize("lists,counts", (({}, {"sources": 1}), ({"source_ids": []}, {"sources": "1"}), + ({"source_ids": []}, {"sources": True}), ({"source_ids": []}, {"sources": -1}), + ({"source_ids": []}, []))) +def test_a_missing_or_malformed_legacy_completeness_record_is_unverified(workflow, lists, counts): + row = legacy_report(workflow, lists, counts) + label = settle_labels([row]).by_stored("artifact", "report") + assert label.unverified is True + assert label.reason == "legacy_counts" + + +@pytest.mark.parametrize("workflow", ("daily_brief", "weekly_synthesis", "connection_finder", "contradiction_finder")) +@pytest.mark.parametrize("lists,counts", (({}, {"sources": 0}), ({"source_ids": []}, {"sources": 0}), + ({"source_ids": []}, {}))) +def test_a_legitimate_empty_legacy_record_is_verified(workflow, lists, counts): + row = legacy_report(workflow, lists, counts) + assert settle_labels([row]).by_stored("artifact", "report").unverified is False From dd2d3871a90eac26632e41651bdd6dd37ca780c8 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 23:20:03 +0200 Subject: [PATCH 48/90] Validate incomplete legacy dependency counts --- .../src/alicebot_api/vnext_derived_labels.py | 10 ++++-- ...test_derived_labels_record_completeness.py | 36 +++++++++++++++++++ 2 files changed, 43 insertions(+), 3 deletions(-) create mode 100644 tests/unit/test_derived_labels_record_completeness.py diff --git a/apps/api/src/alicebot_api/vnext_derived_labels.py b/apps/api/src/alicebot_api/vnext_derived_labels.py index e602e9ede..3391fb4b9 100644 --- a/apps/api/src/alicebot_api/vnext_derived_labels.py +++ b/apps/api/src/alicebot_api/vnext_derived_labels.py @@ -605,8 +605,10 @@ def _legacy_count_problem(kind: str, row: Mapping[str, object], meta: Mapping[st def _counts_against_lists(counts: object, lists: Mapping[str, object]) -> str: - if not isinstance(counts, Mapping): + if counts is None: return "" + if not isinstance(counts, Mapping): + return "malformed" present_lists: dict[str, list[object]] = {} for key, raw in lists.items(): if raw is None: @@ -616,11 +618,13 @@ def _counts_against_lists(counts: object, lists: Mapping[str, object]) -> str: present_lists[key] = list(raw) if any(len(values) != len(set(map(str, values))) for values in present_lists.values()): return "" - for key, values in present_lists.items(): + for key in lists: if key not in counts: continue expected = counts.get(key) - if isinstance(expected, int) and not isinstance(expected, bool) and expected != len(values): + if not isinstance(expected, int) or isinstance(expected, bool) or expected < 0: + return "malformed" + if expected != len(present_lists.get(key, [])): return "counts_disagree" return "" diff --git a/tests/unit/test_derived_labels_record_completeness.py b/tests/unit/test_derived_labels_record_completeness.py new file mode 100644 index 000000000..efac96c30 --- /dev/null +++ b/tests/unit/test_derived_labels_record_completeness.py @@ -0,0 +1,36 @@ +"""Missing legacy lists cannot pass a producer's positive completeness count.""" + +from __future__ import annotations + +import pytest + +from alicebot_api.vnext_derived_labels import settle_labels + + +def legacy_report(workflow, lists, counts): + metadata = {"workflow": workflow} + if workflow in {"daily_brief", "weekly_synthesis"}: + metadata["input_summary"] = {**lists, "counts": counts} + else: + metadata.update(lists) + metadata["input_counts"] = counts + return {"kind": "artifact", "id": "report", "domain": "project", "sensitivity": "public", "metadata_json": metadata} + + +@pytest.mark.parametrize("workflow", ("daily_brief", "weekly_synthesis", "connection_finder", "contradiction_finder")) +@pytest.mark.parametrize("lists,counts", (({}, {"sources": 1}), ({"source_ids": []}, {"sources": "1"}), + ({"source_ids": []}, {"sources": True}), ({"source_ids": []}, {"sources": -1}), + ({"source_ids": []}, []))) +def test_a_missing_or_malformed_legacy_completeness_record_is_unverified(workflow, lists, counts): + row = legacy_report(workflow, lists, counts) + label = settle_labels([row]).by_stored("artifact", "report") + assert label.unverified is True + assert label.reason == "legacy_counts" + + +@pytest.mark.parametrize("workflow", ("daily_brief", "weekly_synthesis", "connection_finder", "contradiction_finder")) +@pytest.mark.parametrize("lists,counts", (({}, {"sources": 0}), ({"source_ids": []}, {"sources": 0}), + ({"source_ids": []}, {}))) +def test_a_legitimate_empty_legacy_record_is_verified(workflow, lists, counts): + row = legacy_report(workflow, lists, counts) + assert settle_labels([row]).by_stored("artifact", "report").unverified is False From 8e6f426d58cb187858313858b3ea82c092e202a5 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 23:27:20 +0200 Subject: [PATCH 49/90] Enforce canonical dependency record completeness --- apps/api/src/alicebot_api/vnext_derived_labels.py | 7 +------ tests/unit/test_derived_labels_record_completeness.py | 10 ++++++++++ 2 files changed, 11 insertions(+), 6 deletions(-) diff --git a/apps/api/src/alicebot_api/vnext_derived_labels.py b/apps/api/src/alicebot_api/vnext_derived_labels.py index 3391fb4b9..97c4b3bc9 100644 --- a/apps/api/src/alicebot_api/vnext_derived_labels.py +++ b/apps/api/src/alicebot_api/vnext_derived_labels.py @@ -534,16 +534,13 @@ def _derived_from_deps(record: object) -> tuple[str, set[tuple[str, str]]]: return "malformed", set() found: set[tuple[str, str]] = set() lists: dict[str, list[str]] = {} - repeated = False for key, kind in _DERIVED_FROM_KIND.items(): if key not in record: lists[key] = [] continue raw = record.get(key) - if not isinstance(raw, list) or any(not isinstance(item, str) for item in raw): + if not isinstance(raw, list) or any(not isinstance(item, str) or not item.strip() for item in raw): return "malformed", set() - if len(raw) != len(set(raw)): - repeated = True ids = _strings(raw) lists[key] = ids _add_ids(found, kind, ids) @@ -554,8 +551,6 @@ def _derived_from_deps(record: object) -> tuple[str, set[tuple[str, str]]]: return "", found if not isinstance(counts, Mapping): return "malformed", found - if repeated: - return "", found for key, ids in lists.items(): if key not in counts: if ids: diff --git a/tests/unit/test_derived_labels_record_completeness.py b/tests/unit/test_derived_labels_record_completeness.py index efac96c30..ff9d6b679 100644 --- a/tests/unit/test_derived_labels_record_completeness.py +++ b/tests/unit/test_derived_labels_record_completeness.py @@ -34,3 +34,13 @@ def test_a_missing_or_malformed_legacy_completeness_record_is_unverified(workflo def test_a_legitimate_empty_legacy_record_is_verified(workflow, lists, counts): row = legacy_report(workflow, lists, counts) assert settle_labels([row]).by_stored("artifact", "report").unverified is False + + +@pytest.mark.parametrize("ids,count", ((["source", "source"], 1), ([""], 1), ([" "], 1))) +def test_canonical_counts_and_identifiers_do_not_use_the_legacy_duplicate_exception(ids, count): + source = {"kind": "source", "id": "source", "domain": "project", "sensitivity": "public", "metadata_json": {}} + row = {"kind": "artifact", "id": "report", "metadata_json": {"derived_from": { + "v": 1, "sources": ids, "memories": [], "open_loops": [], "artifacts": [], "beliefs": [], + "counts": {"sources": count, "memories": 0, "open_loops": 0, "artifacts": 0, "beliefs": 0}, + }}} + assert settle_labels([source, row]).by_stored("artifact", "report").unverified is True From 658560c73ca0604cc0ba9078fc092f7c87faa61c Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 23:17:45 +0200 Subject: [PATCH 50/90] Enforce live label lock order and complete scope move previews --- apps/api/src/alicebot_api/cli/capture.py | 3 + .../src/alicebot_api/vnext_artifact_review.py | 6 + .../src/alicebot_api/vnext_label_writes.py | 128 ++++++++++++++++-- apps/api/src/alicebot_api/vnext_projects.py | 6 + apps/api/src/alicebot_api/vnext_store.py | 33 +++++ .../vnext_stores/postgres/graph_open_loops.py | 13 ++ .../vnext_stores/postgres/memory_access.py | 2 + .../vnext_stores/postgres/memory_lifecycle.py | 6 +- .../vnext_stores/sqlite/graph_open_loops.py | 13 ++ tests/integration/conftest.py | 9 ++ .../test_label_lock_order_postgres.py | 63 +++++++++ tests/unit/test_label_lock_order.py | 86 ++++++++++++ tests/unit/test_label_lock_registry.py | 69 ++++++++++ tests/unit/test_label_writer_registry.py | 78 +++++++++++ tests/unit/test_source_move_label_preview.py | 53 ++++++++ .../test_sqlite_derived_labels_write_path.py | 2 +- 16 files changed, 554 insertions(+), 16 deletions(-) create mode 100644 tests/integration/test_label_lock_order_postgres.py create mode 100644 tests/unit/test_label_lock_order.py create mode 100644 tests/unit/test_label_lock_registry.py create mode 100644 tests/unit/test_label_writer_registry.py create mode 100644 tests/unit/test_source_move_label_preview.py diff --git a/apps/api/src/alicebot_api/cli/capture.py b/apps/api/src/alicebot_api/cli/capture.py index 241566da8..cce5ef0e6 100644 --- a/apps/api/src/alicebot_api/cli/capture.py +++ b/apps/api/src/alicebot_api/cli/capture.py @@ -41,6 +41,7 @@ from alicebot_api.vnext_embeddings import DeferredMemoryEmbedding from alicebot_api.vnext_event_log import append_event from alicebot_api.vnext_store import PostgresVNextStore +from alicebot_api.vnext_label_writes import takes_label_lock as _takes_label_lock from .constants import DEFAULT_VNEXT_DEMO_DATASET_PATH, DEMO_SECRET_MARKERS from .models import CLIContext from .arguments import _object_dict, _object_int, _object_list @@ -394,6 +395,7 @@ def _demo_tag(dataset_id: str) -> JsonObject: return {"demo": True, "demo_dataset_id": dataset_id} +@_takes_label_lock def _reset_vnext_demo_dataset(store: PostgresVNextStore, *, dataset_id: str) -> JsonObject: with store.conn.cursor() as cur: cur.execute( @@ -507,6 +509,7 @@ def _tag_demo_candidate_memories(store: PostgresVNextStore, *, dataset_id: str, return updated +@_takes_label_lock def _tag_demo_artifact(store: PostgresVNextStore, *, artifact_id: str, dataset_id: str) -> None: artifact = store.get_artifact(artifact_id) if artifact is None: diff --git a/apps/api/src/alicebot_api/vnext_artifact_review.py b/apps/api/src/alicebot_api/vnext_artifact_review.py index 3e965e244..b42178dbd 100644 --- a/apps/api/src/alicebot_api/vnext_artifact_review.py +++ b/apps/api/src/alicebot_api/vnext_artifact_review.py @@ -37,6 +37,12 @@ def dispatch_vnext_artifact_review( mutating it, so no caller can route from a stale or forged preloaded row. """ + lock_graph = getattr(store, "lock_graph_mutation", None) + if callable(lock_graph): + lock_graph() + from alicebot_api.vnext_label_writes import acquire_exclusive_label_lock + + acquire_exclusive_label_lock(store) target = store.get_artifact_for_update(artifact_id) if target is None: raise VNextQueueNotFoundError(f"artifact {artifact_id} was not found") diff --git a/apps/api/src/alicebot_api/vnext_label_writes.py b/apps/api/src/alicebot_api/vnext_label_writes.py index 015384ae2..c823feebd 100644 --- a/apps/api/src/alicebot_api/vnext_label_writes.py +++ b/apps/api/src/alicebot_api/vnext_label_writes.py @@ -25,6 +25,7 @@ generation_domain, identifier, is_derived, + input_admitted, labels_raised_payload, settle_labels, stored_scope, @@ -127,6 +128,62 @@ def _in_transaction(store: Any) -> bool: return True +def held_label_locks(store: Any) -> tuple[bool, bool, bool]: + """Read the live S, L and exclusive L grants, including savepoint rollback.""" + + with store.conn.cursor() as cur: + cur.execute(""" + SELECT + coalesce(bool_or(classid = (hashtext('vnext_supersession')::bigint & 4294967295)::oid), false) AS graph, + coalesce(bool_or(classid = (hashtext('vnext_labels')::bigint & 4294967295)::oid), false) AS labels, + coalesce(bool_or(classid = (hashtext('vnext_labels')::bigint & 4294967295)::oid + AND mode = 'ExclusiveLock'), false) AS exclusive + FROM pg_locks + WHERE locktype = 'advisory' AND pid = pg_backend_pid() AND granted + AND objsubid = 2 + AND objid = (hashtext(app.current_user_id()::text)::bigint & 4294967295)::oid + """) + row = cur.fetchone() + if isinstance(row, Mapping): + return bool(row["graph"]), bool(row["labels"]), bool(row["exclusive"]) + return bool(row[0]), bool(row[1]), bool(row[2]) + + +def before_graph_lock(store: Any) -> None: + """Strict tests refuse S after L using the current database grants.""" + + if STRICT_LOCK_ORDER: + graph, labels, _exclusive = held_label_locks(store) + if labels and not graph: + raise LabelLockOrderError("the graph lock must precede the label lock") + + +def require_exclusive_label_lock(store: Any) -> None: + """A changing hook must hold exclusive L before taking any row lock.""" + + if _sqlite(store): + store.lock_label_writes(exclusive=True) + return + _graph, _labels, exclusive = held_label_locks(store) + if exclusive: + return + if STRICT_LOCK_ORDER: + raise LabelLockOrderError("the label change requires the exclusive label lock before row locks") + acquire_exclusive_label_lock(store) + + +def prepare_label_patch( + store: Any, kind: str, before: Mapping[str, object] | None, patch: Mapping[str, object] +) -> JsonObject: + """Check a proposed label change before its UPDATE or FOR UPDATE statement.""" + + proposed = dict(before or {}) + proposed.update({key: value for key, value in patch.items() if value is not None}) + if before and _label_fields(before) != _label_fields(proposed): + require_exclusive_label_lock(store) + return dict(patch) + + def _label_tuple(payload: Mapping[str, object]) -> tuple[str, str, tuple[str, ...], tuple[str, ...]]: metadata = payload.get("metadata_json") meta = metadata if isinstance(metadata, Mapping) else {} @@ -467,6 +524,7 @@ def clamp_owner_patch( and project_scope_identity(stored[2]) == project_scope_identity(settled[2]) and project_scope_identity(stored[3]) == project_scope_identity(settled[3]) ): + require_exclusive_label_lock(store) event = build_event_log_record( event_type=f"{kind}.labels_raised", actor_type="system", @@ -510,7 +568,7 @@ def write_settled_label( """Label-only update. A statement that changes no row refuses the whole relabel.""" table = {"memory": "memories", "open_loop": "open_loops", "artifact": "generated_artifacts", "project": "projects"}[kind] - blob = json.dumps(dict(metadata)) + blob = json.dumps({key: metadata[key] for key in ("project_scope", "project_floor") if key in metadata}) if _sqlite(store): project_sql = ", project_id = ?" if table in {"memories", "open_loops"} else "" params: list[object] = [domain, sensitivity, blob] @@ -520,7 +578,7 @@ def write_settled_label( cursor = store._execute( f""" UPDATE {table} - SET domain = ?, sensitivity = ?, metadata_json = ?{project_sql} + SET domain = ?, sensitivity = ?, metadata_json = json_patch(metadata_json, ?){project_sql} WHERE id = ? AND user_id = ? AND domain = ? AND sensitivity = ? """, # nosec B608 # table comes from the closed kind map; values are bound tuple(params), @@ -532,22 +590,47 @@ def write_settled_label( if project_sql: params.append(project_id) params.extend([str(row_id), expected_domain, expected_sensitivity]) - store._fetch_one( - "write_settled_label", + row = store._fetch_optional_one( f""" UPDATE {table} - SET domain = %s, sensitivity = %s, metadata_json = %s::jsonb{project_sql} + SET domain = %s, sensitivity = %s, metadata_json = metadata_json || %s::jsonb{project_sql} WHERE id = %s::uuid AND domain = %s AND sensitivity = %s RETURNING id """, # nosec B608 # table comes from the closed kind map; values are bound tuple(params), ) + require_changed(int(row is not None), table, str(row_id)) + + +LABEL_TABLE_ORDER = ("generated_artifacts", "projects", "open_loops", "memories") +_LABEL_TABLES = {"artifact": "generated_artifacts", "project": "projects", "open_loop": "open_loops", "memory": "memories"} + + +def lock_settled_label_rows(store: Any, changes: Sequence[Mapping[str, object]]) -> None: + """Lock exactly the changed rows in a stable table and UUID order.""" + + if _sqlite(store): + return + grouped: dict[str, set[str]] = {} + for row in changes: + grouped.setdefault(_LABEL_TABLES[str(row["kind"])], set()).add(str(row["id"])) + with store.conn.cursor() as cur: + for table in LABEL_TABLE_ORDER: + ids = sorted(grouped.get(table, ())) + if ids: + cur.execute( + f"SELECT id FROM {table} WHERE id = ANY(%s::uuid[]) ORDER BY id FOR UPDATE", # nosec B608 # closed internal table map + (ids,), + ) + locked = cur.fetchall() + if len(locked) != len(ids): + raise DerivedDomainRepairError("a planned label row disappeared before it could be locked") def propagate(store: Any, changed: Sequence[tuple[str, str]], *, cause: str) -> int: """Recompute dependants of ``changed`` rows and write the ones that rise.""" - store.lock_label_writes(exclusive=True) + require_exclusive_label_lock(store) roots = [row_id for _kind, row_id in changed] affected = walk_dependants(store, roots) if not affected: @@ -561,7 +644,7 @@ def propagate(store: Any, changed: Sequence[tuple[str, str]], *, cause: str) -> if exceeded: raise LabelPropagationTooLarge(f"label propagation stopped after {PROPAGATION_BOUND} rows") settled = settle_labels(nodes) - written = 0 + changes: list[tuple[Mapping[str, object], Any, tuple[str, str, tuple[str, ...], tuple[str, ...]]]] = [] for row in affected: label = settled.by_stored(str(row.get("kind")), str(row.get("id"))) if label.unverified: @@ -586,6 +669,10 @@ def propagate(store: Any, changed: Sequence[tuple[str, str]], *, cause: str) -> and project_scope_identity(previous[3]) == project_scope_identity(current[3]) ): continue + changes.append((row, label, previous)) + lock_settled_label_rows(store, [row for row, _label, _previous in changes]) + written = 0 + for row, label, previous in sorted(changes, key=lambda item: (LABEL_TABLE_ORDER.index(_LABEL_TABLES[str(item[0]["kind"])]), str(item[0]["id"]))): raw_metadata = row.get("metadata_json") metadata = dict(raw_metadata) if isinstance(raw_metadata, Mapping) else {} metadata["project_scope"] = list(label.project_scope) @@ -660,15 +747,21 @@ def count_rows_hidden_by_scope_move(store: Any, source: Mapping[str, object], ne metadata = dict(raw_metadata) if isinstance(raw_metadata, Mapping) else {} metadata["project_scope"] = list(new_scope) moved["metadata_json"] = metadata - before = settle_labels([current, *[dict(row) for row in affected]]) - after = settle_labels([moved, *[dict(row) for row in affected]]) + nodes, exceeded = collect_label_rows(store, [current, *[dict(row) for row in affected]], max_nodes=PROPAGATION_BOUND) + if exceeded: + raise LabelPropagationTooLarge("the source move preview exceeded the label propagation bound") + before = settle_labels(nodes, on_cycle="unverified") + moved_nodes = [moved if str(row.get("kind")) == "source" and str(row.get("id")) == source_id else row for row in nodes] + after = settle_labels(moved_nodes, on_cycle="unverified") hidden = 0 for row in affected: old = before.by_stored(str(row.get("kind")), str(row.get("id"))) new = after.by_stored(str(row.get("kind")), str(row.get("id"))) - old_ids = set(project_scope_identity(old.project_scope)) - new_ids = set(project_scope_identity(new.project_scope)) - if old_ids - new_ids: + if old.unverified or not old.project_scope: + continue + binding = project_scope_identity([*old.project_scope, *old.project_floor]) + new_row = {**row, "metadata_json": {"project_scope": list(new.project_scope), "project_floor": list(new.project_floor)}} + if new.unverified or not new.project_scope or not input_admitted(str(row["kind"]), new_row, binding): hidden += 1 return hidden @@ -695,10 +788,17 @@ def raise_source_to_replacement(store: Any, old: Mapping[str, object], replaceme def label_error_response(exc: BaseException) -> tuple[int, str, str | None] | None: """``(status, detail, retry_after)`` for a relabel failure, or None.""" - if isinstance(exc, (LabelPropagationTooLarge, DerivedDomainRepairError)): - return 409, REFUSED_DETAIL, None + if isinstance(exc, LabelPropagationTooLarge): + return 409, REFUSED_DETAIL + "; cause: propagation_bound", None + if isinstance(exc, DerivedDomainRepairError): + cause = "row_changed" if "changed no row" in str(exc) or "disappeared" in str(exc) else "dependency_cycle" + return 409, REFUSED_DETAIL + "; cause: " + cause, None + if isinstance(exc, LabelLockOrderError): + return 409, REFUSED_DETAIL + "; cause: lock_order", None if type(exc).__name__ in {"LockNotAvailable", "DeadlockDetected", "SerializationFailure"}: return 503, RETRYABLE_DETAIL, "2" + if type(exc).__module__.startswith("psycopg"): + return 409, REFUSED_DETAIL + "; cause: database_error", None return None diff --git a/apps/api/src/alicebot_api/vnext_projects.py b/apps/api/src/alicebot_api/vnext_projects.py index b34b1a88b..8f0785c92 100644 --- a/apps/api/src/alicebot_api/vnext_projects.py +++ b/apps/api/src/alicebot_api/vnext_projects.py @@ -824,6 +824,12 @@ def review_project_update( ) -> JsonObject: if action not in PROJECT_UPDATE_ACTIONS: raise VNextProjectValidationError("project update action must be accept, edit, or reject") + lock_graph = getattr(self.store, "lock_graph_mutation", None) + if callable(lock_graph): + lock_graph() + from alicebot_api.vnext_label_writes import acquire_exclusive_label_lock + + acquire_exclusive_label_lock(self.store) # The artifact is the review decision's serialization point. Every # accept/edit/reject path must inspect and transition the same locked # row so stale reviewers cannot split project, memory, and artifact diff --git a/apps/api/src/alicebot_api/vnext_store.py b/apps/api/src/alicebot_api/vnext_store.py index b8d7572ca..788b172e1 100644 --- a/apps/api/src/alicebot_api/vnext_store.py +++ b/apps/api/src/alicebot_api/vnext_store.py @@ -480,6 +480,10 @@ def lock_label_writes(self, *, exclusive: bool = False) -> None: cur.execute( f"SELECT {mode}(hashtext('vnext_labels'), hashtext(app.current_user_id()::text))" ) + from alicebot_api.vnext_label_writes import _in_transaction, LabelLockOrderError + + if not _in_transaction(self): + raise LabelLockOrderError("label writes require an open transaction") def read_label_rows(self, kind: str, ids: Sequence[str]) -> list[VNextRow]: """Narrow label rows for the insert floor. No text columns.""" @@ -1250,6 +1254,9 @@ def get_source_by_dedupe_key(self, dedupe_key: str) -> VNextRow | None: @takes_label_lock def update_source(self, *, source_id: str, patch: JsonObject, actor_type: str = "system") -> VNextRow: + from alicebot_api.vnext_label_writes import prepare_label_patch + + patch = prepare_label_patch(self, "source", self.get_source(source_id), patch) with self.conn.cursor() as cur: cur.execute( f""" @@ -1723,6 +1730,9 @@ def search_sources( @takes_label_lock def create_project(self, project: JsonObject, *, actor_type: str = "system") -> VNextRow: + from alicebot_api.vnext_label_writes import apply_insert_floor, remember_floor_event + + project, floor_event = apply_insert_floor(self, "project", project) row = self._fetch_one( "create_project", f""" @@ -1771,6 +1781,7 @@ def create_project(self, project: JsonObject, *, actor_type: str = "system") -> target_id=row["id"], payload={"operation": "create", "fields": _sorted_field_names(project)}, ) + remember_floor_event(self, floor_event, row["id"]) return row def get_project(self, project_id: str) -> VNextRow | None: @@ -1843,6 +1854,26 @@ def list_projects( @takes_label_lock def update_project(self, *, project_id: str, patch: JsonObject, actor_type: str = "system") -> VNextRow: + from alicebot_api.vnext_label_writes import ( + apply_insert_floor, merge_protected_metadata, prepare_label_patch, + propagate_after_write, remember_floor_event, + ) + + before = self.get_project(project_id) + patch = dict(patch) + metadata = patch.get("metadata_json") + floor_event = None + if isinstance(metadata, dict) and before is not None: + patch["metadata_json"] = merge_protected_metadata( + before.get("metadata_json") if isinstance(before.get("metadata_json"), dict) else {}, + metadata, + label_write="derived_from" in metadata, + ) + if "derived_from" in metadata: + floored, floor_event = apply_insert_floor(self, "project", {**before, **patch}) + for key in ("domain", "sensitivity", "metadata_json"): + patch[key] = floored[key] + patch = prepare_label_patch(self, "project", before, patch) row = self._fetch_one( "update_project", f""" @@ -1876,6 +1907,8 @@ def update_project(self, *, project_id: str, patch: JsonObject, actor_type: str target_id=row["id"], payload={"operation": "update", "changes": patch}, ) + remember_floor_event(self, floor_event, row["id"]) + propagate_after_write(self, kind="project", before=before, after=row) return row def create_person(self, person: JsonObject, *, actor_type: str = "system") -> VNextRow: diff --git a/apps/api/src/alicebot_api/vnext_stores/postgres/graph_open_loops.py b/apps/api/src/alicebot_api/vnext_stores/postgres/graph_open_loops.py index 3c4faed08..d20c0a152 100644 --- a/apps/api/src/alicebot_api/vnext_stores/postgres/graph_open_loops.py +++ b/apps/api/src/alicebot_api/vnext_stores/postgres/graph_open_loops.py @@ -1206,6 +1206,18 @@ def list_open_loop_events( @takes_label_lock def update_open_loop(self, *, loop_id: str, patch: JsonObject, actor_type: str = "system") -> VNextRow: + from alicebot_api.vnext_label_writes import clamp_owner_patch, merge_protected_metadata, prepare_label_patch, propagate_after_write + + before = self.get_open_loop(loop_id) + patch = dict(patch) + metadata = patch.get("metadata_json") + if before is not None and isinstance(metadata, dict): + patch["metadata_json"] = merge_protected_metadata( + before.get("metadata_json") if isinstance(before.get("metadata_json"), dict) else {}, + metadata, label_write=False, + ) + patch = prepare_label_patch(self, "open_loop", before, patch) + patch = clamp_owner_patch(self, kind="open_loop", before=before, patch=patch) row = self._fetch_one( "update_open_loop", f""" @@ -1243,6 +1255,7 @@ def update_open_loop(self, *, loop_id: str, patch: JsonObject, actor_type: str = target_id=row["id"], payload={"operation": "update", "changes": patch}, ) + propagate_after_write(self, kind="open_loop", before=before, after=row) return row @takes_label_lock diff --git a/apps/api/src/alicebot_api/vnext_stores/postgres/memory_access.py b/apps/api/src/alicebot_api/vnext_stores/postgres/memory_access.py index ac34bb114..f2d6c1df5 100644 --- a/apps/api/src/alicebot_api/vnext_stores/postgres/memory_access.py +++ b/apps/api/src/alicebot_api/vnext_stores/postgres/memory_access.py @@ -7,6 +7,7 @@ from typing import cast from alicebot_api.store import ContinuityStoreInvariantError +from alicebot_api.vnext_label_writes import takes_label_lock from alicebot_api.vnext_embeddings import ( EMBEDDING_SIGNATURE_METADATA_KEY, memory_embedding_signature_is_current, @@ -216,6 +217,7 @@ def list_memories_referencing_sources( return grouped +@takes_label_lock def list_pending_derived_candidates_for_member( self, *, diff --git a/apps/api/src/alicebot_api/vnext_stores/postgres/memory_lifecycle.py b/apps/api/src/alicebot_api/vnext_stores/postgres/memory_lifecycle.py index 64294a5be..0287fe060 100644 --- a/apps/api/src/alicebot_api/vnext_stores/postgres/memory_lifecycle.py +++ b/apps/api/src/alicebot_api/vnext_stores/postgres/memory_lifecycle.py @@ -362,6 +362,9 @@ def lock_graph_mutation(self) -> None: correction, forgetting, and transitions. Released automatically at commit/rollback. """ + from alicebot_api.vnext_label_writes import before_graph_lock + + before_graph_lock(self) with self.conn.cursor() as cur: cur.execute( "SELECT pg_advisory_xact_lock(hashtext('vnext_supersession'), hashtext(app.current_user_id()::text))" @@ -451,8 +454,9 @@ def update_memory( label_write=label_write, ) if before_label is not None: - from alicebot_api.vnext_label_writes import clamp_owner_patch + from alicebot_api.vnext_label_writes import clamp_owner_patch, prepare_label_patch + patch = prepare_label_patch(self, "memory", before_label, patch) patch = clamp_owner_patch(self, kind="memory", before=before_label, patch=patch) row = self._fetch_one( "update_memory", diff --git a/apps/api/src/alicebot_api/vnext_stores/sqlite/graph_open_loops.py b/apps/api/src/alicebot_api/vnext_stores/sqlite/graph_open_loops.py index cc43135e1..25514f3de 100644 --- a/apps/api/src/alicebot_api/vnext_stores/sqlite/graph_open_loops.py +++ b/apps/api/src/alicebot_api/vnext_stores/sqlite/graph_open_loops.py @@ -977,6 +977,18 @@ def list_open_loop_events( @takes_label_lock def update_open_loop(self, *, loop_id: str, patch: JsonObject, actor_type: str = "system") -> VNextRow: + from alicebot_api.vnext_label_writes import clamp_owner_patch, merge_protected_metadata, prepare_label_patch, propagate_after_write + + before = self.get_open_loop(loop_id) + patch = dict(patch) + metadata = patch.get("metadata_json") + if before is not None and isinstance(metadata, dict): + patch["metadata_json"] = merge_protected_metadata( + before.get("metadata_json") if isinstance(before.get("metadata_json"), dict) else {}, + metadata, label_write=False, + ) + patch = prepare_label_patch(self, "open_loop", before, patch) + patch = clamp_owner_patch(self, kind="open_loop", before=before, patch=patch) cursor = self._execute( """ UPDATE open_loops @@ -1020,6 +1032,7 @@ def update_open_loop(self, *, loop_id: str, patch: JsonObject, actor_type: str = target_id=row["id"], payload={"operation": "update", "changes": patch}, ) + propagate_after_write(self, kind="open_loop", before=before, after=row) return row @takes_label_lock diff --git a/tests/integration/conftest.py b/tests/integration/conftest.py index 0a8f8edf2..404554c6e 100644 --- a/tests/integration/conftest.py +++ b/tests/integration/conftest.py @@ -20,6 +20,15 @@ TEMPLATE_MIGRATION_COUNT = 0 +@pytest.fixture(autouse=True) +def strict_label_lock_order(monkeypatch: pytest.MonkeyPatch) -> None: + """Every integration flow obeys S, L, then label-row locks.""" + + import alicebot_api.vnext_label_writes as label_writes + + monkeypatch.setattr(label_writes, "STRICT_LOCK_ORDER", True) + + def pytest_addoption(parser: pytest.Parser) -> None: parser.addoption( "--require-executed-tests", diff --git a/tests/integration/test_label_lock_order_postgres.py b/tests/integration/test_label_lock_order_postgres.py new file mode 100644 index 000000000..db9215843 --- /dev/null +++ b/tests/integration/test_label_lock_order_postgres.py @@ -0,0 +1,63 @@ +"""Live advisory grants enforce S before L and survive savepoint rollback.""" +from uuid import uuid4 + +import pytest + +from alicebot_api.db import user_connection +from alicebot_api.store import ContinuityStore +from alicebot_api.vnext_store import PostgresVNextStore +from alicebot_api.vnext_label_writes import LabelLockOrderError, held_label_locks + + +def _user(url, user): + with user_connection(url, user) as conn: + ContinuityStore(conn).create_user(user, f"locks-{user}@example.test", "Synthetic") + + +def test_real_pg_strict_s_after_l_is_refused(migrated_database_urls): + user = uuid4() + url = migrated_database_urls["app"] + _user(url, user) + with user_connection(url, user) as conn: + store = PostgresVNextStore(conn) + store.lock_label_writes() + assert held_label_locks(store) == (False, True, False) + with pytest.raises(LabelLockOrderError, match="graph lock must precede"): + store.lock_graph_mutation() + + +def test_real_pg_savepoint_rollback_releases_live_grants(migrated_database_urls): + user = uuid4() + url = migrated_database_urls["app"] + _user(url, user) + with user_connection(url, user) as conn: + store = PostgresVNextStore(conn) + with pytest.raises(ValueError, match="rollback"): + with conn.transaction(): + store.lock_graph_mutation() + store.lock_label_writes(exclusive=True) + assert held_label_locks(store) == (True, True, True) + raise ValueError("rollback") + assert held_label_locks(store) == (False, False, False) + store.lock_graph_mutation() + store.lock_label_writes() + assert held_label_locks(store) == (True, True, False) + + +def test_real_pg_changing_hook_needs_exclusive_before_the_update(migrated_database_urls): + user = uuid4() + url = migrated_database_urls["app"] + _user(url, user) + with user_connection(url, user) as conn: + store = PostgresVNextStore(conn) + row = store.create_memory({"memory_key": "original", "canonical_text": "synthetic", "domain": "project", "sensitivity": "public"}) + with user_connection(url, user) as conn: + store = PostgresVNextStore(conn) + with pytest.raises(LabelLockOrderError, match="exclusive label lock"): + store.update_memory(memory_id=str(row["id"]), patch={"sensitivity": "confidential"}) + assert store.get_memory(str(row["id"]))["sensitivity"] == "public" + with user_connection(url, user) as conn: + store = PostgresVNextStore(conn) + store.lock_graph_mutation() + store.lock_label_writes(exclusive=True) + assert store.update_memory(memory_id=str(row["id"]), patch={"sensitivity": "confidential"})["sensitivity"] == "confidential" diff --git a/tests/unit/test_label_lock_order.py b/tests/unit/test_label_lock_order.py new file mode 100644 index 000000000..5ac522ddc --- /dev/null +++ b/tests/unit/test_label_lock_order.py @@ -0,0 +1,86 @@ +"""Strict lock checks read database grants rather than cached state.""" +from types import SimpleNamespace + +import pytest + +from alicebot_api import vnext_label_writes as writes +from alicebot_api.vnext_store import PostgresVNextStore + + +class Cursor: + def __init__(self, conn): + self.conn = conn + def __enter__(self): + return self + def __exit__(self, *args): + return None + def execute(self, query, params=()): + self.conn.queries.append(query) + if "pg_advisory_xact_lock_shared" in query: + self.conn.grants["labels"] = True + elif "pg_advisory_xact_lock(" in query: + self.conn.grants["graph"] = True + def fetchone(self): + return dict(self.conn.grants) + + +def _store(): + conn = SimpleNamespace(grants={"graph": False, "labels": False, "exclusive": False}, queries=[]) + conn.cursor = lambda: Cursor(conn) + return PostgresVNextStore(conn) + + +def test_strict_s_after_l_is_refused(monkeypatch): + monkeypatch.setattr(writes, "STRICT_LOCK_ORDER", True) + store = _store() + store.lock_label_writes() + with pytest.raises(writes.LabelLockOrderError, match="graph lock must precede"): + store.lock_graph_mutation() + assert any("FROM pg_locks" in query for query in store.conn.queries) + + +def test_a_savepoint_rollback_cannot_leave_a_stale_lock_memo(monkeypatch): + monkeypatch.setattr(writes, "STRICT_LOCK_ORDER", True) + store = _store() + store.lock_graph_mutation() + store.lock_label_writes() + store.conn.grants.update(graph=False, labels=False, exclusive=False) + store.lock_graph_mutation() + store.conn.grants.update(graph=False, labels=True) + with pytest.raises(writes.LabelLockOrderError): + store.lock_graph_mutation() + assert sum("FROM pg_locks" in query for query in store.conn.queries) == 3 + + +def test_strict_changing_hook_requires_exclusive_l(monkeypatch): + monkeypatch.setattr(writes, "STRICT_LOCK_ORDER", True) + store = _store() + before = {"domain": "project", "sensitivity": "public"} + with pytest.raises(writes.LabelLockOrderError, match="exclusive label lock"): + writes.prepare_label_patch(store, "memory", before, {"sensitivity": "confidential"}) + store.conn.grants["exclusive"] = True + assert writes.prepare_label_patch(store, "memory", before, {"sensitivity": "confidential"}) == {"sensitivity": "confidential"} + + +def test_propagation_locks_tables_and_rows_in_order(): + store = _store() + statements = [] + class RowsCursor(Cursor): + def execute(self, query, params=()): + statements.append((query, params)) + self.ids = params[0] + def fetchall(self): + return [{"id": row_id} for row_id in self.ids] + store.conn.cursor = lambda: RowsCursor(store.conn) + changes = [{"kind": kind, "id": row_id} for kind, row_id in [("memory", "b"), ("artifact", "c"), ("open_loop", "d"), ("project", "e"), ("memory", "a")]] + writes.lock_settled_label_rows(store, changes) + assert [query.split("FROM ")[1].split()[0] for query, _params in statements] == list(writes.LABEL_TABLE_ORDER) + assert all("ORDER BY id FOR UPDATE" in query for query, _params in statements) + assert statements[-1][1] == (["a", "b"],) + + +def test_compare_and_set_miss_refuses_the_whole_label_write(): + from alicebot_api.vnext_derived_domain_backfill import DerivedDomainRepairError + store = SimpleNamespace(_fetch_optional_one=lambda *_: None) + with pytest.raises(DerivedDomainRepairError, match="changed no row"): + writes.write_settled_label(store, kind="memory", row_id="missing", domain="health", sensitivity="confidential", metadata={}, project_id=None, expected_domain="project", expected_sensitivity="public") diff --git a/tests/unit/test_label_lock_registry.py b/tests/unit/test_label_lock_registry.py new file mode 100644 index 000000000..8e73f72b8 --- /dev/null +++ b/tests/unit/test_label_lock_registry.py @@ -0,0 +1,69 @@ +"""Discover every store SQL write or row lock on a label table.""" +import ast +import re +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[2] / "apps/api/src/alicebot_api" +TABLE = r"(?:memories|sources|open_loops|generated_artifacts|projects)" +WRITE = re.compile(rf"\b(?:INSERT\s+INTO|UPDATE|DELETE\s+FROM)\s+{TABLE}\b", re.I) +FROM = re.compile(rf"\b(?:FROM|JOIN)\s+{TABLE}\b", re.I) + + +def sql_text(node): + if isinstance(node, ast.JoinedStr): + return " ".join(item.value if isinstance(item, ast.Constant) and isinstance(item.value, str) else "{}" for item in node.values) + return node.value if isinstance(node, ast.Constant) and isinstance(node.value, str) else "" + + +def label_sql_functions(tree): + for fn in ast.walk(tree): + if not isinstance(fn, (ast.FunctionDef, ast.AsyncFunctionDef)): + continue + for node in ast.walk(fn): + query = " ".join(sql_text(node).split()) + if WRITE.search(query) or ("FOR UPDATE" in query.upper() and FROM.search(query)): + yield fn, node + break + + +def test_all_discovered_label_writers_and_row_lockers_take_l(): + paths = [ROOT / "vnext_store.py", *sorted((ROOT / "vnext_stores/postgres").glob("*.py"))] + found = set() + missing = [] + for path in paths: + for fn, node in label_sql_functions(ast.parse(path.read_text())): + found.add((path.name, fn.name)) + if not any(isinstance(dec, ast.Name) and dec.id == "takes_label_lock" for dec in fn.decorator_list): + missing.append(f"{path.name}:{node.lineno} {fn.name}") + assert ("memory_access.py", "list_pending_derived_candidates_for_member") in found + assert ("memory_lifecycle.py", "lock_project_update_artifacts_for_redaction") in found + assert not missing, "label SQL without L: " + ", ".join(missing) + + +# Each application SQL exception has a dedicated transaction protocol. +DIRECT_SQL_PROTOCOLS = { + ("vnext_label_writes.py", "write_settled_label"): "exclusive L and label compare-and-set", + ("vnext_label_writes.py", "lock_settled_label_rows"): "exclusive L and deterministic table order", + ("vnext_label_repair.py", "relabel_labels_sqlite"): "SQLite immediate writer transaction", + ("vnext_derived_domain_backfill.py", "relabel_derived_rows_sqlite"): "frozen historical SQLite repair", + ("labels.py", "repair_labels_postgres"): "S, exclusive L, ordered rows and compare-and-set", + ("sqlite_schema.py", "_backfill_legacy_memory_project_scopes"): "schema bootstrap writer transaction", + ("sqlite_schema.py", "_backfill_source_dedupe_keys"): "schema bootstrap writer transaction", + ("sqlite_schema.py", "_repair_source_dedupe_identity"): "schema bootstrap writer transaction", + ("sqlite_schema.py", "_backfill_memory_agent_attribution"): "schema bootstrap writer transaction", + ("sqlite_schema.py", "_deduplicate_memory_lookup_values"): "schema bootstrap writer transaction", + ("sqlite_schema.py", "_repair_tombstone_lookup_value_holders"): "schema bootstrap writer transaction", +} + + +def test_application_code_cannot_write_label_tables_directly(): + missing = [] + for path in ROOT.rglob("*.py"): + if path.name in {"vnext_store.py", "sqlite_store.py", "store.py"} or "vnext_stores" in path.parts or "legacy_store" in path.parts: + continue + for fn, node in label_sql_functions(ast.parse(path.read_text())): + if any(isinstance(dec, ast.Name) and dec.id in {"takes_label_lock", "_takes_label_lock"} for dec in fn.decorator_list): + continue + if (path.name, fn.name) not in DIRECT_SQL_PROTOCOLS: + missing.append(f"{path.relative_to(ROOT)}:{node.lineno} {fn.name}") + assert not missing, "application label SQL without protocol: " + ", ".join(missing) diff --git a/tests/unit/test_label_writer_registry.py b/tests/unit/test_label_writer_registry.py new file mode 100644 index 000000000..e047e7e5a --- /dev/null +++ b/tests/unit/test_label_writer_registry.py @@ -0,0 +1,78 @@ +"""Discover label-changing SQL, update patches and the derived insert floors.""" +import ast +import re +from pathlib import Path + +from tests.unit.test_label_lock_registry import ROOT, label_sql_functions, sql_text + +LABEL_KEYS = {"domain", "sensitivity", "project_id", "project_scope", "project_floor", "derived_from"} +UPDATE_METHODS = {"update_memory", "update_source", "update_open_loop", "update_project"} +# Explicit function identities make every newly added caller reviewable. +LABEL_CALLERS = { + ("routers/vnext_memories.py", "review_vnext_memory"), + ("routers/vnext_memories.py", "review_vnext_source"), + ("vnext_projects.py", "review_project_update"), + ("vnext_label_writes.py", "raise_source_to_replacement"), + ("sqlite_store.py", "supersede_source"), + ("cli/smokes.py", "_run_vnext_smoke_operator_console"), +} + + +def keys_in(node): + return {item.value for item in ast.walk(node) if isinstance(item, ast.Constant) and isinstance(item.value, str)} & LABEL_KEYS + + +def test_every_label_changing_update_caller_is_registered(): + missing = [] + for path in ROOT.rglob("*.py"): + if "vnext_stores" in path.parts or path.name in {"sqlite_store.py", "vnext_store.py"}: + continue + tree = ast.parse(path.read_text()) + for fn in ast.walk(tree): + if not isinstance(fn, (ast.FunctionDef, ast.AsyncFunctionDef)): + continue + patch_nodes = [node for node in ast.walk(fn) if isinstance(node, (ast.Assign, ast.AnnAssign)) and any(isinstance(item, ast.Name) and "patch" in item.id for item in ast.walk(node))] + for call in ast.walk(fn): + if not isinstance(call, ast.Call) or not isinstance(call.func, ast.Attribute) or call.func.attr not in UPDATE_METHODS: + continue + patch = next((kw.value for kw in call.keywords if kw.arg == "patch"), None) + if patch is not None and (keys_in(patch) or (isinstance(patch, ast.Name) and any(keys_in(item) for item in patch_nodes))): + key = (str(path.relative_to(ROOT)), fn.name) + if key not in LABEL_CALLERS: + missing.append(f"{key[0]}:{call.lineno} {fn.name}") + assert not missing, "unregistered label caller: " + ", ".join(missing) + + +def test_every_create_has_the_insert_floor_and_every_update_has_its_hook(): + floors = { + "vnext_store.py": {"create_artifact", "upsert_artifact_by_workflow_digest", "create_project", "update_project"}, + "vnext_stores/postgres/memory_lifecycle.py": {"create_memory"}, + "vnext_stores/postgres/graph_open_loops.py": {"create_open_loop"}, + "vnext_stores/sqlite/memory_lifecycle.py": {"create_memory"}, + "vnext_stores/sqlite/graph_open_loops.py": {"create_open_loop"}, + } + hooks = { + "vnext_store.py": {"update_source", "update_project"}, + "sqlite_store.py": {"update_source"}, + "vnext_stores/postgres/memory_lifecycle.py": {"update_memory"}, + "vnext_stores/postgres/graph_open_loops.py": {"update_open_loop"}, + "vnext_stores/sqlite/memory_lifecycle.py": {"update_memory"}, + "vnext_stores/sqlite/graph_open_loops.py": {"update_open_loop"}, + } + for registry, helper in ((floors, "apply_insert_floor"), (hooks, "propagate_after_write")): + for name, functions in registry.items(): + tree = ast.parse((ROOT / name).read_text()) + for function in functions: + fn = next(node for node in ast.walk(tree) if isinstance(node, ast.FunctionDef) and node.name == function) + calls = {node.func.id for node in ast.walk(fn) if isinstance(node, ast.Call) and isinstance(node.func, ast.Name)} + assert helper in calls, f"{name}:{fn.lineno} {function} lacks {helper}" + + +def test_a_new_derived_insert_cannot_bypass_the_floor(): + for path in [ROOT / "vnext_store.py", ROOT / "sqlite_store.py", *sorted((ROOT / "vnext_stores").rglob("*.py"))]: + for fn, node in label_sql_functions(ast.parse(path.read_text())): + inserts = [sql_text(item) for item in ast.walk(fn)] + if not any(re.search(r"\bINSERT\s+INTO\s+(?:memories|open_loops|generated_artifacts|projects)\b", query, re.I) for query in inserts): + continue + calls = {item.func.id for item in ast.walk(fn) if isinstance(item, ast.Call) and isinstance(item.func, ast.Name)} + assert "apply_insert_floor" in calls, f"{path.name}:{node.lineno} {fn.name} inserts without a floor" diff --git a/tests/unit/test_source_move_label_preview.py b/tests/unit/test_source_move_label_preview.py new file mode 100644 index 000000000..cb6caa1ce --- /dev/null +++ b/tests/unit/test_source_move_label_preview.py @@ -0,0 +1,53 @@ +"""A source move previews loss of full project admission before any write.""" +from copy import deepcopy +from uuid import uuid4 + +import pytest + +from alicebot_api import vnext_label_writes as writes + + +def _source(scope): + return {"id": str(uuid4()), "kind": "source", "user_id": "synthetic", "domain": "project", "sensitivity": "public", "metadata_json": {"project_scope": scope}} + + +def _report(sources, scope): + return {"id": str(uuid4()), "kind": "artifact", "user_id": "synthetic", "artifact_type": "daily_brief", "domain": "project", "sensitivity": "public", "metadata_json": {"source_ids": [row["id"] for row in sources], "project_scope": scope}} + + +class Store: + def __init__(self, rows): + self.rows = rows + self.reads = [] + def read_label_rows(self, kind, ids): + self.reads.append((kind, ids)) + return [row for row in self.rows if row["kind"] == kind and row["id"] in ids] + + +def test_stored_scope_unchanged_floor_move_still_counts_hidden_row(monkeypatch): + source = _source(["alpha"]) + report = _report([source], ["alpha"]) + store = Store([source, report]) + original = deepcopy(store.rows) + monkeypatch.setattr(writes, "walk_dependants", lambda *_: [report]) + assert writes.count_rows_hidden_by_scope_move(store, source, ["beta"]) == 1 + assert store.rows == original + + +def test_preview_reads_the_other_parent_and_its_ancestry(monkeypatch): + moved, other = _source(["alpha"]), _source(["alpha"]) + parent = _report([other], ["alpha"]) + report = _report([moved], ["alpha"]) + report["metadata_json"]["artifact_ids"] = [parent["id"]] + store = Store([moved, other, parent, report]) + monkeypatch.setattr(writes, "walk_dependants", lambda *_: [report]) + assert writes.count_rows_hidden_by_scope_move(store, moved, ["beta"]) == 1 + assert any(other["id"] in ids for _kind, ids in store.reads) + assert any(parent["id"] in ids for _kind, ids in store.reads) + + +def test_preview_does_not_count_a_row_already_unverified(monkeypatch): + source = _source(["alpha"]) + report = _report([source, _source(["alpha"])], ["alpha"]) + monkeypatch.setattr(writes, "walk_dependants", lambda *_: [report]) + assert writes.count_rows_hidden_by_scope_move(Store([source, report]), source, ["beta"]) == 0 diff --git a/tests/unit/test_sqlite_derived_labels_write_path.py b/tests/unit/test_sqlite_derived_labels_write_path.py index 3bfd61cca..ad231c68a 100644 --- a/tests/unit/test_sqlite_derived_labels_write_path.py +++ b/tests/unit/test_sqlite_derived_labels_write_path.py @@ -76,7 +76,7 @@ def test_an_update_that_omits_a_marker_keeps_it(tmp_path: Path) -> None: path = tmp_path / "vault.sqlite3" with _vault(path) as conn: store = SQLiteVNextStore(conn, USER) - health = add_memory(store, key="health", text="A restricted observation", domain="health") + health = add_memory(store, key="health", text="A restricted observation", domain="health", scope=(ALPHA,)) with without_insert_floor(): derived = store.create_memory( { From 6bf053b3267f88fd82898ea412557415fc04eb57 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 23:27:25 +0200 Subject: [PATCH 51/90] Regenerate source candidates at current labels and lock review adapters first --- apps/api/src/alicebot_api/main.py | 1 + .../alicebot_api/mcp/evidence_artifacts.py | 4 + .../openapi_operation_contracts.py | 7 + .../alicebot_api/routers/vnext_memories.py | 32 ++++- .../src/alicebot_api/routers/vnext_review.py | 4 + .../src/alicebot_api/vnext_label_writes.py | 10 +- .../alicebot_api/vnext_source_regeneration.py | 73 ++++++++++ apps/api/src/alicebot_api/vnext_store.py | 17 ++- ...test_source_move_label_preview_postgres.py | 128 ++++++++++++++++++ 9 files changed, 270 insertions(+), 6 deletions(-) create mode 100644 apps/api/src/alicebot_api/vnext_source_regeneration.py create mode 100644 tests/integration/test_source_move_label_preview_postgres.py diff --git a/apps/api/src/alicebot_api/main.py b/apps/api/src/alicebot_api/main.py index 2b2be310a..1378f61a4 100644 --- a/apps/api/src/alicebot_api/main.py +++ b/apps/api/src/alicebot_api/main.py @@ -779,6 +779,7 @@ async def receive() -> dict[str, object]: ("POST", "/v0/vnext/projects/update-candidates/{artifact_id}/review"), ("POST", "/v0/vnext/queue/process-next"), ("POST", "/v0/vnext/sources/{source_id}/review"), + ("POST", "/v0/vnext/sources/{source_id}/regenerate"), ("PUT", "/v0/vnext/settings/brain-charter"), } ) diff --git a/apps/api/src/alicebot_api/mcp/evidence_artifacts.py b/apps/api/src/alicebot_api/mcp/evidence_artifacts.py index 0a384563a..c6efae11f 100644 --- a/apps/api/src/alicebot_api/mcp/evidence_artifacts.py +++ b/apps/api/src/alicebot_api/mcp/evidence_artifacts.py @@ -791,6 +791,10 @@ def _handle_alice_vnext_artifact_review(context: MCPRuntimeContext, arguments: M actor_id: str | None = None trace_id: str | None = None with _vnext_store_context(context) as store: + store.lock_graph_mutation() + from alicebot_api.vnext_label_writes import acquire_exclusive_label_lock + + acquire_exclusive_label_lock(store) _target, actor_type, actor_id, decision = _authorize_vnext_artifact_target( store, identity=identity, diff --git a/apps/api/src/alicebot_api/openapi_operation_contracts.py b/apps/api/src/alicebot_api/openapi_operation_contracts.py index edf9d9e61..e04734f5c 100644 --- a/apps/api/src/alicebot_api/openapi_operation_contracts.py +++ b/apps/api/src/alicebot_api/openapi_operation_contracts.py @@ -1148,6 +1148,10 @@ def _closed_source_schema( closed=True, ), ), + ("POST", "/v0/vnext/sources/{source_id}/regenerate"): ( + "RegenerateVnextSourceSuccessResponse", + _operation_schema("RegenerateVnextSourceSuccessResponse", ("source_id", "memory_ids", "open_loop_ids", "memory_count", "open_loop_count"), closed=True), + ), ("GET", "/v0/vnext/traces/sources/{source_id}"): ( "GetVnextSourceTraceSuccessResponse", _operation_schema( @@ -2082,6 +2086,9 @@ def _closed_source_schema( ("POST", "/v0/vnext/sources/{source_id}/review"): _typed_properties( objects=("source", "trace"), booleans=("archived",) ), + ("POST", "/v0/vnext/sources/{source_id}/regenerate"): _typed_properties( + strings=("source_id",), string_arrays=("memory_ids", "open_loop_ids"), integers=("memory_count", "open_loop_count"), + ), ("POST", "/v0/vnext/memories/{memory_id}/review"): _typed_properties( objects=("memory",), nullable_objects=("consolidation_acceptance",) ), diff --git a/apps/api/src/alicebot_api/routers/vnext_memories.py b/apps/api/src/alicebot_api/routers/vnext_memories.py index b0156174c..c58712977 100644 --- a/apps/api/src/alicebot_api/routers/vnext_memories.py +++ b/apps/api/src/alicebot_api/routers/vnext_memories.py @@ -116,7 +116,11 @@ class VNextSourceReviewRequest(VNextAgentRequest): sensitivity: VNextSensitivity | None = None project_id: str | None = Field(default=None, min_length=1, max_length=120) review_note: str | None = Field(default=None, min_length=1, max_length=4000) - confirm_label_hide: bool = False + confirm_label_hide: bool = Field(default=False, description="Confirm a source project move after previewing the number of derived rows hidden from project-bound keys.") + + +class VNextSourceRegenerateRequest(VNextAgentRequest): + user_id: UUID = Field(description="Owner of the stored source whose candidate memories and open loops are regenerated. Earlier rows keep their labels and provenance.") class VNextConnectorSyncRequest(VNextAgentRequest): @@ -756,6 +760,32 @@ def get_vnext_source(source_id: UUID, user_id: UUID) -> JSONResponse: ) +@source_review_router.post("/v0/vnext/sources/{source_id}/regenerate", summary="Regenerate fresh candidates from a stored source", description="The local owner or an unbound admin can regenerate candidate memories and open loops from all stored chunks using the source's current labels. Existing sources and outputs remain unchanged. Rerun the report's generation route to rebuild a report.") +def regenerate_vnext_source(source_id: UUID, request: VNextSourceRegenerateRequest, authorization: str | None = Header(default=None)) -> JSONResponse: + from alicebot_api.vnext_label_writes import label_error_response + from alicebot_api.vnext_source_regeneration import regenerate_source_inputs + + settings = get_settings() + try: + with user_connection(settings.database_url, request.user_id) as conn: + store = PostgresVNextStore(conn) + identity = _vnext_authenticated_agent_identity(store, request, user_id=request.user_id, authorization=authorization) + if identity is not None and (identity.permission_profile != "admin_agent" or identity.project_scope_locked or identity.project_scope): + return _vnext_public_error_response(status_code=403, detail="source regeneration requires the owner or an unbound admin") + store.lock_label_writes() + source = store.get_source(str(source_id)) + if source is None: + return _vnext_public_error_response(status_code=404, detail="vNext source was not found") + payload = regenerate_source_inputs(store, source) + except Exception as exc: + mapped = label_error_response(exc) + if mapped is None: + raise + status, detail, retry_after = mapped + return JSONResponse(status_code=status, content={"detail": detail}, headers={"Retry-After": retry_after} if retry_after else None) + return JSONResponse(status_code=201, content=jsonable_encoder(payload)) + + @source_review_router.post("/v0/vnext/sources/{source_id}/review") def review_vnext_source(source_id: UUID, request: VNextSourceReviewRequest) -> JSONResponse: settings = get_settings() diff --git a/apps/api/src/alicebot_api/routers/vnext_review.py b/apps/api/src/alicebot_api/routers/vnext_review.py index cd0a9adb1..6ebc447e0 100644 --- a/apps/api/src/alicebot_api/routers/vnext_review.py +++ b/apps/api/src/alicebot_api/routers/vnext_review.py @@ -634,6 +634,10 @@ def review_vnext_artifact( identity = _vnext_authenticated_agent_identity( store, request, user_id=request.user_id, authorization=authorization ) + store.lock_graph_mutation() + from alicebot_api.vnext_label_writes import acquire_exclusive_label_lock + + acquire_exclusive_label_lock(store) _artifact, decision = _vnext_authorized_artifact( store=store, identity=identity, diff --git a/apps/api/src/alicebot_api/vnext_label_writes.py b/apps/api/src/alicebot_api/vnext_label_writes.py index c823feebd..ebed03e80 100644 --- a/apps/api/src/alicebot_api/vnext_label_writes.py +++ b/apps/api/src/alicebot_api/vnext_label_writes.py @@ -487,10 +487,11 @@ def clamp_owner_patch( for key in ("domain", "sensitivity", "project_id"): if key in proposed_patch and proposed_patch[key] is not None: proposed[key] = proposed_patch[key] - if isinstance(proposed_patch.get("metadata_json"), dict): + patch_metadata = proposed_patch.get("metadata_json") + if isinstance(patch_metadata, dict): stored_meta = before.get("metadata_json") meta = dict(stored_meta) if isinstance(stored_meta, dict) else {} - meta.update(proposed_patch["metadata_json"]) + meta.update(patch_metadata) proposed["metadata_json"] = meta proposed["kind"] = kind nodes, exceeded = collect_label_rows(store, [proposed], max_nodes=PROPAGATION_BOUND) @@ -513,7 +514,8 @@ def clamp_owner_patch( return proposed_patch proposed_patch["domain"] = label.domain proposed_patch["sensitivity"] = label.sensitivity - metadata = dict(proposed.get("metadata_json") or {}) + proposed_metadata = proposed.get("metadata_json") + metadata = dict(proposed_metadata) if isinstance(proposed_metadata, Mapping) else {} metadata["project_scope"] = list(label.project_scope) metadata["project_floor"] = list(label.project_floor) proposed_patch["metadata_json"] = metadata @@ -644,7 +646,7 @@ def propagate(store: Any, changed: Sequence[tuple[str, str]], *, cause: str) -> if exceeded: raise LabelPropagationTooLarge(f"label propagation stopped after {PROPAGATION_BOUND} rows") settled = settle_labels(nodes) - changes: list[tuple[Mapping[str, object], Any, tuple[str, str, tuple[str, ...], tuple[str, ...]]]] = [] + changes: list[tuple[dict[str, object], Any, tuple[str, str, tuple[str, ...], tuple[str, ...]]]] = [] for row in affected: label = settled.by_stored(str(row.get("kind")), str(row.get("id"))) if label.unverified: diff --git a/apps/api/src/alicebot_api/vnext_source_regeneration.py b/apps/api/src/alicebot_api/vnext_source_regeneration.py new file mode 100644 index 000000000..589d72010 --- /dev/null +++ b/apps/api/src/alicebot_api/vnext_source_regeneration.py @@ -0,0 +1,73 @@ +"""Regenerate fresh source copies without changing earlier rows or their provenance.""" +from __future__ import annotations + +from typing import Any +from uuid import UUID, uuid4 + +from alicebot_api.vnext_capture import extract_candidate_memories +from alicebot_api.vnext_derived_labels import with_derived_from +from alicebot_api.vnext_event_log import append_event +from alicebot_api.vnext_project_scope import source_project_scope +from alicebot_api.vnext_projects import _open_loop_candidates +from alicebot_api.vnext_repositories import JsonObject + + +def regenerate_source_inputs(store: Any, source: JsonObject) -> JsonObject: + """Create candidate memories and loops from all stored chunks at the current label.""" + + chunks = store.read_source_chunks_for_regeneration(str(source["id"])) + scope = source_project_scope(source) + project_id = None + if len(scope) == 1: + try: + project_id = str(UUID(scope[0])) + except ValueError: + pass + generation = str(uuid4()) + memories = [] + loops = [] + for index, candidate in enumerate(extract_candidate_memories(chunks)): + metadata = with_derived_from({ + "source_id": str(source["id"]), + "source_chunk_id": candidate.source_chunk_id, + "source_chunk_index": candidate.source_chunk_index, + "extraction_rule": candidate.extraction_rule, + "project_scope": list(scope), + "regeneration_id": generation, + **({"provenance_role": candidate.provenance_role, "assertion_class": candidate.assertion_class} if candidate.provenance_role is not None else {}), + }, {"sources": [source]}) + memory = store.create_memory({ + "memory_key": f"regenerated:{generation}:{index}", + "canonical_text": candidate.text, + "title": candidate.text[:120], + "summary": candidate.text[:280], + "value": {"text": candidate.text, "source_id": str(source["id"]), "source_chunk_id": candidate.source_chunk_id}, + "source_event_ids": [str(source["id"]), candidate.source_chunk_id], + "status": "candidate", "memory_type": candidate.memory_type, + "confidence": candidate.confidence, + "domain": source["domain"], "sensitivity": source["sensitivity"], + "project_id": project_id, "metadata_json": metadata, + }, actor_type="user") + store.create_provenance_link({ + "target_type": "memory", "target_id": str(memory["id"]), + "source_id": str(source["id"]), "source_chunk_id": candidate.source_chunk_id, + "quote": candidate.text, "evidence_role": "quoted_from", "confidence": candidate.confidence, + }, actor_type="user") + memories.append(str(memory["id"])) + source_metadata = source.get("metadata_json") + source_with_text = {**source, "metadata_json": {**(source_metadata if isinstance(source_metadata, dict) else {}), "raw_text": "\n".join(str(chunk["text"]) for chunk in chunks)}} + for loop_candidate in _open_loop_candidates(source_with_text): + loop_candidate["project_id"] = project_id + loop_metadata = loop_candidate.get("metadata_json") + loop_candidate["metadata_json"] = with_derived_from({ + **(loop_metadata if isinstance(loop_metadata, dict) else {}), + "source_id": str(source["id"]), "project_scope": list(scope), "regeneration_id": generation, + }, {"sources": [source]}) + loop = store.create_open_loop(loop_candidate, actor_type="user") + store.create_provenance_link({ + "target_type": "open_loop", "target_id": str(loop["id"]), + "source_id": str(source["id"]), "evidence_role": "quoted_from", + }, actor_type="user") + loops.append(str(loop["id"])) + append_event(store, event_type="source.inputs_regenerated", actor_type="user", target_type="source", target_id=str(source["id"]), payload={"memory_count": len(memories), "open_loop_count": len(loops)}) + return {"source_id": str(source["id"]), "memory_ids": memories, "open_loop_ids": loops, "memory_count": len(memories), "open_loop_count": len(loops)} diff --git a/apps/api/src/alicebot_api/vnext_store.py b/apps/api/src/alicebot_api/vnext_store.py index 788b172e1..59aeee2f9 100644 --- a/apps/api/src/alicebot_api/vnext_store.py +++ b/apps/api/src/alicebot_api/vnext_store.py @@ -467,6 +467,7 @@ class PostgresVNextStore: def __init__(self, conn: UserConnection): self.conn = conn + self._label_floor_applied = False def lock_label_writes(self, *, exclusive: bool = False) -> None: """Shared label lock for a write, or the exclusive lock for a relabel. @@ -1452,6 +1453,19 @@ def list_source_chunks(self, source_id: str, *, limit: int = 500) -> list[VNextR (source_id, bounded_limit), ) + def read_source_chunks_for_regeneration(self, source_id: str) -> list[VNextRow]: + """Read the complete source, or refuse recovery before writing any output.""" + + from alicebot_api.vnext_derived_labels import PROPAGATION_BOUND, LabelPropagationTooLarge + + rows = self._fetch_all( + f"SELECT {SOURCE_CHUNK_COLUMNS} FROM source_chunks WHERE source_id = %s::uuid ORDER BY chunk_index, id LIMIT %s", + (source_id, PROPAGATION_BOUND + 1), + ) + if len(rows) > PROPAGATION_BOUND: + raise LabelPropagationTooLarge("source regeneration exceeded the source chunk bound") + return rows + def search_source_chunks( self, *, @@ -1864,8 +1878,9 @@ def update_project(self, *, project_id: str, patch: JsonObject, actor_type: str metadata = patch.get("metadata_json") floor_event = None if isinstance(metadata, dict) and before is not None: + before_metadata = before.get("metadata_json") patch["metadata_json"] = merge_protected_metadata( - before.get("metadata_json") if isinstance(before.get("metadata_json"), dict) else {}, + before_metadata if isinstance(before_metadata, dict) else {}, metadata, label_write="derived_from" in metadata, ) diff --git a/tests/integration/test_source_move_label_preview_postgres.py b/tests/integration/test_source_move_label_preview_postgres.py new file mode 100644 index 000000000..14056987b --- /dev/null +++ b/tests/integration/test_source_move_label_preview_postgres.py @@ -0,0 +1,128 @@ +"""Owner source moves preview real admission loss and regenerate fresh candidates.""" +import json +from copy import deepcopy +from uuid import UUID, uuid4 + +import anyio +import pytest + +import alicebot_api.main as main +from alicebot_api.config import Settings +from alicebot_api.db import user_connection +from alicebot_api.routers import vnext_memories as router +from alicebot_api.store import ContinuityStore +from alicebot_api.vnext_agent_keys import create_agent_key +from alicebot_api.vnext_capture import VNextCaptureService +from alicebot_api.vnext_store import PostgresVNextStore + +ALPHA = "prj_" + "a" * 16 +BETA = "prj_" + "b" * 16 + + +def _request(path, payload, raw_key): + messages = [] + body = json.dumps(payload).encode() + received = False + async def receive(): + nonlocal received + if received: + return {"type": "http.disconnect"} + received = True + return {"type": "http.request", "body": body, "more_body": False} + async def send(message): + messages.append(message) + scope = {"type": "http", "asgi": {"version": "3.0"}, "http_version": "1.1", "method": "POST", "scheme": "http", "path": path, "raw_path": path.encode(), "query_string": b"", "headers": [(b"host", b"127.0.0.1:8000"), (b"content-type", b"application/json"), (b"authorization", f"Bearer {raw_key}".encode())], "client": ("127.0.0.1", 50000), "server": ("testserver", 80), "root_path": ""} + anyio.run(main.app, scope, receive, send) + status = next(item["status"] for item in messages if item["type"] == "http.response.start") + return status, b"".join(item.get("body", b"") for item in messages if item["type"] == "http.response.body") + + +def _fixture(url): + user = uuid4() + with user_connection(url, user) as conn: + ContinuityStore(conn).create_user(user, f"move-{user}@example.test", "Synthetic") + store = PostgresVNextStore(conn) + result = VNextCaptureService(store, defer_embeddings=True).capture_text("I prefer synthetic blue ink.\nTODO: Review the synthetic draft", domain="project", sensitivity="public", project_scope=[ALPHA]) + source = store.get_source(str(result.source_id)) + report = store.create_artifact({"artifact_type": "daily_brief", "title": "Synthetic report", "content_markdown": "Synthetic report", "domain": "project", "sensitivity": "public", "metadata_json": {"source_ids": [str(result.source_id)], "project_scope": [ALPHA]}}) + return user, source, report + + +def _snapshot(url, user, source): + with user_connection(url, user) as conn: + store = PostgresVNextStore(conn) + rows = {table: store._fetch_all(f"SELECT * FROM {table} ORDER BY id") for table in ("sources", "memories", "open_loops", "generated_artifacts", "provenance_links", "event_log", "graph_edges")} + return rows + + +def test_source_move_preview_is_zero_write_then_confirmation_preserves_provenance(migrated_database_urls, monkeypatch): + url = migrated_database_urls["app"] + user, source, report = _fixture(url) + monkeypatch.setattr(router, "get_settings", lambda: Settings(database_url=url)) + before = _snapshot(url, user, source) + request = router.VNextSourceReviewRequest(user_id=user, action="assign_project", project_id=BETA) + preview = router.review_vnext_source(UUID(str(source["id"])), request) + payload = json.loads(preview.body) + assert preview.status_code == 200 and payload["preview"] is True + assert payload["derived_rows_hidden_from_project_keys"] == len(before["memories"]) + 1 + assert payload["confirm_required"] is True + assert _snapshot(url, user, source) == before + confirmed = router.review_vnext_source(UUID(str(source["id"])), request.model_copy(update={"confirm_label_hide": True})) + assert confirmed.status_code == 200 + after = _snapshot(url, user, source) + assert after["provenance_links"] == before["provenance_links"] + with user_connection(url, user) as conn: + store = PostgresVNextStore(conn) + assert store.get_source(str(source["id"]))["metadata_json"]["project_scope"] == [BETA] + moved_report = store.get_artifact(str(report["id"])) + assert BETA in moved_report["metadata_json"]["project_floor"] + assert moved_report["metadata_json"]["source_ids"] == report["metadata_json"]["source_ids"] + + +def test_regeneration_uses_moved_source_without_lowering_or_rewriting_originals(migrated_database_urls, monkeypatch): + url = migrated_database_urls["app"] + user, source, report = _fixture(url) + monkeypatch.setattr(router, "get_settings", lambda: Settings(database_url=url)) + move = router.VNextSourceReviewRequest(user_id=user, action="assign_project", project_id=BETA, confirm_label_hide=True, sensitivity="confidential", domain="health") + assert router.review_vnext_source(UUID(str(source["id"])), move).status_code == 200 + before = _snapshot(url, user, source) + result = router.regenerate_vnext_source(UUID(str(source["id"])), router.VNextSourceRegenerateRequest(user_id=user), authorization=None) + payload = json.loads(result.body) + assert result.status_code == 201 + assert payload["memory_count"] >= 1 and payload["open_loop_count"] == 1 + after = _snapshot(url, user, source) + assert after["sources"] == before["sources"] + assert after["generated_artifacts"] == before["generated_artifacts"] + for table in ("memories", "open_loops", "provenance_links"): + by_id = {row["id"]: row for row in after[table]} + assert all(by_id[row["id"]] == row for row in before[table]) + with user_connection(url, user) as conn: + store = PostgresVNextStore(conn) + for kind, ids in (("memory", payload["memory_ids"]), ("open_loop", payload["open_loop_ids"])): + rows = store.read_label_rows(kind, ids) + assert len(rows) == len(ids) + for row in rows: + assert row["domain"] == "health" and row["sensitivity"] == "confidential" + assert row["metadata_json"]["project_scope"] == [BETA] + assert row["metadata_json"]["project_floor"] == [BETA] + links = store._fetch_all("SELECT source_id FROM provenance_links WHERE target_id = %s", (str(row["id"]),)) + assert {str(link["source_id"]) for link in links} == {str(source["id"])} + + +@pytest.mark.parametrize("profile,scope,expected", [("trusted_local_agent", None, 403), ("admin_agent", ALPHA, 403), ("admin_agent", None, 201)]) +def test_regeneration_authenticates_and_refuses_trusted_or_bound_keys(migrated_database_urls, monkeypatch, profile, scope, expected): + url = migrated_database_urls["app"] + user, source, _report = _fixture(url) + with user_connection(url, user) as conn: + store = PostgresVNextStore(conn) + _key, raw = create_agent_key(store, user_id=user, agent_id="synthetic-reader", permission_profile=profile, project_scope=scope) + settings = Settings(database_url=url, app_env="test") + monkeypatch.setattr(router, "get_settings", lambda: settings) + monkeypatch.setattr(main, "get_settings", lambda: settings) + before = _snapshot(url, user, source) + status, response = _request(f"/v0/vnext/sources/{source['id']}/regenerate", {"user_id": str(user)}, raw) + assert status == expected, response + if expected == 403: + after = _snapshot(url, user, source) + for table in ("sources", "memories", "open_loops", "generated_artifacts", "provenance_links"): + assert after[table] == before[table] From 06e18f6241bbd8f0a6938468a039b06657f0888d Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 23:37:53 +0200 Subject: [PATCH 52/90] Make pure dependency helpers available to source recovery --- .../src/alicebot_api/vnext_derived_labels.py | 47 +++++++++++++++++++ 1 file changed, 47 insertions(+) diff --git a/apps/api/src/alicebot_api/vnext_derived_labels.py b/apps/api/src/alicebot_api/vnext_derived_labels.py index 97c4b3bc9..14db4f6a5 100644 --- a/apps/api/src/alicebot_api/vnext_derived_labels.py +++ b/apps/api/src/alicebot_api/vnext_derived_labels.py @@ -1219,7 +1219,54 @@ def scope_is_global(scope: object) -> bool: return is_global_scope(scope) +def input_admitted(kind: str, row: Mapping[str, object], projects: object) -> bool: + """Exact-door project test: scope and floor are both inside ``projects``.""" + + if canon_kind(kind) == "source": + scope = source_project_scope(row) + else: + scope = resolve_project_scope(row).values + shape, floor = project_floor_shape(row) + if shape == "malformed": + return False + bound = set(project_scope_identity(projects)) + scope_ids = set(project_scope_identity(scope)) + if not scope_ids or not scope_ids <= bound: + return False + return set(project_scope_identity(floor)) <= bound + + +def stamp_derived_from(payload: dict[str, object], rows_by_kind: Mapping[str, object]) -> None: + """Write the canonical dependency record onto ``payload['metadata_json']``.""" + + metadata = payload.get("metadata_json") + meta = dict(metadata) if isinstance(metadata, Mapping) else {} + record: dict[str, object] = {"v": 1} + counts: dict[str, int] = {} + for key in ("sources", "memories", "open_loops", "artifacts", "beliefs"): + raw_rows = rows_by_kind.get(key) + rows = raw_rows if isinstance(raw_rows, (list, tuple)) else [] + ids = [str(row.get("id")) for row in rows if isinstance(row, Mapping) and row.get("id") is not None] + record[key] = ids + counts[key] = len(ids) + record["counts"] = counts + meta["derived_from"] = record + payload["metadata_json"] = meta + + +def with_derived_from(metadata: Mapping[str, object], rows_by_kind: Mapping[str, object]) -> dict[str, object]: + """A copy of ``metadata`` with ``derived_from`` for the rows a producer used.""" + + payload: dict[str, object] = {"metadata_json": dict(metadata)} + stamp_derived_from(payload, rows_by_kind) + stamped = payload["metadata_json"] + return dict(stamped) if isinstance(stamped, Mapping) else {} + + __all__ = [ + "input_admitted", + "stamp_derived_from", + "with_derived_from", "DERIVED_ARTIFACT_TYPES", "DERIVED_WORKFLOWS", "HOP_BOUND", From 02bcc4506dc34cb2083902607a8f2478b673deff Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 23:38:42 +0200 Subject: [PATCH 53/90] Follow belief aliases in propagation and refresh reviewed carrier receipts --- .../src/alicebot_api/vnext_label_writes.py | 8 +++++ apps/api/src/alicebot_api/vnext_store.py | 21 +++++++++++++- .../test_label_floor_ancestry_postgres.py | 29 +++++++++++++++++++ tests/unit/test_source_move_label_preview.py | 24 +++++++++++++++ .../unit/test_store_graph_open_loops_split.py | 10 ++++--- tests/unit/test_store_memory_access_split.py | 3 +- .../unit/test_store_memory_lifecycle_split.py | 6 ++-- 7 files changed, 93 insertions(+), 8 deletions(-) diff --git a/apps/api/src/alicebot_api/vnext_label_writes.py b/apps/api/src/alicebot_api/vnext_label_writes.py index ebed03e80..5798ac0c3 100644 --- a/apps/api/src/alicebot_api/vnext_label_writes.py +++ b/apps/api/src/alicebot_api/vnext_label_writes.py @@ -440,6 +440,14 @@ def walk_dependants(store: Any, roots: Sequence[str]) -> list[dict[str, object]] raise LabelPropagationTooLarge(f"label propagation stopped after {PROPAGATION_BOUND} rows") batch = pending[:200] pending = pending[200:] + belief_aliases = getattr(store, "list_belief_ids_for_memories", None) + if callable(belief_aliases): + for belief_id in belief_aliases(batch): + alias = identifier(belief_id) + if alias not in seen: + seen.add(alias) + seen.add(_compact_id(belief_id)) + pending.append(str(belief_id)) matched: list[dict[str, object]] = [] batch_ids = {identifier(item) for item in batch} | {_compact_id(item) for item in batch} for row in list_dependants(store, batch): diff --git a/apps/api/src/alicebot_api/vnext_store.py b/apps/api/src/alicebot_api/vnext_store.py index 59aeee2f9..f172b2e6a 100644 --- a/apps/api/src/alicebot_api/vnext_store.py +++ b/apps/api/src/alicebot_api/vnext_store.py @@ -6,7 +6,7 @@ from contextlib import contextmanager from datetime import UTC, datetime from typing import Any, cast -from uuid import uuid4 +from uuid import UUID, uuid4 import psycopg @@ -527,6 +527,25 @@ def read_label_rows(self, kind: str, ids: Sequence[str]) -> list[VNextRow]: (wanted,), ) + def list_belief_ids_for_memories(self, ids: Sequence[str]) -> list[str]: + """Same-user belief aliases that make a memory an indirect report input.""" + + from alicebot_api.vnext_derived_labels import identifier + + wanted = [] + for value in ids: + try: + wanted.append(str(UUID(identifier(value)))) + except ValueError: + continue + if not wanted: + return [] + rows = self._fetch_all( + "SELECT id::text AS id FROM beliefs WHERE user_id = app.current_user_id() AND memory_id = ANY(%s::uuid[]) ORDER BY id", + (wanted,), + ) + return [str(row["id"]) for row in rows] + def _fetch_one( self, operation_name: str, diff --git a/tests/integration/test_label_floor_ancestry_postgres.py b/tests/integration/test_label_floor_ancestry_postgres.py index 657f3a3b5..2a16690f1 100644 --- a/tests/integration/test_label_floor_ancestry_postgres.py +++ b/tests/integration/test_label_floor_ancestry_postgres.py @@ -66,3 +66,32 @@ def test_checked_project_review_moves_scope_and_propagates_labels(migrated_datab assert summary_after["domain"] == "health" assert summary_after["sensitivity"] == "confidential" assert beta in summary_after["metadata_json"]["project_floor"] + + +def test_a_relabel_traverses_the_belief_backing_memory(migrated_database_urls): + user_id = uuid4() + url = migrated_database_urls["app"] + with user_connection(url, user_id) as conn: + ContinuityStore(conn).create_user(user_id, f"belief-relabel-{user_id}@example.test", "Synthetic") + store = PostgresVNextStore(conn) + source = store.create_source({"source_type": "note", "title": "synthetic", "content_hash": str(uuid4()), "domain": "project", "sensitivity": "public"}) + copy = store.create_memory({"memory_key": "copy", "canonical_text": "synthetic", "domain": "project", "sensitivity": "public", "metadata_json": {"source_id": str(source["id"])}}) + belief = store.create_belief({"memory_id": str(copy["id"]), "claim": "Synthetic belief"}) + report = store.create_artifact({"artifact_type": "contradiction_report", "title": "Synthetic", "content_markdown": "Synthetic", "domain": "project", "sensitivity": "public", "metadata_json": {"belief_ids": [str(belief["id"])]}}) + other_user = uuid4() + with user_connection(url, other_user) as conn: + ContinuityStore(conn).create_user(other_user, f"other-belief-{other_user}@example.test", "Synthetic") + other_store = PostgresVNextStore(conn) + other_memory = other_store.create_memory({"memory_key": "other", "canonical_text": "synthetic", "domain": "project", "sensitivity": "public"}) + other_store.create_belief({"memory_id": str(other_memory["id"]), "claim": "Other synthetic belief"}) + with user_connection(url, user_id) as conn: + store = PostgresVNextStore(conn) + alias = "{" + str(copy["id"]).upper() + "}" + assert store.list_belief_ids_for_memories([alias, str(other_memory["id"])]) == [str(belief["id"])] + store.lock_graph_mutation() + store.lock_label_writes(exclusive=True) + store.update_source(source_id=str(source["id"]), patch={"sensitivity": "regulated"}) + with user_connection(url, user_id) as conn: + store = PostgresVNextStore(conn) + assert store.get_memory(str(copy["id"]))["sensitivity"] == "regulated" + assert store.get_artifact(str(report["id"]))["sensitivity"] == "regulated" diff --git a/tests/unit/test_source_move_label_preview.py b/tests/unit/test_source_move_label_preview.py index cb6caa1ce..0ef20c1e7 100644 --- a/tests/unit/test_source_move_label_preview.py +++ b/tests/unit/test_source_move_label_preview.py @@ -51,3 +51,27 @@ def test_preview_does_not_count_a_row_already_unverified(monkeypatch): report = _report([source, _source(["alpha"])], ["alpha"]) monkeypatch.setattr(writes, "walk_dependants", lambda *_: [report]) assert writes.count_rows_hidden_by_scope_move(Store([source, report]), source, ["beta"]) == 0 + + +def test_belief_alias_is_an_intermediate_reverse_edge(monkeypatch): + source = _source(["alpha"]) + report = _report([], ["alpha"]) + belief_id = str(uuid4()) + report["metadata_json"]["belief_ids"] = [belief_id] + store = Store([source, report]) + store.list_belief_ids_for_memories = lambda ids: [belief_id] if source["id"] in ids else [] + monkeypatch.setattr(writes, "list_dependants", lambda _store, ids: [report] if belief_id in ids else []) + assert writes.walk_dependants(store, [source["id"]]) == [report] + + +def test_sqlite_reverse_walk_needs_no_unsupported_belief_table(tmp_path): + from alicebot_api.onramp import bootstrap_database + from alicebot_api.sqlite_store import SQLiteVNextStore, sqlite_user_connection + from tests.unit.test_derived_domain_fence import USER + path = tmp_path / "labels.sqlite3" + bootstrap_database(path, user_id=USER, user_email="synthetic@example.test") + with sqlite_user_connection(path, USER) as conn: + store = SQLiteVNextStore(conn, USER) + source = store.create_source({"source_type": "note", "title": "synthetic", "content_hash": "synthetic", "domain": "project", "sensitivity": "public"}) + copy = store.create_memory({"memory_key": "copy", "canonical_text": "synthetic", "domain": "project", "sensitivity": "public", "metadata_json": {"source_id": str(source["id"])}}) + assert [row["id"] for row in writes.walk_dependants(store, [str(source["id"])])] == [str(copy["id"])] diff --git a/tests/unit/test_store_graph_open_loops_split.py b/tests/unit/test_store_graph_open_loops_split.py index eae9d1a4d..7013c4cff 100644 --- a/tests/unit/test_store_graph_open_loops_split.py +++ b/tests/unit/test_store_graph_open_loops_split.py @@ -104,13 +104,15 @@ "OPEN_LOOP_COLUMNS", ) +# Reviewed label hooks: both open-loop updaters preserve protected metadata, +# clamp derived labels and propagate a stricter label to later rows. SOURCE_RECEIPTS = { # Re-minted for the filter-before-cut fix (2026-10-03): ``list_open_loop_events`` takes ``domains`` and # ``sensitivity_allowed`` as required arguments, and the SQLite one applies them in the join before ``LIMIT`` # (an empty ceiling returns no rows without a query). The Postgres reader takes the same two arguments so that # the shared unscoped call site can state ``None`` for both, and it refuses anything else, since the Postgres # runtime resolves no project view (reviewed change, not drift). - POSTGRES_CARRIER_PATH: "e4724ba1ec3b8917c5be74b619259ddf1c282825b8e938ce4fa9947491a90a6f", + POSTGRES_CARRIER_PATH: "a9fecb0462324a46610f01134146fa73a2ab63692c7a6daa44ff56760840a812", # The SQLite carrier is re-minted, with its method AST manifest below, for # ``list_open_loops`` and ``list_open_loop_events``: they bind a query through # ``literal_match_operand`` and so refuse one past the LIKE operand limit. @@ -128,13 +130,13 @@ # delete preview and the scrub count and blank the same loops. The rule text moved unchanged to # ``vnext_stores/sqlite/open_loop_source_reference.py``; only the reader function and the receipt of the file # change (reviewed change, not drift). - SQLITE_CARRIER_PATH: "9a2634bef621d32262b845c046820d8b19c64801ec9f9b462e978f364f16f643", + SQLITE_CARRIER_PATH: "ea8a177e5037cfe40682def81b4cc1d6116b754b01e942d08a4367e5e54c8d2b", POSTGRES_COLUMNS_PATH: "5b0d972a55abf8590ce14394a37fd71b9b88ba7ab3de82d61efc1bddfc022b71", SQLITE_COLUMNS_PATH: "be81b8628d0831d3d02b280b5455fb02333db5740ebef8d85d58024384ae6556", } EXPECTED_METHOD_AST_MANIFESTS = { - POSTGRES_CARRIER_PATH: "2558088459f1b9a565e1b366ffe0b7c4025c623a9e2ea78007d06a46793ce1b8", - SQLITE_CARRIER_PATH: "2850ba6057b1510759613aaa3798a226808a42470ee11cfb9c6e3afbf3e98e66", + POSTGRES_CARRIER_PATH: "69d4776f91829f5dc96f751c7d513c13149f851d878476c1c2c17c245d8cf0a3", + SQLITE_CARRIER_PATH: "2d29f3668f52984f860b25fb6db5b37b04a2e4395631e584f70c93fa016540a1", } EXPECTED_METADATA_MANIFESTS = { POSTGRES_CARRIER_PATH: "6edb6a10e7a37dbbbbde97e5550422718a0112257666de8e23d49c60490fa13f", diff --git a/tests/unit/test_store_memory_access_split.py b/tests/unit/test_store_memory_access_split.py index 7b2647bbf..0e669eb6c 100644 --- a/tests/unit/test_store_memory_access_split.py +++ b/tests/unit/test_store_memory_access_split.py @@ -23,6 +23,7 @@ POSTGRES_FACADE_PATH = REPO_ROOT / "apps/api/src/alicebot_api/vnext_store.py" SQLITE_FACADE_PATH = REPO_ROOT / "apps/api/src/alicebot_api/sqlite_store.py" +# Reviewed lock boundary: the pending-candidate row locker takes L first. SOURCE_RECEIPTS = { "apps/api/src/alicebot_api/vnext_stores/retrieval_common.py": ( "fa1a3a90511b5c61754ba29560e91b7b3058a48d47c143b09d8505d52025b8cc" @@ -37,7 +38,7 @@ # which takes a keyword-only ``include_deleted`` (false by default, so every caller reads what it read before) # and drops the ``deleted_at IS NULL`` clause only when it is true. Previous Postgres receipt f642880f... "apps/api/src/alicebot_api/vnext_stores/postgres/memory_access.py": ( - "46946cc087de35f54474adf47cadcd67b862685bfa67a38be277d8e58a00c47e" + "9d42ac3a44b33f067306e7903f711cddf7ba538529cf03c16baf00473d861839" ), # Re-minted for per-project memory S2 (2026-10-02): the project fence builders read the reserved global # marker and take the domains to leave out, and the single-scan partition SQL and the materialized-CTE hint diff --git a/tests/unit/test_store_memory_lifecycle_split.py b/tests/unit/test_store_memory_lifecycle_split.py index 2f8b6c208..18b15e917 100644 --- a/tests/unit/test_store_memory_lifecycle_split.py +++ b/tests/unit/test_store_memory_lifecycle_split.py @@ -82,9 +82,11 @@ # mutators take the label lock. Metadata receipts include the label_write # keyword and lock wrappers. Facades add only lock_label_writes/read_label_rows; # removing those two names reproduces each previous class-order receipt. +# Reviewed strict lock change: the graph lock reads live advisory grants, +# and the memory update checks exclusive L before changing labels. SOURCE_RECEIPTS = { COMMON_PATH: "8fc077dc71f0e631a2df81de2ebeec1fb6c768f341c2e7891309e4753eef7bb5", - POSTGRES_CARRIER_PATH: "371f92595d2f72f0cfa225a49c03aa39498caf094df4f26f4f9ac4cd926e0de1", + POSTGRES_CARRIER_PATH: "23ab87cde159a285bf8c71dbc6d2eb4e0e15035ce0f509bef38ba799f3f92de3", # SQLite carrier re-minted for the Phase 4 Stage 2 resident vector cache # (reviewed change): redaction paths that NULL a live embedding now bump # the embedding_stamp token in the same transaction (prompt eviction). @@ -98,7 +100,7 @@ SQLITE_CARRIER_PATH: "f893f1faeeb9a87135c108992aa91ff036c5f5dccb1bbdaf315ae5afe1b89b73", } EXPECTED_METHOD_AST_MANIFESTS = { - "postgres": "d4969140e86b136b29708dc3bb6b4bca635b73b4016c4e633c4d5d3da841e784", + "postgres": "827c4f391a1efe50dcb265b7bb50a5b191d2bae32c3f817dcc237d51bfb10d29", "sqlite": "d0b6024f0803ca9d3c6f6ea7f5022a28453b2b8b83833f0f05343cb3eff89a57", } EXPECTED_METADATA_MANIFESTS = { From 7da4765673c2bd5d278106de085fd306cf903344 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 23:49:45 +0200 Subject: [PATCH 54/90] Check canonical project labels before changing legacy project pointers --- .../api/src/alicebot_api/vnext_label_writes.py | 7 +++++-- tests/unit/test_label_lock_order.py | 18 ++++++++++++++++++ 2 files changed, 23 insertions(+), 2 deletions(-) diff --git a/apps/api/src/alicebot_api/vnext_label_writes.py b/apps/api/src/alicebot_api/vnext_label_writes.py index 5798ac0c3..04dba3302 100644 --- a/apps/api/src/alicebot_api/vnext_label_writes.py +++ b/apps/api/src/alicebot_api/vnext_label_writes.py @@ -179,7 +179,10 @@ def prepare_label_patch( proposed = dict(before or {}) proposed.update({key: value for key, value in patch.items() if value is not None}) - if before and _label_fields(before) != _label_fields(proposed): + proposed["kind"] = kind + old = _label_fields({**before, "kind": kind}) if before else None + new = _label_fields(proposed) + if old and (old[:2] != new[:2] or project_scope_identity(old[2]) != project_scope_identity(new[2]) or project_scope_identity(old[3]) != project_scope_identity(new[3])): require_exclusive_label_lock(store) return dict(patch) @@ -467,7 +470,7 @@ def walk_dependants(store: Any, roots: Sequence[str]) -> list[dict[str, object]] def _label_fields(row: Mapping[str, object]) -> tuple[str, str, tuple[str, ...], tuple[str, ...]]: metadata = row.get("metadata_json") meta = metadata if isinstance(metadata, Mapping) else {} - scope = meta.get("project_scope", ()) + scope = source_project_scope(row) if row.get("kind") == "source" or "source_type" in row else resolve_project_scope(row).values floor = meta.get("project_floor", ()) return ( str(row.get("domain") or "unknown"), diff --git a/tests/unit/test_label_lock_order.py b/tests/unit/test_label_lock_order.py index 5ac522ddc..e2e021eca 100644 --- a/tests/unit/test_label_lock_order.py +++ b/tests/unit/test_label_lock_order.py @@ -84,3 +84,21 @@ def test_compare_and_set_miss_refuses_the_whole_label_write(): store = SimpleNamespace(_fetch_optional_one=lambda *_: None) with pytest.raises(DerivedDomainRepairError, match="changed no row"): writes.write_settled_label(store, kind="memory", row_id="missing", domain="health", sensitivity="confidential", metadata={}, project_id=None, expected_domain="project", expected_sensitivity="public") + + +def test_strict_hook_checks_the_legacy_project_pointer_before_any_update(monkeypatch): + monkeypatch.setattr(writes, "STRICT_LOCK_ORDER", True) + store = _store() + before = {"domain": "project", "sensitivity": "public", "project_id": "11111111-1111-4111-8111-111111111111"} + with pytest.raises(writes.LabelLockOrderError, match="exclusive label lock"): + writes.prepare_label_patch(store, "memory", before, {"project_id": "22222222-2222-4222-8222-222222222222"}) + + +def test_named_refusal_causes_never_disclose_error_content(): + from psycopg.errors import DivisionByZero, LockNotAvailable + from alicebot_api.vnext_derived_domain_backfill import DerivedDomainRepairError + for error, cause in ((writes.LabelPropagationTooLarge("synthetic-private-text"), "propagation_bound"), (DerivedDomainRepairError("changed no row: synthetic-private-text"), "row_changed"), (DerivedDomainRepairError("cycle: synthetic-private-text"), "dependency_cycle"), (writes.LabelLockOrderError("synthetic-private-text"), "lock_order"), (DivisionByZero("synthetic-private-text"), "database_error")): + status, detail, retry_after = writes.label_error_response(error) + assert status == 409 and detail.endswith("cause: " + cause) and retry_after is None + assert "synthetic-private-text" not in detail + assert writes.label_error_response(LockNotAvailable("synthetic-private-text")) == (503, writes.RETRYABLE_DETAIL, "2") From a3711a6c8be2d5cdc6473eecead8bc20ce160b16 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 23:58:36 +0200 Subject: [PATCH 55/90] Document source regeneration as a creation response --- apps/api/src/alicebot_api/routers/vnext_memories.py | 2 +- .../integration/test_source_move_label_preview_postgres.py | 6 ++++++ 2 files changed, 7 insertions(+), 1 deletion(-) diff --git a/apps/api/src/alicebot_api/routers/vnext_memories.py b/apps/api/src/alicebot_api/routers/vnext_memories.py index c58712977..f8d44a3ca 100644 --- a/apps/api/src/alicebot_api/routers/vnext_memories.py +++ b/apps/api/src/alicebot_api/routers/vnext_memories.py @@ -760,7 +760,7 @@ def get_vnext_source(source_id: UUID, user_id: UUID) -> JSONResponse: ) -@source_review_router.post("/v0/vnext/sources/{source_id}/regenerate", summary="Regenerate fresh candidates from a stored source", description="The local owner or an unbound admin can regenerate candidate memories and open loops from all stored chunks using the source's current labels. Existing sources and outputs remain unchanged. Rerun the report's generation route to rebuild a report.") +@source_review_router.post("/v0/vnext/sources/{source_id}/regenerate", status_code=201, summary="Regenerate fresh candidates from a stored source", description="The local owner or an unbound admin can regenerate candidate memories and open loops from all stored chunks using the source's current labels. Existing sources and outputs remain unchanged. Rerun the report's generation route to rebuild a report.") def regenerate_vnext_source(source_id: UUID, request: VNextSourceRegenerateRequest, authorization: str | None = Header(default=None)) -> JSONResponse: from alicebot_api.vnext_label_writes import label_error_response from alicebot_api.vnext_source_regeneration import regenerate_source_inputs diff --git a/tests/integration/test_source_move_label_preview_postgres.py b/tests/integration/test_source_move_label_preview_postgres.py index 14056987b..a280c7f4c 100644 --- a/tests/integration/test_source_move_label_preview_postgres.py +++ b/tests/integration/test_source_move_label_preview_postgres.py @@ -19,6 +19,12 @@ BETA = "prj_" + "b" * 16 +def test_regeneration_openapi_documents_creation_status(): + responses = main.app.openapi()["paths"]["/v0/vnext/sources/{source_id}/regenerate"]["post"]["responses"] + assert "201" in responses + assert responses["201"]["content"]["application/json"]["schema"]["$ref"].endswith("RegenerateVnextSourceSuccessResponse") + + def _request(path, payload, raw_key): messages = [] body = json.dumps(payload).encode() From 9b0bebb6a44f0b3f0f6e35a0778f1c8bc1b81152 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Tue, 6 Oct 2026 00:06:05 +0200 Subject: [PATCH 56/90] Refresh reviewed write backport carrier receipts --- tests/unit/test_store_events_revisions_split.py | 4 +++- tests/unit/test_store_graph_open_loops_split.py | 6 ++++-- tests/unit/test_store_memory_access_split.py | 6 ++++-- tests/unit/test_store_memory_lifecycle_split.py | 10 +++++++--- 4 files changed, 18 insertions(+), 8 deletions(-) diff --git a/tests/unit/test_store_events_revisions_split.py b/tests/unit/test_store_events_revisions_split.py index dde8c9474..ee09f4e66 100644 --- a/tests/unit/test_store_events_revisions_split.py +++ b/tests/unit/test_store_events_revisions_split.py @@ -139,6 +139,8 @@ }, } EXPECTED_CLASS_KEY_SHA256 = { + # Reviewed recovery and reverse propagation add only the two Postgres + # methods; all prior runtime class keys retain their order. # Source owner methods are additive; the existing member order is unchanged. # Re-minted for the paired browser-clip capability façade methods. # The sqlite hash is re-minted again for ``check_source_search_query``. It is @@ -148,7 +150,7 @@ # Re-minted for the per-file importer savepoint (2026-10-02): ``savepoint`` is appended last on both façades. # Previous receipt: 650e2e0ff088d67c... Proof: the class key list equals the list at origin/main 040a2a10 # plus ``savepoint`` before ``__dict__``, with every other key in the same order. - "postgres": "dda7be1a316b3c525195f4682a608d0bab24b57f635db2a7aebc0c7ea187fc68", + "postgres": "e4a921e06bc707a362fb9c2b3050d292b119cd0b92b346ce5997a679b3bd61f0", # Re-minted for the merge of #500 and #502: ``check_literal_match_query``. # Re-minted again for per-project memory S2 (2026-10-02): the two single-scan partition reads # ``list_memories_view_partitions`` and ``list_open_loops_view_partitions``, SQLite only on purpose diff --git a/tests/unit/test_store_graph_open_loops_split.py b/tests/unit/test_store_graph_open_loops_split.py index 7013c4cff..ec64e2cba 100644 --- a/tests/unit/test_store_graph_open_loops_split.py +++ b/tests/unit/test_store_graph_open_loops_split.py @@ -149,12 +149,14 @@ SQLITE_CARRIER_PATH: (6, "970028b5c929f0e749d8b40bdee571c872600de7a713e58d60e0da86f022af8a"), } EXPECTED_CLASS_ORDERS = { + # Reviewed source recovery and belief propagation add only two Postgres + # methods. The SQLite facade and every existing member retain their order. # Two paired browser-clip capability methods extend both façades, and one # more paired method, ``list_memories_referencing_sources``. # Per-file importer savepoint (2026-10-02): one paired method more, ``savepoint``, appended last. # Previous receipt: (171, 526374782104a2a1...). Proof: the member list equals the list at origin/main # 040a2a10 with ``savepoint`` added at the end and nothing else moved (reviewed change, not drift). - "PostgresVNextStore": (172, "6f1a459fcf4319cf4281f6cc0d4e81679c3e05d851fd0a874a2d90298d7c2569"), + "PostgresVNextStore": (176, "3549d6ea179f4ebc2341f1c251f139dfb747d56700a4554aef114150efed5c3f"), # One SQLite-only method more, ``check_source_search_query``: the Postgres # source search has no expression-depth or LIKE-length limit to check. # Merge of #500 and #502 (2026-10-01): one more SQLite-only method, @@ -169,7 +171,7 @@ # 040a2a10 with ``savepoint`` added at the end and nothing else moved (reviewed change, not drift). # Proof: the replacement branch gains only scrub_source, source_inventory and # prunable_sources here; every pre-existing class member keeps its order. - "SQLiteVNextStore": (134, "1301272026897057cf071009cc21787543ddc326f1e06f1a75a763f3e344767e"), + "SQLiteVNextStore": (136, "1d99dbaf69e4ea888ca7beb2389ac176650a2e573c067bf0686adc9e609132f5"), } EXPECTED_COLUMN_AST = { POSTGRES_COLUMNS_PATH: { diff --git a/tests/unit/test_store_memory_access_split.py b/tests/unit/test_store_memory_access_split.py index 0e669eb6c..90a0dc8bb 100644 --- a/tests/unit/test_store_memory_access_split.py +++ b/tests/unit/test_store_memory_access_split.py @@ -194,6 +194,8 @@ ) EXPECTED_CLASS_ORDERS = { + # Reviewed source recovery and belief propagation add only two Postgres + # methods. The SQLite facade and every existing member retain their order. # Two paired browser-clip capability methods extend both façades. One more # paired method, ``list_memories_referencing_sources``, is the batched form # of ``list_memories_referencing_source``; both carrier receipts above were @@ -201,7 +203,7 @@ # Per-file importer savepoint (2026-10-02): one paired method more, ``savepoint``, appended last. # Previous receipt: (171, 526374782104a2a1...). Proof: the member list equals the list at origin/main # 040a2a10 with ``savepoint`` added at the end and nothing else moved (reviewed change, not drift). - "PostgresVNextStore": (172, "6f1a459fcf4319cf4281f6cc0d4e81679c3e05d851fd0a874a2d90298d7c2569"), + "PostgresVNextStore": (176, "3549d6ea179f4ebc2341f1c251f139dfb747d56700a4554aef114150efed5c3f"), # One SQLite-only method more, ``check_source_search_query``: the Postgres # source search has no expression-depth or LIKE-length limit to check. # Merge of #500 and #502 (2026-10-01): one more SQLite-only method, @@ -216,7 +218,7 @@ # 040a2a10 with ``savepoint`` added at the end and nothing else moved (reviewed change, not drift). # Proof: the replacement branch gains only scrub_source, source_inventory and # prunable_sources here; every pre-existing class member keeps its order. - "SQLiteVNextStore": (134, "1301272026897057cf071009cc21787543ddc326f1e06f1a75a763f3e344767e"), + "SQLiteVNextStore": (136, "1d99dbaf69e4ea888ca7beb2389ac176650a2e573c067bf0686adc9e609132f5"), } diff --git a/tests/unit/test_store_memory_lifecycle_split.py b/tests/unit/test_store_memory_lifecycle_split.py index 18b15e917..4f9f9489d 100644 --- a/tests/unit/test_store_memory_lifecycle_split.py +++ b/tests/unit/test_store_memory_lifecycle_split.py @@ -82,6 +82,8 @@ # mutators take the label lock. Metadata receipts include the label_write # keyword and lock wrappers. Facades add only lock_label_writes/read_label_rows; # removing those two names reproduces each previous class-order receipt. +# Reviewed SQLite owner clamp: update_memory settles an edit at its inputs +# and records whether the clamp applied; its signature and metadata stay fixed. # Reviewed strict lock change: the graph lock reads live advisory grants, # and the memory update checks exclusive L before changing labels. SOURCE_RECEIPTS = { @@ -97,23 +99,25 @@ # back between two reads cannot fail memories_seen_range_check. Previous # sqlite receipt 67adaa61..., method AST 3f134ac9...; the metadata # manifests are unchanged. - SQLITE_CARRIER_PATH: "f893f1faeeb9a87135c108992aa91ff036c5f5dccb1bbdaf315ae5afe1b89b73", + SQLITE_CARRIER_PATH: "759cf44762c388e599b4e8c377fa3415ca2fdf9ba7884698259171ef3d2b8138", } EXPECTED_METHOD_AST_MANIFESTS = { "postgres": "827c4f391a1efe50dcb265b7bb50a5b191d2bae32c3f817dcc237d51bfb10d29", - "sqlite": "d0b6024f0803ca9d3c6f6ea7f5022a28453b2b8b83833f0f05343cb3eff89a57", + "sqlite": "3577659fcf9e583bdb957bb1a959ac1d8cae07ce8b50321bc36697dec47acec4", } EXPECTED_METADATA_MANIFESTS = { "postgres": "07a567e26d0f7c4f51ae2a1910d059397b513a575d85f06c2f8c0396ac1bb2ef", "sqlite": "1270ef0115988552418349aa9e94a7442ba04be41443f278f68a1fa81857903a", } EXPECTED_CLASS_ORDERS = { + # Reviewed source recovery and belief propagation add only two Postgres + # methods. The SQLite facade and every existing member retain their order. # Two paired browser-clip capability methods extend both façades, and one # more paired method, ``list_memories_referencing_sources``. # Per-file importer savepoint (2026-10-02): one paired method more, ``savepoint``, appended last. # Previous receipt: (171, 526374782104a2a1...). Proof: the member list equals the list at origin/main # 040a2a10 with ``savepoint`` added at the end and nothing else moved (reviewed change, not drift). - "PostgresVNextStore": (174, "095250b8a77d6c0a32d2783343916b947bcae8ae86e3a1693e47c2fb11802211"), + "PostgresVNextStore": (176, "3549d6ea179f4ebc2341f1c251f139dfb747d56700a4554aef114150efed5c3f"), # One SQLite-only method more, ``check_source_search_query``: the Postgres # source search has no expression-depth or LIKE-length limit to check. # Merge of #500 and #502 (2026-10-01): one more SQLite-only method, From a860d07963f2e394d69ffd14b294e58ce935faf8 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 23:27:20 +0200 Subject: [PATCH 57/90] Enforce canonical dependency record completeness --- apps/api/src/alicebot_api/vnext_derived_labels.py | 7 +------ tests/unit/test_derived_labels_record_completeness.py | 10 ++++++++++ 2 files changed, 11 insertions(+), 6 deletions(-) diff --git a/apps/api/src/alicebot_api/vnext_derived_labels.py b/apps/api/src/alicebot_api/vnext_derived_labels.py index 3391fb4b9..97c4b3bc9 100644 --- a/apps/api/src/alicebot_api/vnext_derived_labels.py +++ b/apps/api/src/alicebot_api/vnext_derived_labels.py @@ -534,16 +534,13 @@ def _derived_from_deps(record: object) -> tuple[str, set[tuple[str, str]]]: return "malformed", set() found: set[tuple[str, str]] = set() lists: dict[str, list[str]] = {} - repeated = False for key, kind in _DERIVED_FROM_KIND.items(): if key not in record: lists[key] = [] continue raw = record.get(key) - if not isinstance(raw, list) or any(not isinstance(item, str) for item in raw): + if not isinstance(raw, list) or any(not isinstance(item, str) or not item.strip() for item in raw): return "malformed", set() - if len(raw) != len(set(raw)): - repeated = True ids = _strings(raw) lists[key] = ids _add_ids(found, kind, ids) @@ -554,8 +551,6 @@ def _derived_from_deps(record: object) -> tuple[str, set[tuple[str, str]]]: return "", found if not isinstance(counts, Mapping): return "malformed", found - if repeated: - return "", found for key, ids in lists.items(): if key not in counts: if ids: diff --git a/tests/unit/test_derived_labels_record_completeness.py b/tests/unit/test_derived_labels_record_completeness.py index efac96c30..ff9d6b679 100644 --- a/tests/unit/test_derived_labels_record_completeness.py +++ b/tests/unit/test_derived_labels_record_completeness.py @@ -34,3 +34,13 @@ def test_a_missing_or_malformed_legacy_completeness_record_is_unverified(workflo def test_a_legitimate_empty_legacy_record_is_verified(workflow, lists, counts): row = legacy_report(workflow, lists, counts) assert settle_labels([row]).by_stored("artifact", "report").unverified is False + + +@pytest.mark.parametrize("ids,count", ((["source", "source"], 1), ([""], 1), ([" "], 1))) +def test_canonical_counts_and_identifiers_do_not_use_the_legacy_duplicate_exception(ids, count): + source = {"kind": "source", "id": "source", "domain": "project", "sensitivity": "public", "metadata_json": {}} + row = {"kind": "artifact", "id": "report", "metadata_json": {"derived_from": { + "v": 1, "sources": ids, "memories": [], "open_loops": [], "artifacts": [], "beliefs": [], + "counts": {"sources": count, "memories": 0, "open_loops": 0, "artifacts": 0, "beliefs": 0}, + }}} + assert settle_labels([source, row]).by_stored("artifact", "report").unverified is True From 1431ad7b4350f9d0215de2cdf7d29a3493dc6d9b Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Tue, 6 Oct 2026 00:13:34 +0200 Subject: [PATCH 58/90] Verify replacement memories preserve dependency edges --- .../test_sqlite_derived_labels_write_path.py | 39 +++++++++++++++++++ 1 file changed, 39 insertions(+) diff --git a/tests/unit/test_sqlite_derived_labels_write_path.py b/tests/unit/test_sqlite_derived_labels_write_path.py index ad231c68a..ee6a00024 100644 --- a/tests/unit/test_sqlite_derived_labels_write_path.py +++ b/tests/unit/test_sqlite_derived_labels_write_path.py @@ -232,6 +232,45 @@ def test_merge_protected_metadata_keeps_label_keys_unless_the_write_is_a_relabel assert relabel["consolidation"] == {"cluster_member_ids": ["m"]} +def test_a_replacement_memory_keeps_dependencies_and_floor(tmp_path: Path, monkeypatch) -> None: + from uuid import UUID + + from alicebot_api.mcp.registry import call_mcp_tool + from alicebot_api.mcp.types import MCPRuntimeContext + from alicebot_api.onramp import sqlite_url_for_path + from alicebot_api.vnext_derived_labels import with_derived_from + + monkeypatch.delenv("ALICE_AGENT_API_KEY", raising=False) + monkeypatch.delenv("ALICE_EMBEDDINGS_BASE_URL", raising=False) + monkeypatch.setenv("ALICE_MCP_FULL_TOOLS", "1") + path = tmp_path / "replacement.sqlite3" + with _vault(path) as conn: + store = SQLiteVNextStore(conn, USER) + source = store.create_source({"source_type": "note", "title": "Synthetic observation", "content_hash": "replacement", "domain": "health", "sensitivity": "confidential", "metadata_json": {"project_scope": [ALPHA]}}) + metadata = with_derived_from({"source_id": str(source["id"]), "project_scope": [ALPHA]}, {"sources": [source]}) + original = store.create_memory({"memory_key": "replacement-original", "canonical_text": "Synthetic original observation", "status": "active", "domain": "health", "sensitivity": "confidential", "metadata_json": metadata}) + link = store.create_provenance_link({"target_type": "memory", "target_id": str(original["id"]), "source_id": str(source["id"]), "evidence_role": "supports", "confidence": 1.0}) + result = call_mcp_tool( + MCPRuntimeContext(database_url=sqlite_url_for_path(path), user_id=UUID(USER)), + name="alice_memory_correct", + arguments={"review_item_id": str(original["id"]), "action": "supersede-existing", "replacement_title": "Synthetic corrected observation", "replacement_provenance": {"source_id": str(source["id"]), "evidence_role": "supports", "confidence": 1.0}}, + ) + replacement_id = str(result["replacement_object"]["id"]) + with sqlite_user_connection(path, USER) as conn: + store = SQLiteVNextStore(conn, USER) + replacement = store.get_memory(replacement_id) + assert replacement["metadata_json"]["derived_from"] == metadata["derived_from"] + assert replacement["metadata_json"]["source_id"] == source["id"] + assert replacement["metadata_json"]["project_scope"] == [ALPHA] + assert replacement["metadata_json"]["project_floor"] == [ALPHA] + assert replacement["sensitivity"] == "confidential" + assert store.get_memory(str(original["id"]))["canonical_text"] == original["canonical_text"] + assert str(store.list_provenance_links(target_type="memory", target_id=str(original["id"]))[0]["id"]) == str(link["id"]) + assert str(store.list_provenance_links(target_type="memory", target_id=replacement_id)[0]["source_id"]) == str(source["id"]) + store.update_source(source_id=str(source["id"]), patch={"sensitivity": "regulated"}) + assert store.get_memory(replacement_id)["sensitivity"] == "regulated" + + def test_insert_floor_survives_two_hops(tmp_path: Path): db = tmp_path / 'labels.sqlite3' bootstrap_database(db, user_id=USER, user_email='synthetic@example.test') From 18cd836c43e5bd8555a861d58b6ab024a7f0fb1a Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Tue, 6 Oct 2026 00:19:50 +0200 Subject: [PATCH 59/90] Test project floor views and derived group card lookup --- ...est_derived_labels_group_scope_postgres.py | 35 ++++++++ tests/unit/test_group_scope_sqlite.py | 84 +++++++++++++++++++ tests/unit/test_project_floor_views.py | 75 +++++++++++++++++ 3 files changed, 194 insertions(+) create mode 100644 tests/integration/test_derived_labels_group_scope_postgres.py create mode 100644 tests/unit/test_group_scope_sqlite.py create mode 100644 tests/unit/test_project_floor_views.py diff --git a/tests/integration/test_derived_labels_group_scope_postgres.py b/tests/integration/test_derived_labels_group_scope_postgres.py new file mode 100644 index 000000000..39c093d19 --- /dev/null +++ b/tests/integration/test_derived_labels_group_scope_postgres.py @@ -0,0 +1,35 @@ +"""The PostgreSQL group consumers have the same controls as SQLite.""" + +from uuid import uuid4 + +import pytest + +from alicebot_api.db import user_connection +from alicebot_api.store import ContinuityStore +from alicebot_api.vnext_derived_labels import group_scope +from alicebot_api.vnext_memory_commit import VNextMemoryCommitService +from alicebot_api.vnext_rollups import VNextRollupService +from alicebot_api.vnext_store import PostgresVNextStore +from tests.unit.test_group_scope_sqlite import ALPHA, seed_members + + +@pytest.mark.parametrize("accept", (False, True)) +def test_a_second_scoped_rollup_run_over_the_same_group_finds_its_card_and_inserts_nothing(migrated_database_urls, accept): + user_id = uuid4() + with user_connection(migrated_database_urls["app"], user_id) as conn: + ContinuityStore(conn).create_user(user_id, f"group-{user_id}@example.invalid", "Group") + store = PostgresVNextStore(conn) + members = seed_members(store) + service = VNextRollupService(store) + first = service.propose_rollups(projects=(ALPHA,)) + assert len(first.candidate_ids) == 1 + candidate = store.get_memory(first.candidate_ids[0]) + assert candidate["metadata_json"]["project_scope"] == [] + assert group_scope(candidate) == group_scope(members[0]) + if accept: + assert VNextMemoryCommitService(store).accept_consolidation_candidate(first.candidate_ids[0], reason="Reviewed synthetic group")["status"] == "accepted" + before = conn.execute("SELECT count(*) AS n FROM memories").fetchone()["n"] + second = service.propose_rollups(projects=(ALPHA,)) + assert second.proposals == [] + assert conn.execute("SELECT count(*) AS n FROM memories").fetchone()["n"] == before + assert any(group["state"] == ("already_covered_by_accepted" if accept else "existing_candidate") for group in second.groups), second diff --git a/tests/unit/test_group_scope_sqlite.py b/tests/unit/test_group_scope_sqlite.py new file mode 100644 index 000000000..a982ae6ed --- /dev/null +++ b/tests/unit/test_group_scope_sqlite.py @@ -0,0 +1,84 @@ +"""Group consumers keep global aggregate cards usable without widening reads.""" + +from __future__ import annotations + +import sqlite3 +import json +from uuid import uuid4 + +import pytest + +from alicebot_api.sqlite_schema import bootstrap_sqlite_schema +from alicebot_api.sqlite_store import SQLiteVNextStore, ensure_sqlite_user +from alicebot_api.vnext_derived_labels import group_scope +from alicebot_api.vnext_memory_commit import VNextMemoryCommitService, VNextMemoryCommitValidationError +from alicebot_api.vnext_rollups import VNextRollupService + +ALPHA = "prj_" + "a" * 16 +BETA = "prj_" + "b" * 16 + + +def seed_members(store): + members = [] + for index, (text, day) in enumerate((("I played Hollow Knight for 25 hours", "2023-06-02"), + ("I played Stardew Valley for 85 hours", "2023-06-20"), + ("I played Celeste for 10 hours", "2023-07-01"))): + members.append(store.create_memory({"memory_key": f"group-{index}", "memory_type": "episode", + "title": text, "canonical_text": text, "summary": text, "status": "active", "value": {"text": text}, + "domain": "personal", "sensitivity": "internal", + "metadata_json": {"session_date": day, "project_scope": [ALPHA, BETA]}})) + return members + + +@pytest.fixture +def sqlite_group_store(): + conn = sqlite3.connect(":memory:") + bootstrap_sqlite_schema(conn) + user_id = str(uuid4()) + ensure_sqlite_user(conn, user_id, "group@example.invalid", "Group") + yield SQLiteVNextStore(conn, user_id) + conn.close() + + +def test_a_consolidation_candidate_over_a_two_project_group_is_accepted(sqlite_group_store): + store = sqlite_group_store + members = seed_members(store) + first = VNextRollupService(store).propose_rollups(projects=(ALPHA,)) + assert len(first.candidate_ids) == 1 + candidate = store.get_memory(first.candidate_ids[0]) + assert candidate["metadata_json"]["project_scope"] == [] + assert group_scope(candidate) == group_scope(members[0]) + accepted = VNextMemoryCommitService(store).accept_consolidation_candidate(str(candidate["id"]), reason="Reviewed synthetic group") + assert accepted["status"] == "accepted" + + +@pytest.mark.parametrize("accept", (False, True)) +def test_a_second_scoped_rollup_run_over_the_same_group_finds_its_card_and_inserts_nothing(sqlite_group_store, accept): + store = sqlite_group_store + seed_members(store) + service = VNextRollupService(store) + first = service.propose_rollups(projects=(ALPHA,)) + assert len(first.candidate_ids) == 1 + if accept: + VNextMemoryCommitService(store).accept_consolidation_candidate(first.candidate_ids[0], reason="Reviewed synthetic group") + before = store.conn.execute("SELECT count(*) FROM memories").fetchone()[0] + second = service.propose_rollups(projects=(ALPHA,)) + assert second.proposals == [] + assert store.conn.execute("SELECT count(*) FROM memories").fetchone()[0] == before + assert any(group["state"] == ("already_covered_by_accepted" if accept else "existing_candidate") for group in second.groups), second + + +def test_a_cluster_of_promoted_copies_with_different_floors_is_refused_as_crossing_project_scopes(sqlite_group_store): + store = sqlite_group_store + members = seed_members(store) + first = VNextRollupService(store).propose_rollups(projects=(ALPHA,)) + candidate_id = first.candidate_ids[0] + # Preserve the snapshot apart from the group labels so the scope check is reached. + member = members[0] + metadata = dict(member["metadata_json"]) + metadata["project_floor"] = [ALPHA] + metadata["project_scope"] = [] + store.conn.execute("UPDATE memories SET metadata_json=? WHERE id=?", (json.dumps(metadata), member["id"])) + candidate = store.get_memory(candidate_id) + with pytest.raises(VNextMemoryCommitValidationError, match="crosses project scopes"): + VNextMemoryCommitService(store).accept_consolidation_candidate(candidate_id, reason="Reviewed synthetic group") diff --git a/tests/unit/test_project_floor_views.py b/tests/unit/test_project_floor_views.py new file mode 100644 index 000000000..0914388c4 --- /dev/null +++ b/tests/unit/test_project_floor_views.py @@ -0,0 +1,75 @@ +"""Project views agree across both SQL builders and every Python mirror.""" + +from __future__ import annotations + +import inspect +import itertools +import json +import sqlite3 + +import pytest + +from alicebot_api.mcp.retrieval_shared import _resource_matches_project_scope +from alicebot_api.session_briefing import _memory_honours_fence +from alicebot_api.vnext_project_scope import GLOBAL_PROJECT_MARKER, project_scopes_overlap +from alicebot_api.vnext_retrieval import _ResolvedRetrievalScope, _project_scope_meets, _row_matches_scope +from alicebot_api.vnext_stores.sqlite.query_predicates import ( + _ensure_project_scope_identity_sqlite, + _project_view_sql, + _view_membership_sql, +) + +ALPHA = "prj_" + "a" * 16 +BETA = "prj_" + "b" * 16 +SCOPES = ((), (ALPHA,), (BETA,), (ALPHA, BETA), ("Alice",), (ALPHA.upper(),)) +VIEWS = ((), (ALPHA,), (BETA,), (ALPHA, GLOBAL_PROJECT_MARKER), + (BETA, GLOBAL_PROJECT_MARKER), (GLOBAL_PROJECT_MARKER,), (ALPHA, BETA, GLOBAL_PROJECT_MARKER)) + + +def expected(scope, floor, view): + if not view: + return True + ids = {value.lower() for value in view if value != GLOBAL_PROJECT_MARKER} + stored = {value.lower() for value in scope} + alice = lambda values: {value for value in values if value in {ALPHA, BETA}} + return bool(stored & ids) or ( + GLOBAL_PROJECT_MARKER in view and not alice(stored) and alice(value.lower() for value in floor) <= ids + ) + + +@pytest.mark.parametrize("scope,floor,view", itertools.product(SCOPES, SCOPES, VIEWS)) +def test_sql_and_python_project_view_membership_grid(scope, floor, view): + row = {"domain": "project", "sensitivity": "public", "metadata_json": { + "project_scope": list(scope), "project_floor": list(floor), + }} + want = expected(scope, floor, view) + assert project_scopes_overlap(scope, view, floor=floor) == (want if view else False) + assert _project_scope_meets(set(scope), view, floor=floor) == (want if view else False) + assert _resource_matches_project_scope(row, view) == want + resolved = _ResolvedRetrievalScope(frozenset(view), frozenset(), None, None, frozenset()) + assert _row_matches_scope(row, resolved) == want + assert _memory_honours_fence(row, effective_domains=(), effective_sensitivity_allowed=("public",), + effective_project_scope=view, exclude_global_domains=frozenset()) == want + with sqlite3.connect(":memory:") as conn: + _ensure_project_scope_identity_sqlite(conn) + conn.execute("CREATE TABLE rows(metadata_json TEXT, project_id TEXT, domain TEXT)") + conn.execute("INSERT INTO rows VALUES (?, NULL, 'project')", (json.dumps(row["metadata_json"]),)) + placeholders = lambda values: ",".join("?" for _ in values) + kwargs = dict(placeholders=placeholders, scope_expression="alice_project_scope_identity(metadata_json,project_id)", + text_expressions=("metadata_json", "project_id"), domain_expression="domain", + floor_expression="alice_project_floor_identity(metadata_json)") + clause, params = _project_view_sql(**kwargs, projects=view, global_excluded_domains=()) + assert bool(conn.execute("SELECT count(*) FROM rows WHERE 1=1" + clause, params).fetchone()[0]) == want + if view: + ids = tuple(value.lower() for value in view if value != GLOBAL_PROJECT_MARKER) + for partition in (False, True): + sql, params = _view_membership_sql(**kwargs, ids=ids, wants_global=GLOBAL_PROJECT_MARKER in view, + global_excluded_domains=(), partition=partition) + got = conn.execute("SELECT " + sql + " FROM rows", params).fetchone()[0] + assert (got is not None if partition else bool(got)) == want + + +def test_every_marker_aware_python_mirror_passes_the_floor(): + for function in (_project_scope_meets, _row_matches_scope, _resource_matches_project_scope, _memory_honours_fence, + _view_membership_sql, _project_view_sql): + assert "floor" in inspect.getsource(function), function.__name__ From 9ddcb6ab3d668e05c291d8cc712b788fc27649c8 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Tue, 6 Oct 2026 00:19:08 +0200 Subject: [PATCH 60/90] Cover missing canonical dependency count records --- tests/unit/test_derived_labels_record_completeness.py | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/tests/unit/test_derived_labels_record_completeness.py b/tests/unit/test_derived_labels_record_completeness.py index ff9d6b679..6bc06813d 100644 --- a/tests/unit/test_derived_labels_record_completeness.py +++ b/tests/unit/test_derived_labels_record_completeness.py @@ -44,3 +44,13 @@ def test_canonical_counts_and_identifiers_do_not_use_the_legacy_duplicate_except "counts": {"sources": count, "memories": 0, "open_loops": 0, "artifacts": 0, "beliefs": 0}, }}} assert settle_labels([source, row]).by_stored("artifact", "report").unverified is True + + +@pytest.mark.parametrize("counts", (None, {}, {"sources": True}, {"sources": "1"}, [])) +def test_a_nonempty_canonical_record_requires_well_formed_complete_counts(counts): + source = {"kind": "source", "id": "source", "domain": "project", "sensitivity": "public", "metadata_json": {}} + row = {"kind": "artifact", "id": "report", "metadata_json": {"derived_from": { + "v": 1, "sources": ["source"], "memories": [], "open_loops": [], "artifacts": [], "beliefs": [], + "counts": counts, + }}} + assert settle_labels([source, row]).by_stored("artifact", "report").unverified is True From 1ba95930b46ab71279bd75f0b5faf712825c94bb Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Tue, 6 Oct 2026 00:23:12 +0200 Subject: [PATCH 61/90] Retain reviewed facade additions in producer carrier receipts --- tests/unit/test_store_graph_open_loops_split.py | 2 +- tests/unit/test_store_memory_access_split.py | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/tests/unit/test_store_graph_open_loops_split.py b/tests/unit/test_store_graph_open_loops_split.py index fade2ff5b..91cea70ad 100644 --- a/tests/unit/test_store_graph_open_loops_split.py +++ b/tests/unit/test_store_graph_open_loops_split.py @@ -162,7 +162,7 @@ # Previous receipt: (171, 526374782104a2a1...). Proof: the member list equals the list at origin/main # 040a2a10 with ``savepoint`` added at the end and nothing else moved (reviewed change, not drift). # lock_label_writes and read_label_rows follow __init__. Previous receipt (172, 6f1a459f...). - "PostgresVNextStore": (174, "095250b8a77d6c0a32d2783343916b947bcae8ae86e3a1693e47c2fb11802211"), + "PostgresVNextStore": (176, "3549d6ea179f4ebc2341f1c251f139dfb747d56700a4554aef114150efed5c3f"), # One SQLite-only method more, ``check_source_search_query``: the Postgres # source search has no expression-depth or LIKE-length limit to check. # Merge of #500 and #502 (2026-10-01): one more SQLite-only method, diff --git a/tests/unit/test_store_memory_access_split.py b/tests/unit/test_store_memory_access_split.py index 8fde85850..205f6510b 100644 --- a/tests/unit/test_store_memory_access_split.py +++ b/tests/unit/test_store_memory_access_split.py @@ -220,7 +220,7 @@ # 040a2a10 with ``savepoint`` added at the end and nothing else moved (reviewed change, not drift). # Label lock: lock_label_writes and read_label_rows follow __init__. Dropping # those two names restores the previous receipt (172, 6f1a459f...). - "PostgresVNextStore": (174, "095250b8a77d6c0a32d2783343916b947bcae8ae86e3a1693e47c2fb11802211"), + "PostgresVNextStore": (176, "3549d6ea179f4ebc2341f1c251f139dfb747d56700a4554aef114150efed5c3f"), # One SQLite-only method more, ``check_source_search_query``: the Postgres # source search has no expression-depth or LIKE-length limit to check. # Merge of #500 and #502 (2026-10-01): one more SQLite-only method, From bf42c2a5fae6f63885e2e6f383addbb4c9cd2831 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Mon, 5 Oct 2026 23:33:00 +0200 Subject: [PATCH 62/90] Fix effective project scope and exact audit authorization --- .../alicebot_api/routers/vnext_memories.py | 29 +++++++++++++++++-- apps/api/src/alicebot_api/sqlite_store.py | 2 +- .../api/src/alicebot_api/vnext_label_guard.py | 6 ++++ apps/api/src/alicebot_api/vnext_store.py | 2 +- 4 files changed, 35 insertions(+), 4 deletions(-) diff --git a/apps/api/src/alicebot_api/routers/vnext_memories.py b/apps/api/src/alicebot_api/routers/vnext_memories.py index 27ff748e4..ce23ff584 100644 --- a/apps/api/src/alicebot_api/routers/vnext_memories.py +++ b/apps/api/src/alicebot_api/routers/vnext_memories.py @@ -1,5 +1,6 @@ from __future__ import annotations +from collections.abc import Mapping from datetime import UTC, datetime from typing import Literal from uuid import UUID @@ -42,6 +43,7 @@ _vnext_agent_identity, _vnext_agent_record, _vnext_authenticated_agent_identity, + _vnext_exact_resource_policy, _vnext_load_source_trace, _vnext_metadata, _vnext_permission_response, @@ -1855,12 +1857,35 @@ def list_vnext_recent_memory_commits(user_id: UUID, limit: int = Query(default=2 @memory_router.get("/v0/vnext/memories/{memory_id}/audit") -def get_vnext_memory_audit(memory_id: UUID, user_id: UUID) -> JSONResponse: +def get_vnext_memory_audit( + memory_id: UUID, + user_id: UUID, + authorization: str | None = Header(default=None), +) -> JSONResponse: + from alicebot_api.vnext_label_guard import apply_unverified_rule, effective_row_for_fence + settings = get_settings() try: with user_connection(settings.database_url, user_id) as conn: store = PostgresVNextStore(conn) - payload = VNextMemoryCommitService(store).audit(memory_id=str(memory_id)) + identity = resolve_protected_agent_identity( + store, user_id=user_id, raw_key=agent_key_from_authorization(authorization), payload={}, + ) + + def authorize_memory(memory: Mapping[str, object]) -> None: + effective = effective_row_for_fence(store, identity, "memory", memory) + decision = _vnext_exact_resource_policy(identity=identity, action="memory.audit", resource=dict(effective)) + decision = apply_unverified_rule(decision, effective, identity) + append_policy_events(store, identity=identity, decision=decision, target_type="memory", target_id=str(memory["id"])) + if decision.decision == "blocked": + raise AgentPolicyBlockedError(decision) + + try: + payload = VNextMemoryCommitService(store).audit(memory_id=str(memory_id), authorize_memory=authorize_memory) + except AgentPolicyBlockedError as exc: + return _vnext_permission_response(exc.decision) + except AgentKeyAuthenticationError as exc: + return _vnext_agent_auth_error_response(exc) except VNextMemoryCommitValidationError as exc: return public_exception_response(exc, status_code=404) return JSONResponse(status_code=200, content=jsonable_encoder(payload)) diff --git a/apps/api/src/alicebot_api/sqlite_store.py b/apps/api/src/alicebot_api/sqlite_store.py index 4808aceb8..05d3be249 100644 --- a/apps/api/src/alicebot_api/sqlite_store.py +++ b/apps/api/src/alicebot_api/sqlite_store.py @@ -419,7 +419,7 @@ def read_label_rows(self, kind: str, ids: Sequence[str]) -> list[VNextRow]: return [] extra = "" if table == "memories": - extra = ", value, project_id" + extra = ", value, project_id, source_event_ids, deleted_at, status" elif table == "open_loops": extra = ", project_id, source_id, memory_id" from uuid import UUID diff --git a/apps/api/src/alicebot_api/vnext_label_guard.py b/apps/api/src/alicebot_api/vnext_label_guard.py index e6c3afb4f..2723ce2be 100644 --- a/apps/api/src/alicebot_api/vnext_label_guard.py +++ b/apps/api/src/alicebot_api/vnext_label_guard.py @@ -110,6 +110,12 @@ def effective_row(self, kind: str, row: Mapping[str, object] | None) -> Mapping[ metadata["project_floor"] = list(label.project_floor) copy["unverified"] = False copy["metadata_json"] = metadata + # Store records expose scope and floor at the top level as well. The + # resolver reads those first, so both representations must agree. + copy["project_scope"] = list(metadata["project_scope"]) + copy["project_floor"] = list(metadata["project_floor"]) + if not copy["project_scope"]: + copy["project_id"] = None return copy def admit_rows(self, kind: str, rows: Sequence[Mapping[str, object]]) -> list[Mapping[str, object]]: diff --git a/apps/api/src/alicebot_api/vnext_store.py b/apps/api/src/alicebot_api/vnext_store.py index 0baadf2a7..5ab808b1a 100644 --- a/apps/api/src/alicebot_api/vnext_store.py +++ b/apps/api/src/alicebot_api/vnext_store.py @@ -514,7 +514,7 @@ def read_label_rows(self, kind: str, ids: Sequence[str]) -> list[VNextRow]: ) extra = "" if table == "memories": - extra = ", value, project_id" + extra = ", value, project_id, source_event_ids, deleted_at, status" elif table == "open_loops": extra = ", project_id, source_id, memory_id" elif table == "beliefs": From 4fe175a9237fa5cb2b8ccc8defcd3a2c2eac2d76 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Tue, 6 Oct 2026 00:27:11 +0200 Subject: [PATCH 63/90] Follow canonical encoded references during label propagation --- .../src/alicebot_api/vnext_label_writes.py | 18 +++- apps/api/src/alicebot_api/vnext_store.py | 2 +- ...est_encoded_label_dependencies_postgres.py | 95 +++++++++++++++++++ tests/unit/test_encoded_label_dependencies.py | 76 +++++++++++++++ 4 files changed, 186 insertions(+), 5 deletions(-) create mode 100644 tests/integration/test_encoded_label_dependencies_postgres.py create mode 100644 tests/unit/test_encoded_label_dependencies.py diff --git a/apps/api/src/alicebot_api/vnext_label_writes.py b/apps/api/src/alicebot_api/vnext_label_writes.py index 04dba3302..9265da97e 100644 --- a/apps/api/src/alicebot_api/vnext_label_writes.py +++ b/apps/api/src/alicebot_api/vnext_label_writes.py @@ -366,11 +366,14 @@ def _sqlite_dependants(store: Any, table: str, kind: str, compacts: Sequence[str if table == "memories": extra = ", value, project_id, NULL AS source_id, NULL AS memory_id" text_clause, _ = _like_clause("metadata_json", len(compacts), qmark=True) + # A nested JSON string may encode every character of an id. Keep all + # escaped candidates for the canonical dependency parser below. + text_clause += " OR instr(coalesce(metadata_json, ''), char(92)) > 0" params: list[object] = [store.user_id, *[f"%{item}%" for item in compacts]] value_sql = "" if with_value: value_clause, _ = _like_clause("value", len(compacts), qmark=True) - value_sql = f" OR {value_clause}" + value_sql = f" OR {value_clause} OR instr(coalesce(value, ''), char(92)) > 0" params.extend(f"%{item}%" for item in compacts) column_sql = "" if table == "open_loops": @@ -396,23 +399,30 @@ def _postgres_dependants(store: Any, table: str, kind: str, compacts: Sequence[s if table == "memories": extra = ", value, project_id" elif table == "open_loops": - extra = ", NULL::jsonb AS value, project_id, source_id, memory_id" + extra = ", NULL::jsonb AS value, project_id, source_id::text AS source_id, memory_id::text AS memory_id" elif table == "generated_artifacts": extra = ", NULL::jsonb AS value, artifact_type" else: extra = ", NULL::jsonb AS value" text_clause, _ = _like_clause("metadata_json::text", len(compacts), qmark=False) + text_clause += " OR strpos(coalesce(metadata_json::text, ''), chr(92)) > 0" params: list[object] = [f"%{item}%" for item in compacts] value_sql = "" if with_value: value_clause, _ = _like_clause("value::text", len(compacts), qmark=False) - value_sql = f" OR {value_clause}" + value_sql = f" OR {value_clause} OR strpos(coalesce(value::text, ''), chr(92)) > 0" params.extend(f"%{item}%" for item in compacts) + column_sql = "" + if table == "open_loops": + for column in ("source_id", "memory_id"): + clause, _ = _like_clause(f"{column}::text", len(compacts), qmark=False) + column_sql += f" OR {clause}" + params.extend(f"%{item}%" for item in compacts) rows = store._fetch_all( f""" SELECT id::text AS id, user_id::text AS user_id, domain, sensitivity, metadata_json{extra} FROM {table} - WHERE ({text_clause}{value_sql}) + WHERE ({text_clause}{value_sql}{column_sql}) """, # nosec B608 # internal literal table/columns; every external value is bound tuple(params), ) diff --git a/apps/api/src/alicebot_api/vnext_store.py b/apps/api/src/alicebot_api/vnext_store.py index f172b2e6a..2a6a4e209 100644 --- a/apps/api/src/alicebot_api/vnext_store.py +++ b/apps/api/src/alicebot_api/vnext_store.py @@ -513,7 +513,7 @@ def read_label_rows(self, kind: str, ids: Sequence[str]) -> list[VNextRow]: if table == "memories": extra = ", value, project_id" elif table == "open_loops": - extra = ", project_id, source_id, memory_id" + extra = ", project_id, source_id::text AS source_id, memory_id::text AS memory_id" elif table == "beliefs": extra = ", memory_id" elif table == "generated_artifacts": diff --git a/tests/integration/test_encoded_label_dependencies_postgres.py b/tests/integration/test_encoded_label_dependencies_postgres.py new file mode 100644 index 000000000..cd7c5f701 --- /dev/null +++ b/tests/integration/test_encoded_label_dependencies_postgres.py @@ -0,0 +1,95 @@ +"""Actual PostgreSQL encoded dependency and confirmation controls.""" +import json +from uuid import UUID, uuid4 + +import pytest +from psycopg.types.json import Jsonb + +from alicebot_api.config import Settings +from alicebot_api.db import user_connection +from alicebot_api.routers import vnext_memories as router +from alicebot_api.store import ContinuityStore +from alicebot_api.vnext_derived_labels import dependencies_of +from alicebot_api.vnext_label_writes import walk_dependants +from alicebot_api.vnext_store import PostgresVNextStore +from tests.integration.test_source_move_label_preview_postgres import ALPHA, BETA, _snapshot +from tests.unit.test_encoded_label_dependencies import encoded_object + + +def test_encoded_source_move_previews_without_writes_then_confirms(migrated_database_urls, monkeypatch): + url = migrated_database_urls["app"] + user = uuid4() + with user_connection(url, user) as conn: + ContinuityStore(conn).create_user(user, f"encoded-{user}@example.test", "Synthetic") + store = PostgresVNextStore(conn) + source = store.create_source({"source_type": "note", "title": "Synthetic", "content_hash": str(user), "domain": "project", "sensitivity": "public", "metadata_json": {"project_scope": [ALPHA]}}) + reference = encoded_object("source_id", str(source["id"])) + copy = store.create_memory({"memory_key": "encoded", "canonical_text": "Synthetic", "status": "active", "domain": "project", "sensitivity": "public", "metadata_json": {"source_id": reference, "project_scope": [ALPHA]}}) + unrelated = store.create_memory({"memory_key": "unrelated", "canonical_text": "Synthetic", "domain": "project", "sensitivity": "public", "metadata_json": {"note": encoded_object("source_id", str(uuid4()))}}) + assert ("source", str(source["id"])) in dependencies_of("memory", copy) + assert {str(row["id"]) for row in walk_dependants(store, [str(source["id"])])} == {str(copy["id"])} + monkeypatch.setattr(router, "get_settings", lambda: Settings(database_url=url)) + before = _snapshot(url, user, source) + request = router.VNextSourceReviewRequest(user_id=user, action="assign_project", project_id=BETA, sensitivity="confidential") + preview = router.review_vnext_source(UUID(str(source["id"])), request) + payload = json.loads(preview.body) + assert preview.status_code == 200 and payload["preview"] is True + assert payload["derived_rows_hidden_from_project_keys"] == 1 + assert payload["confirm_required"] is True + assert _snapshot(url, user, source) == before + confirmed = router.review_vnext_source(UUID(str(source["id"])), request.model_copy(update={"confirm_label_hide": True})) + assert confirmed.status_code == 200 + with user_connection(url, user) as conn: + store = PostgresVNextStore(conn) + stored = store.get_memory(str(copy["id"])) + assert stored["sensitivity"] == "confidential" + assert stored["metadata_json"]["source_id"] == reference + assert BETA in stored["metadata_json"]["project_floor"] + assert store.get_memory(str(unrelated["id"]))["sensitivity"] == "public" + assert _snapshot(url, user, source)["provenance_links"] == before["provenance_links"] + + +@pytest.mark.parametrize("key,kind", [("source_id", "source"), ("memory_id", "memory"), ("artifact_id", "artifact"), ("belief_ids", "belief")]) +def test_encoded_metadata_references_keep_the_canonical_reverse_edge(migrated_database_urls, key, kind): + user = uuid4() + root_id = str(uuid4()) + with user_connection(migrated_database_urls["app"], user) as conn: + ContinuityStore(conn).create_user(user, f"encoded-keys-{user}@example.test", "Synthetic") + store = PostgresVNextStore(conn) + row = store.create_memory({"memory_key": "encoded-key", "canonical_text": "Synthetic", "domain": "project", "sensitivity": "public"}) + metadata = {key: [root_id] if key.endswith("ids") else root_id, "consolidation": {}} + raw = json.dumps(metadata) + encoded = "".join("\\u%04x" % ord(char) if char.isalnum() or char == "_" else char for char in raw) + conn.execute("UPDATE memories SET metadata_json = %s::jsonb WHERE id = %s", (encoded, row["id"])) + stored = store.get_memory(str(row["id"])) + assert (kind, root_id) in dependencies_of("memory", stored) + assert {str(item["id"]) for item in walk_dependants(store, [root_id])} == {str(row["id"])} + + +@pytest.mark.parametrize("key,kind", [("source_id", "source"), ("artifact_id", "artifact")]) +def test_encoded_value_object_keeps_its_canonical_reverse_edge(migrated_database_urls, key, kind): + user = uuid4() + root_id = str(uuid4()) + with user_connection(migrated_database_urls["app"], user) as conn: + ContinuityStore(conn).create_user(user, f"encoded-value-{user}@example.test", "Synthetic") + store = PostgresVNextStore(conn) + row = store.create_memory({"memory_key": "encoded-value", "canonical_text": "Synthetic", "domain": "project", "sensitivity": "public"}) + conn.execute("UPDATE memories SET value = %s, metadata_json = %s WHERE id = %s", (Jsonb(encoded_object(key, root_id)), Jsonb({"consolidation": {}}), row["id"])) + stored = store.get_memory(str(row["id"])) + assert (kind, root_id) in dependencies_of("memory", stored) + assert {str(item["id"]) for item in walk_dependants(store, [root_id])} == {str(row["id"])} + + +def test_a_candidate_open_loop_direct_source_column_is_a_reverse_edge(migrated_database_urls): + user = uuid4() + with user_connection(migrated_database_urls["app"], user) as conn: + ContinuityStore(conn).create_user(user, f"loop-edge-{user}@example.test", "Synthetic") + store = PostgresVNextStore(conn) + store.lock_graph_mutation() + source = store.create_source({"source_type": "note", "title": "Synthetic", "content_hash": str(user), "domain": "project", "sensitivity": "public"}) + loop = store.create_open_loop({"title": "Synthetic", "loop_type": "task", "source_id": str(source["id"]), "domain": "project", "sensitivity": "public", "metadata_json": {"discovered_by": "synthetic"}}) + assert {str(row["id"]) for row in walk_dependants(store, [str(source["id"])])} == {str(loop["id"])} + store.lock_graph_mutation() + store.lock_label_writes(exclusive=True) + store.update_source(source_id=str(source["id"]), patch={"sensitivity": "confidential"}) + assert store.get_open_loop(str(loop["id"]))["sensitivity"] == "confidential" diff --git a/tests/unit/test_encoded_label_dependencies.py b/tests/unit/test_encoded_label_dependencies.py new file mode 100644 index 000000000..fb94635ff --- /dev/null +++ b/tests/unit/test_encoded_label_dependencies.py @@ -0,0 +1,76 @@ +"""The reverse lookup remains a superset of canonical encoded references.""" +import json +from uuid import uuid4 + +import pytest + +from alicebot_api.onramp import bootstrap_database +from alicebot_api.sqlite_store import SQLiteVNextStore, sqlite_user_connection +from alicebot_api.vnext_derived_labels import dependencies_of +from alicebot_api.vnext_label_writes import count_rows_hidden_by_scope_move, walk_dependants +from tests.unit.test_derived_domain_fence import USER + +ALPHA = "prj_" + "a" * 16 +BETA = "prj_" + "b" * 16 + + +def encoded_object(key, row_id): + escaped = "".join("\\u%04x" % ord(character) for character in row_id) + return '{"' + key + '":"' + escaped + '"}' + + +def test_an_encoded_source_is_previewed_and_raised_without_losing_its_reference(tmp_path): + path = tmp_path / "encoded.sqlite3" + bootstrap_database(path, user_id=USER, user_email="synthetic@example.test") + with sqlite_user_connection(path, USER) as conn: + store = SQLiteVNextStore(conn, USER) + source = store.create_source({"source_type": "note", "title": "Synthetic", "content_hash": "encoded", "domain": "project", "sensitivity": "public", "metadata_json": {"project_scope": [ALPHA]}}) + reference = encoded_object("source_id", str(source["id"])) + copy = store.create_memory({"memory_key": "encoded", "canonical_text": "Synthetic", "status": "active", "domain": "project", "sensitivity": "public", "metadata_json": {"source_id": reference, "project_scope": [ALPHA]}}) + unrelated = store.create_memory({"memory_key": "unrelated", "canonical_text": "Synthetic", "domain": "project", "sensitivity": "public", "metadata_json": {"note": encoded_object("source_id", str(uuid4()))}}) + before = {table: list(conn.execute(f"SELECT * FROM {table} ORDER BY id")) for table in ("sources", "memories", "event_log", "provenance_links")} + assert ("source", str(source["id"])) in dependencies_of("memory", copy) + assert {str(row["id"]) for row in walk_dependants(store, [str(source["id"])])} == {str(copy["id"])} + assert count_rows_hidden_by_scope_move(store, source, [BETA]) == 1 + assert {table: list(conn.execute(f"SELECT * FROM {table} ORDER BY id")) for table in before} == before + store.update_source(source_id=str(source["id"]), patch={"sensitivity": "confidential", "metadata_json": {"project_scope": [BETA]}}) + stored = store.get_memory(str(copy["id"])) + assert stored["sensitivity"] == "confidential" + assert BETA in stored["metadata_json"]["project_floor"] + assert stored["metadata_json"]["source_id"] == reference + assert store.get_memory(str(unrelated["id"]))["sensitivity"] == "public" + + +@pytest.mark.parametrize("key,kind", [("source_id", "source"), ("memory_id", "memory"), ("artifact_id", "artifact"), ("belief_ids", "belief")]) +def test_unicode_encoded_metadata_keys_and_ids_are_not_cut_before_the_parser(tmp_path, key, kind): + path = tmp_path / "encoded-keys.sqlite3" + bootstrap_database(path, user_id=USER, user_email="synthetic@example.test") + row_id = str(uuid4()) + value = [row_id] if key.endswith("ids") else row_id + metadata = json.dumps({key: value, "consolidation": {}}) + encoded = "".join("\\u%04x" % ord(character) if character.isalnum() or character == "_" else character for character in metadata) + with sqlite_user_connection(path, USER) as conn: + store = SQLiteVNextStore(conn, USER) + row = store.create_memory({"memory_key": "raw-encoded", "canonical_text": "Synthetic", "domain": "project", "sensitivity": "public"}) + conn.execute("UPDATE memories SET metadata_json = ? WHERE id = ?", (encoded, row["id"])) + stored = store.get_memory(str(row["id"])) + assert (kind, row_id) in dependencies_of("memory", stored) + assert {str(item["id"]) for item in walk_dependants(store, [row_id])} == {str(row["id"])} + + +@pytest.mark.parametrize("key,kind", [("source_id", "source"), ("artifact_id", "artifact")]) +def test_encoded_value_object_keeps_the_same_reverse_edge(tmp_path, key, kind): + path = tmp_path / "encoded-value.sqlite3" + bootstrap_database(path, user_id=USER, user_email="synthetic@example.test") + row_id = str(uuid4()) + with sqlite_user_connection(path, USER) as conn: + store = SQLiteVNextStore(conn, USER) + row = store.create_memory({"memory_key": "encoded-value", "canonical_text": "Synthetic", "domain": "project", "sensitivity": "public"}) + # The dependency lives in the encoded value alone. A marker without + # the root id states the row class without making SQL's id term pass. + metadata = {"consolidation": {}} + conn.execute("UPDATE memories SET value = ?, metadata_json = ? WHERE id = ?", (json.dumps(encoded_object(key, row_id)), json.dumps(metadata), row["id"])) + stored = store.get_memory(str(row["id"])) + assert (kind, row_id) in dependencies_of("memory", stored) + assert {str(item["id"]) for item in walk_dependants(store, [row_id])} == {str(row["id"])} + From 84778ae91f377576b0438563b0afd647c0a79e7f Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Tue, 6 Oct 2026 00:43:44 +0200 Subject: [PATCH 64/90] Normalize native PostgreSQL loop references at exact read boundaries --- apps/api/src/alicebot_api/vnext_stores/postgres/columns.py | 4 ++-- tests/unit/test_store_graph_open_loops_split.py | 5 +++-- 2 files changed, 5 insertions(+), 4 deletions(-) diff --git a/apps/api/src/alicebot_api/vnext_stores/postgres/columns.py b/apps/api/src/alicebot_api/vnext_stores/postgres/columns.py index f97ff71d9..f5ca74801 100644 --- a/apps/api/src/alicebot_api/vnext_stores/postgres/columns.py +++ b/apps/api/src/alicebot_api/vnext_stores/postgres/columns.py @@ -172,7 +172,7 @@ OPEN_LOOP_COLUMNS = """ id, user_id, - memory_id, + memory_id::text AS memory_id, title, status, opened_at, @@ -185,7 +185,7 @@ priority, project_id, person_id, - source_id, + source_id::text AS source_id, closed_at, domain, sensitivity, diff --git a/tests/unit/test_store_graph_open_loops_split.py b/tests/unit/test_store_graph_open_loops_split.py index 91cea70ad..bd68d493d 100644 --- a/tests/unit/test_store_graph_open_loops_split.py +++ b/tests/unit/test_store_graph_open_loops_split.py @@ -134,7 +134,7 @@ # Re-minted so the open-loop partition read passes the floor identity. # Previous receipt 9a2634be... SQLITE_CARRIER_PATH: "a050eda266e928f9289730a941f6f10d8ad048c6eb4b288fa57e8121e717b83c", - POSTGRES_COLUMNS_PATH: "5b0d972a55abf8590ce14394a37fd71b9b88ba7ab3de82d61efc1bddfc022b71", + POSTGRES_COLUMNS_PATH: "1a782bf3eb87f68f67434508baab0f82d49cb40a0c547cde49bbc972e2f1d182", SQLITE_COLUMNS_PATH: "be81b8628d0831d3d02b280b5455fb02333db5740ebef8d85d58024384ae6556", } EXPECTED_METHOD_AST_MANIFESTS = { @@ -188,7 +188,8 @@ "c333a0dacf8733a16fb86f3acc1bbf61bd25fa66ce96cc3bc25805c4a85d9203" ), "BELIEF_COLUMNS": "32bac57e9e38fead1af29b9324777978f3b03bfe20d5306695fae98079d82dc7", - "OPEN_LOOP_COLUMNS": "651275ee48d37e13228bf339ac4f260a50333fc7b86197ab16573cc099f912bf", + # Reviewed: normalize the two direct dependency UUID columns to text. + "OPEN_LOOP_COLUMNS": "44970b53460b10e1e414a9c1c7905f15ea7ecfed4570b1fcd5becf65952b3c64", }, SQLITE_COLUMNS_PATH: { "GRAPH_EDGE_COLUMNS": "587b88564c446c03420441371a180e11618ea6bf192e5e20d2ad5d426ce890f2", From 314104dedcf4122fe544a177f7914859306b792d Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Tue, 6 Oct 2026 00:43:13 +0200 Subject: [PATCH 65/90] Normalize reverse dependency candidates once per column --- .../src/alicebot_api/vnext_label_writes.py | 23 +++- .../test_label_reverse_search_postgres.py | 128 ++++++++++++++++++ 2 files changed, 145 insertions(+), 6 deletions(-) create mode 100644 tests/integration/test_label_reverse_search_postgres.py diff --git a/apps/api/src/alicebot_api/vnext_label_writes.py b/apps/api/src/alicebot_api/vnext_label_writes.py index 9265da97e..ca5ca7cdc 100644 --- a/apps/api/src/alicebot_api/vnext_label_writes.py +++ b/apps/api/src/alicebot_api/vnext_label_writes.py @@ -404,20 +404,31 @@ def _postgres_dependants(store: Any, table: str, kind: str, compacts: Sequence[s extra = ", NULL::jsonb AS value, artifact_type" else: extra = ", NULL::jsonb AS value" - text_clause, _ = _like_clause("metadata_json::text", len(compacts), qmark=False) + # Normalize each column once per row, rather than once per frontier id. + # This remains only a superset lookup; the canonical parser selects exact edges. + pattern = "(?:" + "|".join(re.escape(item) for item in compacts) + ")" + + def candidate_clause(column: str) -> str: + return ( + "replace(replace(replace(replace(lower(coalesce(" + + column + + ",'')),'-',''),'{',''),'}',''),' ','') ~ %s" + ) + + text_clause = candidate_clause("metadata_json::text") text_clause += " OR strpos(coalesce(metadata_json::text, ''), chr(92)) > 0" - params: list[object] = [f"%{item}%" for item in compacts] + params: list[object] = [pattern] value_sql = "" if with_value: - value_clause, _ = _like_clause("value::text", len(compacts), qmark=False) + value_clause = candidate_clause("value::text") value_sql = f" OR {value_clause} OR strpos(coalesce(value::text, ''), chr(92)) > 0" - params.extend(f"%{item}%" for item in compacts) + params.append(pattern) column_sql = "" if table == "open_loops": for column in ("source_id", "memory_id"): - clause, _ = _like_clause(f"{column}::text", len(compacts), qmark=False) + clause = candidate_clause(f"{column}::text") column_sql += f" OR {clause}" - params.extend(f"%{item}%" for item in compacts) + params.append(pattern) rows = store._fetch_all( f""" SELECT id::text AS id, user_id::text AS user_id, domain, sensitivity, metadata_json{extra} diff --git a/tests/integration/test_label_reverse_search_postgres.py b/tests/integration/test_label_reverse_search_postgres.py new file mode 100644 index 000000000..f68ebd541 --- /dev/null +++ b/tests/integration/test_label_reverse_search_postgres.py @@ -0,0 +1,128 @@ +"""The faster reverse search preserves the previous superset and exact closure.""" + +from uuid import uuid4 + +from alicebot_api.db import user_connection +from alicebot_api.store import ContinuityStore +from alicebot_api.vnext_label_writes import _compact_id, _like_clause, _postgres_dependants, walk_dependants +from alicebot_api.vnext_store import PostgresVNextStore +from tests.unit.test_encoded_label_dependencies import encoded_object + + +def test_the_regex_candidate_search_matches_the_previous_scan_for_aliases_and_encoded_objects(migrated_database_urls): + user = uuid4() + roots = [str(uuid4()) for _ in range(3)] + with user_connection(migrated_database_urls["app"], user) as conn: + ContinuityStore(conn).create_user(user, f"reverse-{user}@example.invalid", "Reverse") + store = PostgresVNextStore(conn) + expected = set() + for index, root in enumerate(roots): + for variant, reference in enumerate( + (root, root.upper(), root.replace("-", ""), "{" + root.upper() + "}", encoded_object("source_id", root)) + ): + row = store.create_memory( + { + "memory_key": f"alias.{index}.{variant}", + "canonical_text": "Synthetic alias", + "domain": "project", + "sensitivity": "public", + "metadata_json": {"source_id": reference}, + } + ) + expected.add(str(row["id"])) + decoy = store.create_memory( + { + "memory_key": "decoy", + "canonical_text": "Synthetic decoy", + "domain": "project", + "sensitivity": "public", + "metadata_json": {"note": roots[0]}, + } + ) + unrelated = store.create_memory( + { + "memory_key": "unrelated", + "canonical_text": "Synthetic unrelated", + "domain": "project", + "sensitivity": "public", + "metadata_json": {"note": encoded_object("source_id", str(uuid4()))}, + } + ) + compacts = [_compact_id(root) for root in roots] + text, _ = _like_clause("metadata_json::text", len(compacts), qmark=False) + value, _ = _like_clause("value::text", len(compacts), qmark=False) + previous = conn.execute( + f"SELECT id::text AS id FROM memories WHERE ({text}) OR ({value}) " + "OR strpos(metadata_json::text,chr(92))>0 OR strpos(value::text,chr(92))>0", + [*(f"%{item}%" for item in compacts), *(f"%{item}%" for item in compacts)], + ).fetchall() + current = _postgres_dependants(store, "memories", "memory", compacts, with_value=True) + assert {row["id"] for row in current} == {row["id"] for row in previous} + assert {row["id"] for row in walk_dependants(store, roots)} == expected + assert str(decoy["id"]) not in expected + assert str(unrelated["id"]) not in expected + + +def test_multiple_frontier_batches_keep_memory_chains_and_both_direct_loop_columns(migrated_database_urls): + user = uuid4() + root = str(uuid4()) + with user_connection(migrated_database_urls["app"], user) as conn: + ContinuityStore(conn).create_user(user, f"frontier-{user}@example.invalid", "Frontier") + store = PostgresVNextStore(conn) + store.lock_graph_mutation() + store.create_source( + { + "id": root, + "source_type": "note", + "title": "Synthetic source", + "content_hash": str(user), + "domain": "project", + "sensitivity": "public", + } + ) + roots = [str(uuid4()) for _ in range(220)] + conn.execute( + """INSERT INTO memories(id,user_id,memory_key,value,source_event_ids,canonical_text, + status,domain,sensitivity,metadata_json) + SELECT id,app.current_user_id(),'frontier.'||id,'{}'::jsonb,'[]'::jsonb,'Synthetic frontier', + 'active','project','public', + jsonb_build_object('source_id',%s::text) FROM unnest(%s::uuid[]) id""", + (root, roots), + ) + descendants = [] + for index in (0, 199, 219): + row = store.create_memory( + { + "memory_key": f"descendant.{index}", + "canonical_text": "Synthetic descendant", + "domain": "project", + "sensitivity": "public", + "metadata_json": {"derived_from": {"v": 1, "memories": [roots[index]], "counts": {"memories": 1}}}, + } + ) + descendants.append(str(row["id"])) + direct_source = store.create_open_loop( + { + "title": "Synthetic source loop", + "source_id": root, + "domain": "project", + "sensitivity": "public", + "metadata_json": {"discovered_by": "synthetic"}, + } + ) + direct_memory = store.create_open_loop( + { + "title": "Synthetic memory loop", + "source_id": root, + "memory_id": roots[-1], + "domain": "project", + "sensitivity": "public", + "metadata_json": {"discovered_by": "synthetic"}, + } + ) + expected = {*roots, *descendants, str(direct_source["id"]), str(direct_memory["id"])} + assert {row["id"] for row in walk_dependants(store, [root])} == expected + memory_candidates = _postgres_dependants( + store, "open_loops", "open_loop", [_compact_id(roots[-1])], with_value=False + ) + assert str(direct_memory["id"]) in {row["id"] for row in memory_candidates} From a6a79dd7d6e4b4b182c0fa31d8d82e0af41603fd Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Tue, 6 Oct 2026 00:46:33 +0200 Subject: [PATCH 66/90] Reuse source label inputs during bounded capture writes --- apps/api/src/alicebot_api/vnext_capture.py | 147 ++++++++-------- .../src/alicebot_api/vnext_label_writes.py | 81 ++++++++- apps/api/src/alicebot_api/vnext_store.py | 3 + .../test_capture_label_batch_postgres.py | 159 ++++++++++++++++++ 4 files changed, 316 insertions(+), 74 deletions(-) create mode 100644 tests/integration/test_capture_label_batch_postgres.py diff --git a/apps/api/src/alicebot_api/vnext_capture.py b/apps/api/src/alicebot_api/vnext_capture.py index 301341ccc..7a4503c27 100644 --- a/apps/api/src/alicebot_api/vnext_capture.py +++ b/apps/api/src/alicebot_api/vnext_capture.py @@ -1542,84 +1542,87 @@ def _capture_source(self, source_input: SourceCaptureInput) -> CaptureResult: sensitivity=source_input.sensitivity, ) memory_rows: list[JsonObject] = [] - for candidate in candidates: - # Speaker provenance is only stamped when a role was derived, - # so provenance-free captures keep byte-identical metadata. - provenance_metadata: JsonObject = ( - { - "provenance_role": candidate.provenance_role, - "assertion_class": candidate.assertion_class, - } - if candidate.provenance_role is not None - else {} - ) - memory = self.store.create_memory( - { - "memory_key": _memory_key( - content_hash=content_hash, - candidate=candidate, - domain=source_input.domain, - sensitivity=source_input.sensitivity, - ), - "value": { - "text": candidate.text, + from alicebot_api.vnext_label_writes import capture_label_inputs + + with capture_label_inputs(self.store, source_id): + for candidate in candidates: + # Speaker provenance is only stamped when a role was derived, + # so provenance-free captures keep byte-identical metadata. + provenance_metadata: JsonObject = ( + { + "provenance_role": candidate.provenance_role, + "assertion_class": candidate.assertion_class, + } + if candidate.provenance_role is not None + else {} + ) + memory = self.store.create_memory( + { + "memory_key": _memory_key( + content_hash=content_hash, + candidate=candidate, + domain=source_input.domain, + sensitivity=source_input.sensitivity, + ), + "value": { + "text": candidate.text, + "source_id": source_id, + "source_chunk_id": candidate.source_chunk_id, + }, + "status": "candidate", + "source_event_ids": [source_id, candidate.source_chunk_id], + "memory_type": candidate.memory_type, + "confidence": candidate.confidence, + "title": _truncate(candidate.text, max_length=120), + "canonical_text": candidate.text, + "summary": _truncate(candidate.text, max_length=280), + "domain": source_input.domain, + "sensitivity": source_input.sensitivity, + "project_id": project_scope[0] if len(project_scope) == 1 else None, + "created_by_agent_id": self.actor_id if self.actor_type == "agent" else None, + "run_id": self.run_id if self.actor_type == "agent" else None, + "metadata_json": { + "source_id": source_id, + "source_chunk_id": candidate.source_chunk_id, + "source_chunk_index": candidate.source_chunk_index, + "extraction_rule": candidate.extraction_rule, + "capture_content_hash": content_hash, + **provenance_metadata, + **project_scope_metadata, + "generated_by": self.actor_type, + "agent_identity": self.agent_identity, + "agent_id": self.actor_id if self.actor_type == "agent" else None, + "agent_run_id": self.run_id if self.actor_type == "agent" else None, + "trace_id": self.trace_id, + "policy_decision": self.policy_decision, + }, + }, + actor_type=self.actor_type, + ) + memory_rows.append(memory) + self.store.create_provenance_link( + { + "target_type": "memory", + "target_id": str(memory["id"]), "source_id": source_id, "source_chunk_id": candidate.source_chunk_id, + "quote": candidate.text, + "evidence_role": "quoted_from", + "confidence": candidate.confidence, }, - "status": "candidate", - "source_event_ids": [source_id, candidate.source_chunk_id], - "memory_type": candidate.memory_type, - "confidence": candidate.confidence, - "title": _truncate(candidate.text, max_length=120), - "canonical_text": candidate.text, - "summary": _truncate(candidate.text, max_length=280), - "domain": source_input.domain, - "sensitivity": source_input.sensitivity, - "project_id": project_scope[0] if len(project_scope) == 1 else None, - "created_by_agent_id": self.actor_id if self.actor_type == "agent" else None, - "run_id": self.run_id if self.actor_type == "agent" else None, - "metadata_json": { + actor_type=self.actor_type, + ) + self._log_event( + event_type="memory.candidate_created", + target_type="memory", + target_id=str(memory["id"]), + payload={ "source_id": source_id, "source_chunk_id": candidate.source_chunk_id, - "source_chunk_index": candidate.source_chunk_index, - "extraction_rule": candidate.extraction_rule, - "capture_content_hash": content_hash, - **provenance_metadata, - **project_scope_metadata, - "generated_by": self.actor_type, - "agent_identity": self.agent_identity, - "agent_id": self.actor_id if self.actor_type == "agent" else None, - "agent_run_id": self.run_id if self.actor_type == "agent" else None, - "trace_id": self.trace_id, - "policy_decision": self.policy_decision, + "memory_type": candidate.memory_type, + "confidence": candidate.confidence, }, - }, - actor_type=self.actor_type, - ) - memory_rows.append(memory) - self.store.create_provenance_link( - { - "target_type": "memory", - "target_id": str(memory["id"]), - "source_id": source_id, - "source_chunk_id": candidate.source_chunk_id, - "quote": candidate.text, - "evidence_role": "quoted_from", - "confidence": candidate.confidence, - }, - actor_type=self.actor_type, - ) - self._log_event( - event_type="memory.candidate_created", - target_type="memory", - target_id=str(memory["id"]), - payload={ - "source_id": source_id, - "source_chunk_id": candidate.source_chunk_id, - "memory_type": candidate.memory_type, - "confidence": candidate.confidence, - }, - ) + ) # Capture writes candidate memories only, and recall cannot return a # candidate, so no text is sent to the embeddings endpoint here: the diff --git a/apps/api/src/alicebot_api/vnext_label_writes.py b/apps/api/src/alicebot_api/vnext_label_writes.py index ca5ca7cdc..28c54c159 100644 --- a/apps/api/src/alicebot_api/vnext_label_writes.py +++ b/apps/api/src/alicebot_api/vnext_label_writes.py @@ -10,8 +10,9 @@ import re from collections.abc import Iterator, Mapping, Sequence from contextlib import contextmanager +from contextvars import ContextVar from functools import wraps -from dataclasses import replace +from dataclasses import dataclass, field, replace from typing import Any from alicebot_api.vnext_agent_control import RESTRICTED_DOMAINS @@ -59,6 +60,73 @@ class LabelLockOrderError(RuntimeError): REFUSED_DETAIL = "the label change could not be applied to every dependent row; nothing was changed" +@dataclass +class _CaptureLabelInputs: + conn: Any + transaction_id: str + rollback_counter: int + source_id: str + rows: dict[tuple[str, str], list[dict[str, object]]] = field(default_factory=dict) + + @property + def key(self) -> tuple[str, int]: + return self.transaction_id, self.rollback_counter + + +_CAPTURE_LABEL_INPUTS: ContextVar[_CaptureLabelInputs | None] = ContextVar("capture_label_inputs", default=None) + + +def invalidate_capture_label_inputs(store: Any) -> None: + batch = _CAPTURE_LABEL_INPUTS.get() + if batch is not None and batch.conn is getattr(store, "conn", None): + batch.rows.clear() + + +def label_savepoint_rolled_back(store: Any) -> None: + """Invalidate capture inputs whenever the store rolls a savepoint back.""" + + conn = store.conn + conn._alice_label_rollback_counter = int(getattr(conn, "_alice_label_rollback_counter", 0)) + 1 + invalidate_capture_label_inputs(store) + + +def _current_capture_inputs(store: Any) -> _CaptureLabelInputs | None: + batch = _CAPTURE_LABEL_INPUTS.get() + if batch is None or batch.conn is not getattr(store, "conn", None) or not _in_transaction(store): + return None + counter = int(getattr(batch.conn, "_alice_label_rollback_counter", 0)) + if counter != batch.rollback_counter: + batch.rows.clear() + batch.rollback_counter = counter + return batch + + +@contextmanager +def capture_label_inputs(store: Any, source_id: str) -> Iterator[None]: + """Reuse only the new source's label row during capture's candidate loop. + + A managed transaction fixes the actual transaction id for this synchronous + loop. Savepoint rollback and label updates invalidate its input rows. Every + writer still executes its advisory lock, and no grant is memoized. + """ + + conn = getattr(store, "conn", None) + if conn is None or _sqlite(store) or not callable(getattr(conn, "transaction", None)): + yield + return + with conn.transaction(): + with conn.cursor() as cur: + cur.execute("SELECT pg_current_xact_id()::text AS transaction_id") + row = cur.fetchone() + transaction_id = str(row["transaction_id"] if isinstance(row, Mapping) else row[0]) + batch = _CaptureLabelInputs(conn, transaction_id, int(getattr(conn, "_alice_label_rollback_counter", 0)), identifier(source_id)) + token = _CAPTURE_LABEL_INPUTS.set(batch) + try: + yield + finally: + _CAPTURE_LABEL_INPUTS.reset(token) + + def takes_label_lock(fn: Any) -> Any: """Take the shared label lock before a store method writes or row-locks a label table.""" @@ -177,6 +245,7 @@ def prepare_label_patch( ) -> JsonObject: """Check a proposed label change before its UPDATE or FOR UPDATE statement.""" + invalidate_capture_label_inputs(store) proposed = dict(before or {}) proposed.update({key: value for key, value in patch.items() if value is not None}) proposed["kind"] = kind @@ -220,7 +289,14 @@ def apply_insert_floor(store: Any, kind: str, payload: Mapping[str, object]) -> own["kind"] = kind own["id"] = own_id own["user_id"] = user_id - nodes, exceeded = collect_label_rows(store, [own], max_nodes=PROPAGATION_BOUND) + batch = _current_capture_inputs(store) + cache = batch.rows if batch is not None else None + nodes, exceeded = collect_label_rows(store, [own], max_nodes=PROPAGATION_BOUND, cache=cache) + if batch is not None: + # Keep only this capture's source. Other dependencies are read afresh. + for key in list(batch.rows): + if key != ("source", batch.source_id): + del batch.rows[key] if exceeded: raise LabelPropagationTooLarge(f"label propagation stopped after {PROPAGATION_BOUND} rows") if not user_id: @@ -601,6 +677,7 @@ def write_settled_label( ) -> None: """Label-only update. A statement that changes no row refuses the whole relabel.""" + invalidate_capture_label_inputs(store) table = {"memory": "memories", "open_loop": "open_loops", "artifact": "generated_artifacts", "project": "projects"}[kind] blob = json.dumps({key: metadata[key] for key in ("project_scope", "project_floor") if key in metadata}) if _sqlite(store): diff --git a/apps/api/src/alicebot_api/vnext_store.py b/apps/api/src/alicebot_api/vnext_store.py index 2a6a4e209..c34354a85 100644 --- a/apps/api/src/alicebot_api/vnext_store.py +++ b/apps/api/src/alicebot_api/vnext_store.py @@ -3802,6 +3802,9 @@ def savepoint(self) -> Iterator[None]: try: yield except BaseException: + from alicebot_api.vnext_label_writes import label_savepoint_rolled_back + + label_savepoint_rolled_back(self) try: self.conn.execute(f"ROLLBACK TO SAVEPOINT {name}") self.conn.execute(f"RELEASE SAVEPOINT {name}") diff --git a/tests/integration/test_capture_label_batch_postgres.py b/tests/integration/test_capture_label_batch_postgres.py new file mode 100644 index 000000000..e7a6d8549 --- /dev/null +++ b/tests/integration/test_capture_label_batch_postgres.py @@ -0,0 +1,159 @@ +"""Capture reuses source inputs while every writer takes a live label lock.""" +from uuid import uuid4 + +import psycopg +import pytest +from psycopg.rows import dict_row + +from alicebot_api import vnext_label_writes as writes +from alicebot_api.db import set_current_user, user_connection +from alicebot_api.store import ContinuityStore +from alicebot_api.vnext_capture import VNextCaptureService +from alicebot_api.vnext_store import PostgresVNextStore + + +def _user(conn, user): + ContinuityStore(conn).create_user(user, f"batch-{user}@example.test", "Synthetic") + + +def _source(store, user): + return store.create_source({"source_type": "note", "title": "Synthetic", "content_hash": str(user), "domain": "project", "sensitivity": "public"}) + + +def _memory(store, key, source): + return store.create_memory({"memory_key": key, "canonical_text": "Synthetic", "domain": "project", "sensitivity": "public", "metadata_json": {"source_id": str(source["id"])}}) + + +def test_capture_reuses_only_source_inputs_and_keeps_every_writer_lock(migrated_database_urls, monkeypatch): + monkeypatch.setattr(writes, "STRICT_LOCK_ORDER", False) + user = uuid4() + with user_connection(migrated_database_urls["app"], user) as conn: + _user(conn, user) + store = PostgresVNextStore(conn) + calls, reads, identities = [], [], [] + real_lock, real_reader, real_memory = store.lock_label_writes, store.read_label_rows, store.create_memory + def lock(*, exclusive=False): + calls.append(exclusive) + return real_lock(exclusive=exclusive) + def reader(kind, ids): + reads.append((kind, ids)) + return real_reader(kind, ids) + def memory(payload, **kwargs): + identities.append(conn.execute("SELECT pg_current_xact_id()::text AS tx, app.current_user_id()::text AS tenant").fetchone()) + frame = writes._current_capture_inputs(store) + assert frame.key[0] == identities[-1]["tx"] + return real_memory(payload, **kwargs) + monkeypatch.setattr(store, "lock_label_writes", lock) + monkeypatch.setattr(store, "read_label_rows", reader) + monkeypatch.setattr(store, "create_memory", memory) + result = VNextCaptureService(store).capture_text("\n".join(f"Decision: Synthetic item {i} is assigned to synthetic route {i}." for i in range(200)), domain="project", sensitivity="public") + assert result.candidate_memory_count == 200 + assert len(calls) == 401 # Source, 200 memories, 200 provenance links. + assert len(reads) == 1 and reads[0][0] == "source" + assert len({(row["tx"], row["tenant"]) for row in identities}) == 1 + assert identities[0]["tenant"] == str(user) + assert writes._current_capture_inputs(store) is None + assert writes.held_label_locks(store)[1] is True + assert conn.execute("SELECT count(*) AS n FROM memories").fetchone()["n"] == 200 + + +def test_source_input_memo_is_keyed_by_transaction_and_rollback_counter(migrated_database_urls): + user = uuid4() + with user_connection(migrated_database_urls["app"], user) as conn: + _user(conn, user) + store = PostgresVNextStore(conn) + store.lock_graph_mutation() + store.lock_label_writes(exclusive=True) + source = _source(store, user) + with writes.capture_label_inputs(store, str(source["id"])): + _memory(store, "before", source) + frame = writes._current_capture_inputs(store) + before = frame.key + assert frame.rows + with pytest.raises(ValueError): + with store.savepoint(): + store.update_source(source_id=str(source["id"]), patch={"sensitivity": "confidential"}) + assert _memory(store, "rolled-back", source)["sensitivity"] == "confidential" + raise ValueError("Synthetic rollback") + assert _memory(store, "after", source)["sensitivity"] == "public" + after = writes._current_capture_inputs(store).key + assert after == (before[0], before[1] + 1) + assert writes.held_label_locks(store)[1] is True + assert {row["memory_key"] for row in conn.execute("SELECT memory_key FROM memories").fetchall()} == {"before", "after"} + + +def test_failed_native_savepoint_batch_leaves_no_memo_or_rows(migrated_database_urls): + user = uuid4() + with user_connection(migrated_database_urls["app"], user) as conn: + _user(conn, user) + store = PostgresVNextStore(conn) + with pytest.raises(ValueError): + with conn.transaction(): + source = _source(store, user) + with writes.capture_label_inputs(store, str(source["id"])): + _memory(store, "failed", source) + assert writes.held_label_locks(store)[1] is True + raise ValueError("Synthetic failure") + assert writes._current_capture_inputs(store) is None + assert writes.held_label_locks(store)[1] is False + assert conn.execute("SELECT count(*) AS n FROM memories").fetchone()["n"] == 0 + source = _source(store, user) + _memory(store, "fresh", source) + assert writes.held_label_locks(store)[1] is True + + +def test_source_raise_invalidates_inputs_in_the_same_transaction(migrated_database_urls, monkeypatch): + monkeypatch.setattr(writes, "STRICT_LOCK_ORDER", False) + user = uuid4() + with user_connection(migrated_database_urls["app"], user) as conn: + _user(conn, user) + store = PostgresVNextStore(conn) + store.lock_graph_mutation() + store.lock_label_writes(exclusive=True) + source = _source(store, user) + with writes.capture_label_inputs(store, str(source["id"])): + assert _memory(store, "before", source)["sensitivity"] == "public" + store.update_source(source_id=str(source["id"]), patch={"sensitivity": "confidential"}) + assert writes._current_capture_inputs(store).rows == {} + assert _memory(store, "after", source)["sensitivity"] == "confidential" + + +def test_strict_mode_still_locks_every_writer_with_source_input_reuse(migrated_database_urls, monkeypatch): + user = uuid4() + with user_connection(migrated_database_urls["app"], user) as conn: + _user(conn, user) + store = PostgresVNextStore(conn) + source = _source(store, user) + count = 0 + real_lock = store.lock_label_writes + def lock(*, exclusive=False): + nonlocal count + count += 1 + return real_lock(exclusive=exclusive) + monkeypatch.setattr(store, "lock_label_writes", lock) + assert writes.STRICT_LOCK_ORDER is True + with writes.capture_label_inputs(store, str(source["id"])): + _memory(store, "first", source) + _memory(store, "second", source) + assert writes.held_label_locks(store)[1] is True + assert count == 2 + + +def test_one_store_starts_fresh_source_inputs_in_each_transaction(migrated_database_urls): + user = uuid4() + keys = [] + with psycopg.connect(migrated_database_urls["app"], row_factory=dict_row) as conn: + store = PostgresVNextStore(conn) + for index in range(2): + set_current_user(conn, user) + if index == 0: + _user(conn, user) + source = _source(store, uuid4()) + with writes.capture_label_inputs(store, str(source["id"])): + keys.append(writes._current_capture_inputs(store).key) + with pytest.raises(psycopg.ProgrammingError): + conn.commit() + _memory(store, str(index), source) + assert writes._current_capture_inputs(store) is None + conn.commit() + assert keys[0][0] != keys[1][0] From 97576fcf5c94191ab4192c2cddda65f1dc7f1a2c Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Tue, 6 Oct 2026 00:48:26 +0200 Subject: [PATCH 67/90] Bound large PostgreSQL reverse frontier query cost --- .../src/alicebot_api/vnext_label_writes.py | 12 +++-- .../test_label_reverse_search_postgres.py | 50 ++++++++++++++++++- 2 files changed, 57 insertions(+), 5 deletions(-) diff --git a/apps/api/src/alicebot_api/vnext_label_writes.py b/apps/api/src/alicebot_api/vnext_label_writes.py index 28c54c159..0efd18e34 100644 --- a/apps/api/src/alicebot_api/vnext_label_writes.py +++ b/apps/api/src/alicebot_api/vnext_label_writes.py @@ -482,7 +482,12 @@ def _postgres_dependants(store: Any, table: str, kind: str, compacts: Sequence[s extra = ", NULL::jsonb AS value" # Normalize each column once per row, rather than once per frontier id. # This remains only a superset lookup; the canonical parser selects exact edges. - pattern = "(?:" + "|".join(re.escape(item) for item in compacts) + ")" + if len(compacts) >= 32 and all(re.fullmatch(r"[0-9a-f]{32}", item) for item in compacts): + # A constant pattern avoids constructing a large automaton for UUID + # frontiers. More candidates are safe because exact edges are parsed below. + pattern = r"[0-9a-f]{32}" + else: + pattern = "(?:" + "|".join(re.escape(item) for item in compacts) + ")" def candidate_clause(column: str) -> str: return ( @@ -538,8 +543,9 @@ def walk_dependants(store: Any, roots: Sequence[str]) -> list[dict[str, object]] while pending: if len(seen) > PROPAGATION_BOUND: raise LabelPropagationTooLarge(f"label propagation stopped after {PROPAGATION_BOUND} rows") - batch = pending[:200] - pending = pending[200:] + batch_size = 200 if _sqlite(store) else 2000 + batch = pending[:batch_size] + pending = pending[batch_size:] belief_aliases = getattr(store, "list_belief_ids_for_memories", None) if callable(belief_aliases): for belief_id in belief_aliases(batch): diff --git a/tests/integration/test_label_reverse_search_postgres.py b/tests/integration/test_label_reverse_search_postgres.py index f68ebd541..ef62b33d3 100644 --- a/tests/integration/test_label_reverse_search_postgres.py +++ b/tests/integration/test_label_reverse_search_postgres.py @@ -80,7 +80,7 @@ def test_multiple_frontier_batches_keep_memory_chains_and_both_direct_loop_colum "sensitivity": "public", } ) - roots = [str(uuid4()) for _ in range(220)] + roots = [str(uuid4()) for _ in range(2200)] conn.execute( """INSERT INTO memories(id,user_id,memory_key,value,source_event_ids,canonical_text, status,domain,sensitivity,metadata_json) @@ -90,7 +90,7 @@ def test_multiple_frontier_batches_keep_memory_chains_and_both_direct_loop_colum (root, roots), ) descendants = [] - for index in (0, 199, 219): + for index in (0, 1999, 2199): row = store.create_memory( { "memory_key": f"descendant.{index}", @@ -126,3 +126,49 @@ def test_multiple_frontier_batches_keep_memory_chains_and_both_direct_loop_colum store, "open_loops", "open_loop", [_compact_id(roots[-1])], with_value=False ) assert str(direct_memory["id"]) in {row["id"] for row in memory_candidates} + + +def test_large_uuid_frontiers_use_a_broader_superset_but_only_canonical_edges_enter_the_closure(migrated_database_urls): + user = uuid4() + roots = [str(uuid4()) for _ in range(32)] + with user_connection(migrated_database_urls["app"], user) as conn: + ContinuityStore(conn).create_user(user, f"broad-{user}@example.invalid", "Broad") + store = PostgresVNextStore(conn) + expected = set() + for index, reference in enumerate((roots[0], roots[-1].upper(), encoded_object("source_id", roots[1]))): + row = store.create_memory( + { + "memory_key": f"broad.{index}", + "canonical_text": "Synthetic dependency", + "domain": "project", + "sensitivity": "public", + "metadata_json": {"source_id": reference}, + } + ) + expected.add(str(row["id"])) + unrelated = store.create_memory( + { + "memory_key": "unrelated.edge", + "canonical_text": "Synthetic unrelated", + "domain": "project", + "sensitivity": "public", + "metadata_json": {"source_id": str(uuid4())}, + } + ) + note = store.create_memory( + { + "memory_key": "unrelated.note", + "canonical_text": "Synthetic unrelated note", + "domain": "project", + "sensitivity": "public", + "metadata_json": {"note": roots[0]}, + } + ) + candidates = _postgres_dependants( + store, "memories", "memory", [_compact_id(root) for root in roots], with_value=True + ) + candidate_ids = {str(row["id"]) for row in candidates} + assert expected <= candidate_ids + assert str(unrelated["id"]) in candidate_ids + assert str(note["id"]) in candidate_ids + assert {str(row["id"]) for row in walk_dependants(store, roots)} == expected From 12a851dfb9c30d99b64963593d1abd9150595fe9 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Tue, 6 Oct 2026 00:48:14 +0200 Subject: [PATCH 68/90] Prove native PostgreSQL owner edit clamp controls --- .../test_label_floor_ancestry_postgres.py | 35 +++++++++++++++++++ 1 file changed, 35 insertions(+) diff --git a/tests/integration/test_label_floor_ancestry_postgres.py b/tests/integration/test_label_floor_ancestry_postgres.py index 2a16690f1..882999373 100644 --- a/tests/integration/test_label_floor_ancestry_postgres.py +++ b/tests/integration/test_label_floor_ancestry_postgres.py @@ -95,3 +95,38 @@ def test_a_relabel_traverses_the_belief_backing_memory(migrated_database_urls): store = PostgresVNextStore(conn) assert store.get_memory(str(copy["id"]))["sensitivity"] == "regulated" assert store.get_artifact(str(report["id"]))["sensitivity"] == "regulated" + + +def test_postgres_owner_edit_is_clamped_in_response_event_and_storage(migrated_database_urls, monkeypatch): + import json + from uuid import UUID + from alicebot_api.config import Settings + from alicebot_api.routers import vnext_memories as router + from alicebot_api.vnext_label_writes import without_insert_floor + + user_id = uuid4() + url = migrated_database_urls["app"] + with user_connection(url, user_id) as conn: + ContinuityStore(conn).create_user(user_id, f"owner-clamp-{user_id}@example.test", "Synthetic") + store = PostgresVNextStore(conn) + source = store.create_source({"source_type": "note", "title": "Synthetic private input", "content_hash": str(user_id), "domain": "health", "sensitivity": "confidential"}) + with without_insert_floor(): + memory = store.create_memory({"memory_key": "stale-copy", "canonical_text": "Synthetic private observation", "status": "active", "domain": "project", "sensitivity": "public", "metadata_json": {"source_id": str(source["id"])}}) + assert memory["sensitivity"] == "public" + monkeypatch.setattr(router, "get_settings", lambda: Settings(database_url=url)) + response = router.review_vnext_memory(UUID(str(memory["id"])), router.VNextMemoryReviewRequest(user_id=user_id, action="edit", domain="project", sensitivity="public"), authorization=None) + assert response.status_code == 200 + payload = json.loads(response.body) + assert payload["label_floor_applied"] is True + assert payload["memory"]["domain"] == "health" + assert payload["memory"]["sensitivity"] == "confidential" + with user_connection(url, user_id) as conn: + store = PostgresVNextStore(conn) + stored = store.get_memory(str(memory["id"])) + assert stored["domain"] == "health" and stored["sensitivity"] == "confidential" + assert stored["metadata_json"]["source_id"] == str(source["id"]) + event = conn.execute("SELECT payload_json FROM event_log WHERE event_type='memory.labels_raised' AND target_id=%s", (str(memory["id"]),)).fetchone() + assert event["payload_json"]["cause"] == "floor_clamped" + encoded = json.dumps(event["payload_json"]) + assert "Synthetic private input" not in encoded + assert "Synthetic private observation" not in encoded From e5bd6fc4b08c5c573f511aec78dc92754149a737 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Tue, 6 Oct 2026 00:58:09 +0200 Subject: [PATCH 69/90] Align write-stage unit fixtures and route inventories --- tests/unit/test_legacy_gated_router_split.py | 14 +++-- tests/unit/test_main.py | 6 +- tests/unit/test_mcp.py | 47 +++++++++++++++ tests/unit/test_providers_router_split.py | 5 +- tests/unit/test_saved_provenance_reader.py | 1 + tests/unit/test_source_refs_read_fence.py | 3 + tests/unit/test_surface_gates.py | 6 +- tests/unit/test_vnext_main.py | 34 ++++++++++- tests/unit/test_vnext_store.py | 63 +++++++++++++------- tests/unit/test_workspaces_router_split.py | 6 +- 10 files changed, 146 insertions(+), 39 deletions(-) diff --git a/tests/unit/test_legacy_gated_router_split.py b/tests/unit/test_legacy_gated_router_split.py index a19305c3d..4def45c37 100644 --- a/tests/unit/test_legacy_gated_router_split.py +++ b/tests/unit/test_legacy_gated_router_split.py @@ -240,8 +240,10 @@ EXPECTED_OWNED_SUPPORT_AST_SHA256 = "8bbbcb0cf52c4bd1da5fce4e50878d8a62599be59ee25e09dcd905c94950d406" EXPECTED_FULL_SUPPORT_AST_SHA256 = "07acb5deafb700e040c459969610e8877e86e62b04d4e9d86493fe314cb02868" EXPECTED_GATED_OPERATION_SHA256 = "55490460fd78990a6872ebf22c6cfd927f7d0a5548b6df90adde8261ede8da65" -EXPECTED_DEFAULT_DEEP_ROUTE_SHA256 = "a865b2264c911ed1b055853777b850e070bd594d79fafb167df7b915b7c3c9ce" -EXPECTED_LEGACY_DEEP_ROUTE_SHA256 = "6743a8b03d649328341eab462aeea0271f997f14c5c558a666b5aff92f297d79" + +# Source regeneration adds one default route before source review. +EXPECTED_DEFAULT_DEEP_ROUTE_SHA256 = "9d9e83c35818387a13c719d455c88f4dcb4d139bf5f4dd5888bc6ff3723077a7" +EXPECTED_LEGACY_DEEP_ROUTE_SHA256 = "5ba780b5ec62584b844cf1fcab8fc35b703f4494c1486db870588ab68a1f57a6" EXPECTED_INTEGRATION_PATCH_COUNTS = { "tests/integration/test_approval_api.py": 8, @@ -667,9 +669,9 @@ def test_main_preserves_frozen_flag_policy_and_five_mount_seams() -> None: def test_flagged_surface_preserves_deep_order_ids_and_import_timing() -> None: default = _isolated_surface_manifest(None) assert default == { - "operation_count": 183, + "operation_count": 184, "legacy_count": 0, - "deep_count": 187, + "deep_count": 188, "deep_digest": EXPECTED_DEFAULT_DEEP_ROUTE_SHA256, "gated_count": 0, "gated_digest": hashlib.sha256(b"[]").hexdigest(), @@ -679,9 +681,9 @@ def test_flagged_surface_preserves_deep_order_ids_and_import_timing() -> None: } legacy = _isolated_surface_manifest("1") assert legacy == { - "operation_count": 232, + "operation_count": 233, "legacy_count": 49, - "deep_count": 236, + "deep_count": 237, "deep_digest": EXPECTED_LEGACY_DEEP_ROUTE_SHA256, "gated_count": 49, "gated_digest": EXPECTED_GATED_OPERATION_SHA256, diff --git a/tests/unit/test_main.py b/tests/unit/test_main.py index 43c657341..563513a30 100644 --- a/tests/unit/test_main.py +++ b/tests/unit/test_main.py @@ -312,7 +312,7 @@ def test_openapi_has_concrete_success_contracts_and_accurate_statuses() -> None: } operations = list(operations_by_key.values()) - assert len(operations) == 183 + assert len(operations) == 184 assert all(operation.get("tags") for operation in operations) assert all(operation.get("description") for operation in operations) assert all("default" in operation["responses"] for operation in operations) @@ -323,7 +323,7 @@ def test_openapi_has_concrete_success_contracts_and_accurate_statuses() -> None: if status.startswith("2") for json_body in [response.get("content", {}).get("application/json", {})] ] - assert len(success_schemas) == 187 + assert len(success_schemas) == 188 assert "APIJsonDocument" not in components assert all(document.get("$ref", "").startswith("#/components/schemas/") for document in success_schemas) resolved_success_schemas = [components[document["$ref"].rsplit("/", 1)[-1]] for document in success_schemas] @@ -347,7 +347,7 @@ def test_openapi_has_concrete_success_contracts_and_accurate_statuses() -> None: assert exact_keys | set(operation_registry) == set(operations_by_key), coverage_report assert exact_keys.isdisjoint(operation_registry), coverage_report assert len(exact_keys) == 42 - assert len(operation_registry) == 141 + assert len(operation_registry) == 142 assert 0 < len(polymorphic_operations) <= 3 assert set(polymorphic_operations) <= set(operation_registry) assert all(reason.strip() for reason in polymorphic_operations.values()) diff --git a/tests/unit/test_mcp.py b/tests/unit/test_mcp.py index 8561365d0..e9dd310e4 100644 --- a/tests/unit/test_mcp.py +++ b/tests/unit/test_mcp.py @@ -1277,8 +1277,48 @@ def _ascii_query_fold(value: str) -> str: return value.translate(_ASCII_QUERY_CASE_TRANSLATION) +class FakeLabelCursor: + """SQL guard responses from the fake store's current lock state.""" + + def __init__(self, store) -> None: + self.store = store + self.query = "" + + def __enter__(self): + return self + + def __exit__(self, *args): + return None + + def execute(self, query, params=None): + assert any(marker in query for marker in ( + "FROM pg_locks", "current_setting('lock_timeout')", "SET LOCAL lock_timeout", "set_config('lock_timeout'", + )), query + self.query = query + + def fetchone(self): + if "FROM pg_locks" in self.query: + return {"graph": self.store.graph_locked, "labels": self.store.labels_locked, + "exclusive": self.store.labels_exclusive} + if "current_setting('lock_timeout')" in self.query: + return {"lock_timeout": "0"} + raise AssertionError(self.query) + + +class FakeLabelConnection: + def __init__(self, store) -> None: + self.store = store + + def cursor(self): + return FakeLabelCursor(self.store) + + class FakeVNextMCPStore: def __init__(self) -> None: + self.graph_locked = False + self.labels_locked = False + self.labels_exclusive = False + self.conn = FakeLabelConnection(self) self.events: list[dict[str, object]] = [] self.sources: list[dict[str, object]] = [] self.chunks: list[dict[str, object]] = [] @@ -1317,6 +1357,13 @@ def __init__(self) -> None: } } + def lock_graph_mutation(self) -> None: + self.graph_locked = True + + def lock_label_writes(self, *, exclusive: bool = False) -> None: + self.labels_locked = True + self.labels_exclusive |= exclusive + @staticmethod def _is_live(row: dict[str, object]) -> bool: return row.get("deleted_at") is None diff --git a/tests/unit/test_providers_router_split.py b/tests/unit/test_providers_router_split.py index a8b88d978..04a176b8d 100644 --- a/tests/unit/test_providers_router_split.py +++ b/tests/unit/test_providers_router_split.py @@ -117,7 +117,10 @@ # lone_surrogates.py and main.py only registers it, so it adds no definition # here. Earlier re-pin (2026-09-26): _rewrite_user_id_json_body writes the # rewritten JSON into request._body before call_next. -EXPECTED_CARRIER_AST_SHA256 = "ab3fc6d61cb81a1b9c1a6573adc8e1e297cbbcf01e230effd4a0824dee2d8e2b" + +# Re-pin 2026-10-06: source regeneration is a new protected write route in the +# central vNext route policy; the app carrier keeps the same definitions. +EXPECTED_CARRIER_AST_SHA256 = "fcd6d722e2d6c3be28b138449555e703f6930168f52b8e94b41dacbcb937b605" EXPECTED_ROUTE_NAME_MANIFEST_SHA256 = "1a438538e16120361f92d30375cc94679d598fe4b78ba5a58a7d8a4dda6af83c" EXPECTED_OPERATION_MANIFEST_SHA256 = "8b79ceaf996b8c51b5bb2f3f38a8c19a4e33796955d8b8f7a66e7ac01ea1732d" EXPECTED_IMPORT_MANIFEST_SHA256 = "17484ccdd460e42e2ad5c82a8ca867664694feaf871118c410a134996a532358" diff --git a/tests/unit/test_saved_provenance_reader.py b/tests/unit/test_saved_provenance_reader.py index 446ca83b1..5fb12c853 100644 --- a/tests/unit/test_saved_provenance_reader.py +++ b/tests/unit/test_saved_provenance_reader.py @@ -730,6 +730,7 @@ def test_review_by_id_reads_saved_provenance_only_through_the_reader() -> None: ("apps/api/src/alicebot_api/mcp/memories.py", "_handle_alice_vnext_commit_memory"): "passes the argument to commit", ("apps/api/src/alicebot_api/cli/memories.py", "_run_vnext_memory_commit"): "passes the argument to commit", ("apps/api/src/alicebot_api/vnext_capture.py", "_capture_source"): "link quote = the candidate's own text", + ("apps/api/src/alicebot_api/vnext_source_regeneration.py", "regenerate_source_inputs"): "fresh candidate quote; SavedProvenanceReader and row readers apply the source fence", ("apps/api/src/alicebot_api/vnext_connectors.py", "ingest_agent_output"): "link quote = the item title", ("apps/api/src/alicebot_api/vnext_retrieval.py", "_supporting_evidence"): "reads the link, fenced by admits_link", ("apps/api/src/alicebot_api/onramp.py", "_apply_import_quarantine"): "replaces a quote with a placeholder", diff --git a/tests/unit/test_source_refs_read_fence.py b/tests/unit/test_source_refs_read_fence.py index 7ffe35cdc..11815f8d2 100644 --- a/tests/unit/test_source_refs_read_fence.py +++ b/tests/unit/test_source_refs_read_fence.py @@ -1509,6 +1509,8 @@ def _provenance_link_call_sites() -> set[tuple[str, str]]: # The source was captured or ingested by this very call, so the caller named no id. ("vnext_capture.py", "_capture_source"), ("vnext_connectors.py", "ingest_agent_output"), + # Regeneration reads the authorized source and its chunks before making new rows. + ("vnext_source_regeneration.py", "regenerate_source_inputs"), } _NAMED_SOURCE_SITES = { # The caller named the id. Each is behind ``resolve_attachable_sources`` and the caller's ``SourceReadFence``. @@ -1702,6 +1704,7 @@ def test_attachable_sources_are_built_only_by_the_resolver() -> None: ("memory.py", "_create_open_loop_for_memory"), ("vnext_brain.py", "_create_candidate_open_loops"), ("vnext_projects.py", "extract_open_loops"), + ("vnext_source_regeneration.py", "regenerate_source_inputs"), ("vnext_scheduler.py", "_publish_mutation"), ("vnext_stores/postgres/graph_open_loops.py", "upsert_open_loop_by_automation_digest"), ("vnext_stores/sqlite/graph_open_loops.py", "upsert_open_loop_by_automation_digest"), diff --git a/tests/unit/test_surface_gates.py b/tests/unit/test_surface_gates.py index e23a12886..92eab0b86 100644 --- a/tests/unit/test_surface_gates.py +++ b/tests/unit/test_surface_gates.py @@ -94,7 +94,7 @@ def _isolated_proxy_execution_posture(flag_value: str | None) -> dict[str, objec @pytest.mark.parametrize("flag_value", [None, "", "0", "true", "yes", "on", "01", " 1"]) def test_http_legacy_surface_gate_fails_closed_for_every_non_exact_value(flag_value: str | None) -> None: assert _isolated_http_inventory(flag_value) == { - "count": 183, + "count": 184, "legacy_count": 0, "removed_count": 0, "runtime_invoke_count": 1, @@ -103,7 +103,7 @@ def test_http_legacy_surface_gate_fails_closed_for_every_non_exact_value(flag_va def test_http_legacy_surface_gate_mounts_exact_inventory_only_for_one() -> None: assert _isolated_http_inventory("1") == { - "count": 232, + "count": 233, "legacy_count": 49, "removed_count": 0, "runtime_invoke_count": 1, @@ -161,7 +161,7 @@ def inventory(): text=True, ) - expected_inventory = {"count": 183, "legacy_count": 0} + expected_inventory = {"count": 184, "legacy_count": 0} assert json.loads(completed.stdout) == { "before": expected_inventory, "after": expected_inventory, diff --git a/tests/unit/test_vnext_main.py b/tests/unit/test_vnext_main.py index 8af8173c1..90335d3dd 100644 --- a/tests/unit/test_vnext_main.py +++ b/tests/unit/test_vnext_main.py @@ -54,6 +54,29 @@ def __init__(self, _conn) -> None: self.agent_api_keys: list[dict[str, object]] = [] self.browser_clip_capabilities: dict[str, dict[str, object]] = {} self.revisions: list[dict[str, object]] = [] + self.graph_locked = False + self.labels_locked = False + self.labels_exclusive = False + + def lock_graph_mutation(self) -> None: + self.graph_locked = True + + def lock_label_writes(self, *, exclusive: bool = False) -> None: + self.labels_locked = True + self.labels_exclusive |= exclusive + + def read_label_rows(self, kind: str, ids: list[str]) -> list[dict[str, object]]: + collection = {"source": self.sources.values(), "memory": self.memories, "open_loop": self.open_loops, + "artifact": self.artifacts.values(), "belief": self.beliefs.values(), "project": self.projects.values()}.get(kind, []) + return [dict(row) for row in collection if str(row.get("id")) in ids] + + def _fetch_all(self, query: str, _params: tuple[object, ...]) -> list[dict[str, object]]: + # The label dependant walker performs its exact canonical reference filter after this prefilter. + for table, kind in (("memories", "memory"), ("open_loops", "open_loop"), ("generated_artifacts", "artifact"), ("projects", "project")): + if f"FROM {table}" in query: + collection = {"memory": self.memories, "open_loop": self.open_loops, "artifact": self.artifacts.values(), "project": self.projects.values()}[kind] + return [{**row, "kind": kind} for row in collection] + raise AssertionError(query) def create_browser_clip_capability( self, @@ -879,6 +902,12 @@ def list_connector_states(self) -> list[dict[str, object]]: def _install_fake_vnext_store(monkeypatch, store: FakeVNextStore) -> None: + from alicebot_api import vnext_label_writes + monkeypatch.setattr(vnext_label_writes, "acquire_exclusive_label_lock", lambda target: target.lock_label_writes(exclusive=True)) + monkeypatch.setattr( + vnext_label_writes, "held_label_locks", + lambda target: (target.graph_locked, target.labels_locked, target.labels_exclusive), + ) @contextmanager def fake_user_connection(database_url, current_user_id): assert database_url == "postgresql://db" @@ -1198,7 +1227,7 @@ def test_vnext_route_inventory_fails_closed_without_route_local_policy() -> None } assert not (main_module._VNEXT_ROUTE_LOCAL_POLICY & main_module._VNEXT_CENTRAL_OPERATOR_ROUTES) assert (main_module._VNEXT_ROUTE_LOCAL_POLICY | main_module._VNEXT_CENTRAL_OPERATOR_ROUTES) == registered - assert len(registered) == 71 + assert len(registered) == 72 project_bound = main_module.AgentIdentity( agent_id="project-reader", @@ -1347,6 +1376,7 @@ def test_vnext_memories_router_partitions_preserve_global_route_sequence() -> No vnext_memories_router.source_review_router, [ ("GET", "/v0/vnext/sources/{source_id}"), + ("POST", "/v0/vnext/sources/{source_id}/regenerate"), ("POST", "/v0/vnext/sources/{source_id}/review"), ], ), @@ -1912,7 +1942,7 @@ def test_create_vnext_source_threads_project_scope_into_captured_memory(monkeypa candidates = store.list_memories(status="candidate") assert candidates, "capture must promote at least one candidate memory" - assert memory_project_scope(candidates[0]) == ("Project-Helios", "project-helios") + assert memory_project_scope(candidates[0]) == ("Project-Helios",) for memory in candidates: store.update_memory(memory_id=str(memory["id"]), patch={"status": "active"}, actor_type="system") diff --git a/tests/unit/test_vnext_store.py b/tests/unit/test_vnext_store.py index de7109f4d..ec0c424b9 100644 --- a/tests/unit/test_vnext_store.py +++ b/tests/unit/test_vnext_store.py @@ -27,6 +27,10 @@ def __init__( self.executed: list[tuple[str, tuple[object, ...] | None]] = [] self.fetchone_results = list(fetchone_results) self.fetchall_result = fetchall_result or [] + self.current_query = "" + self.graph_locked = False + self.labels_locked = False + self.labels_exclusive = False def __enter__(self) -> "RecordingCursor": return self @@ -38,13 +42,28 @@ def execute(self, query: str, params: tuple[object, ...] | None = None) -> None: if params is not None: assert query.count("%s") == len(params) self.executed.append((query, params)) + self.current_query = query + if "pg_advisory_xact_lock" in query: + if "vnext_supersession" in query: + self.graph_locked = True + elif "vnext_labels" in query: + self.labels_locked = True + self.labels_exclusive |= "pg_advisory_xact_lock_shared" not in query @property def statements(self) -> list[tuple[str, tuple[object, ...] | None]]: """Business SQL, with transaction guards retained separately in executed.""" - return [(query, params) for query, params in self.executed if "pg_advisory_xact_lock" not in query] + return [(query, params) for query, params in self.executed if not any(marker in query for marker in ( + "pg_advisory_xact_lock", "FROM pg_locks", "current_setting('lock_timeout')", + "SET LOCAL lock_timeout", "set_config('lock_timeout'", + ))] def fetchone(self) -> dict[str, Any] | None: + if "current_setting('lock_timeout')" in self.current_query: + return {"lock_timeout": "0"} + if "FROM pg_locks" in self.current_query: + return {"graph": self.graph_locked, "labels": self.labels_locked, + "exclusive": self.labels_exclusive} if not self.fetchone_results: return None return self.fetchone_results.pop(0) @@ -96,15 +115,12 @@ def test_source_crud_and_chunks_write_audit_events() -> None: {"id": source_id}, _event_row(source_id), {"id": source_id}, - { - "id": source_id, - "content_hash": "sha256:abc", - "dedupe_key": "capture-md5:legacy", - "domain": "project", - "sensitivity": "private", - "metadata_json": {"path": "docs/spec.md"}, - }, - {"id": source_id}, + # The unlocked label pre-read precedes the source row lock. + {"id": source_id, "content_hash": "sha256:abc", "dedupe_key": "capture-md5:legacy", + "domain": "project", "sensitivity": "private", "metadata_json": {"path": "docs/spec.md"}}, + {"id": source_id, "content_hash": "sha256:abc", "dedupe_key": "capture-md5:legacy", + "domain": "project", "sensitivity": "private", "metadata_json": {"path": "docs/spec.md"}}, + {"id": source_id, "domain": "project", "sensitivity": "private", "metadata_json": {"rev": 2}}, _event_row(source_id), {"id": source_id}, _event_row(source_id), @@ -293,7 +309,8 @@ def test_update_source_recomputes_postgres_dedupe_key_with_the_same_statement() } cursor = RecordingCursor( fetchone_results=[ - current, + current, # label pre-read + current, # locked capture identity None, {**current, "domain": "professional"}, _event_row(source_id), @@ -339,7 +356,7 @@ def test_update_source_postgres_collision_fails_before_mutation_event() -> None: "sensitivity": "private", "metadata_json": {"raw_text": raw_text, "project_scope": ["Alpha"]}, } - cursor = RecordingCursor(fetchone_results=[current, {"id": str(uuid4())}]) + cursor = RecordingCursor(fetchone_results=[current, current, {"id": str(uuid4())}]) store = PostgresVNextStore(RecordingConnection(cursor)) with pytest.raises(ContinuityStoreInvariantError, match="already belongs"): @@ -364,7 +381,8 @@ def test_update_source_postgres_releases_key_when_changed_identity_has_no_raw_te } cursor = RecordingCursor( fetchone_results=[ - current, + current, # label pre-read + current, # locked capture identity {**current, "dedupe_key": None, "metadata_json": {"project_scope": ["Beta"]}}, _event_row(source_id), ] @@ -721,7 +739,6 @@ def test_list_artifacts_applies_type_domain_sensitivity_and_limit_filters() -> N ["public", "private"], None, None, - None, 5, ) @@ -1409,6 +1426,7 @@ def test_project_people_belief_and_open_loop_methods_write_audit_events() -> Non {"id": project_id}, _event_row(project_id), {"id": project_id}, + {"id": project_id}, # label pre-read before the update {"id": project_id}, _event_row(project_id), {"id": person_id}, @@ -1426,6 +1444,7 @@ def test_project_people_belief_and_open_loop_methods_write_audit_events() -> Non {"id": loop_id}, {"id": loop_id}, _event_row(loop_id), + {"id": loop_id}, # label pre-read before the final update {"id": loop_id}, _event_row(loop_id), ] @@ -1464,14 +1483,14 @@ def test_project_people_belief_and_open_loop_methods_write_audit_events() -> Non None, 3, ) - assert "UPDATE projects" in cursor.statements[4][0] - assert "INSERT INTO people" in cursor.statements[6][0] - assert "UPDATE people" in cursor.statements[9][0] - assert "INSERT INTO beliefs" in cursor.statements[11][0] - assert "UPDATE beliefs" in cursor.statements[14][0] - assert "INSERT INTO open_loops" in cursor.statements[16][0] - assert "UPDATE open_loops" in cursor.statements[19][0] - assert "UPDATE open_loops" in cursor.statements[21][0] + assert "UPDATE projects" in cursor.statements[5][0] + assert "INSERT INTO people" in cursor.statements[7][0] + assert "UPDATE people" in cursor.statements[10][0] + assert "INSERT INTO beliefs" in cursor.statements[12][0] + assert "UPDATE beliefs" in cursor.statements[15][0] + assert "INSERT INTO open_loops" in cursor.statements[17][0] + assert "UPDATE open_loops" in cursor.statements[20][0] + assert "UPDATE open_loops" in cursor.statements[23][0] def test_get_artifact_for_update_locks_the_persisted_authorization_target() -> None: diff --git a/tests/unit/test_workspaces_router_split.py b/tests/unit/test_workspaces_router_split.py index 9f0aa630c..9718ba3fa 100644 --- a/tests/unit/test_workspaces_router_split.py +++ b/tests/unit/test_workspaces_router_split.py @@ -122,7 +122,9 @@ # lone_surrogates.py and main.py only registers it, so it adds no definition # here. Earlier re-pin (2026-09-26): _rewrite_user_id_json_body writes the # rewritten JSON into request._body before call_next. -EXPECTED_CARRIER_AST_SHA256 = "ab3fc6d61cb81a1b9c1a6573adc8e1e297cbbcf01e230effd4a0824dee2d8e2b" + +# Source regeneration adds one protected write route without new carrier definitions. +EXPECTED_CARRIER_AST_SHA256 = "fcd6d722e2d6c3be28b138449555e703f6930168f52b8e94b41dacbcb937b605" EXPECTED_ROUTE_NODE_SHA256 = { "get_vnext_workspace": "6c2151bf38b1b1311f016c00d14394afc7077a6ea219f7ce3dcfd9b701474ae7", "bootstrap_v1_workspace": "07b1fe2a4cd03a5ba69abe76e258a457e85e92b0bfba592520ee02d01d759c4b", @@ -587,7 +589,7 @@ def test_workspace_routes_preserve_mount_order_origins_and_operation_ids() -> No for method in sorted(getattr(route, "methods", None) or set()) if method in {"GET", "POST", "PUT", "PATCH", "DELETE"} ] - expected_indices = (84, 224, 225) if main_module.LEGACY_SURFACES_ENABLED else (38, 175, 176) + expected_indices = (84, 225, 226) if main_module.LEGACY_SURFACES_ENABLED else (38, 176, 177) assert all(effective_pairs.count((method, path)) == 1 for method, path, _name in EXPECTED_ROUTE_MANIFEST) observed_indices = tuple( effective_pairs.index((method, path)) From 429ea61816ca477ca87bfe8d11c66f0b87941359 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Tue, 6 Oct 2026 00:55:16 +0200 Subject: [PATCH 70/90] Pipeline live capture locks before ordered inserts --- .../src/alicebot_api/vnext_label_writes.py | 5 ++- .../test_capture_label_batch_postgres.py | 43 ++++++++++++++++++- .../test_label_floor_ancestry_postgres.py | 21 ++++++--- 3 files changed, 62 insertions(+), 7 deletions(-) diff --git a/apps/api/src/alicebot_api/vnext_label_writes.py b/apps/api/src/alicebot_api/vnext_label_writes.py index 0efd18e34..68f905528 100644 --- a/apps/api/src/alicebot_api/vnext_label_writes.py +++ b/apps/api/src/alicebot_api/vnext_label_writes.py @@ -122,7 +122,10 @@ def capture_label_inputs(store: Any, source_id: str) -> Iterator[None]: batch = _CaptureLabelInputs(conn, transaction_id, int(getattr(conn, "_alice_label_rollback_counter", 0)), identifier(source_id)) token = _CAPTURE_LABEL_INPUTS.set(batch) try: - yield + # Every lock statement remains live and precedes its INSERT on the + # server. Fetching the INSERT result synchronizes the ordered queue. + with conn.pipeline(): + yield finally: _CAPTURE_LABEL_INPUTS.reset(token) diff --git a/tests/integration/test_capture_label_batch_postgres.py b/tests/integration/test_capture_label_batch_postgres.py index e7a6d8549..49f7410aa 100644 --- a/tests/integration/test_capture_label_batch_postgres.py +++ b/tests/integration/test_capture_label_batch_postgres.py @@ -114,7 +114,6 @@ def test_source_raise_invalidates_inputs_in_the_same_transaction(migrated_databa with writes.capture_label_inputs(store, str(source["id"])): assert _memory(store, "before", source)["sensitivity"] == "public" store.update_source(source_id=str(source["id"]), patch={"sensitivity": "confidential"}) - assert writes._current_capture_inputs(store).rows == {} assert _memory(store, "after", source)["sensitivity"] == "confidential" @@ -157,3 +156,45 @@ def test_one_store_starts_fresh_source_inputs_in_each_transaction(migrated_datab assert writes._current_capture_inputs(store) is None conn.commit() assert keys[0][0] != keys[1][0] + + +def test_pipeline_writer_waits_for_live_lock_and_reads_committed_source(migrated_database_urls): + from concurrent.futures import ThreadPoolExecutor + from threading import Event + from time import monotonic, sleep + + user = uuid4() + url = migrated_database_urls["app"] + with user_connection(url, user) as conn: + _user(conn, user) + source = _source(PostgresVNextStore(conn), user) + queued = Event() + def writer(): + with user_connection(url, user) as conn: + store = PostgresVNextStore(conn) + real_lock = store.lock_label_writes + def lock(*, exclusive=False): + real_lock(exclusive=exclusive) + queued.set() + store.lock_label_writes = lock + with writes.capture_label_inputs(store, str(source["id"])): + return _memory(store, "waited", source) + with ThreadPoolExecutor(max_workers=1) as pool: + with user_connection(url, user) as conn: + store = PostgresVNextStore(conn) + store.lock_graph_mutation() + store.lock_label_writes(exclusive=True) + store.update_source(source_id=str(source["id"]), patch={"sensitivity": "confidential"}) + future = pool.submit(writer) + assert queued.wait(3) + deadline = monotonic() + 3 + while monotonic() < deadline: + waiting = conn.execute("SELECT count(*) AS n FROM pg_locks WHERE locktype='advisory' AND NOT granted AND classid=(hashtext('vnext_labels')::bigint & 4294967295)::oid AND objid=(hashtext(app.current_user_id()::text)::bigint & 4294967295)::oid").fetchone()["n"] + if waiting: + break + sleep(0.01) + assert waiting == 1 + assert future.done() is False + assert conn.execute("SELECT count(*) AS n FROM memories").fetchone()["n"] == 0 + result = future.result(timeout=5) + assert result["sensitivity"] == "confidential" diff --git a/tests/integration/test_label_floor_ancestry_postgres.py b/tests/integration/test_label_floor_ancestry_postgres.py index 882999373..ce42fae48 100644 --- a/tests/integration/test_label_floor_ancestry_postgres.py +++ b/tests/integration/test_label_floor_ancestry_postgres.py @@ -117,16 +117,27 @@ def test_postgres_owner_edit_is_clamped_in_response_event_and_storage(migrated_d response = router.review_vnext_memory(UUID(str(memory["id"])), router.VNextMemoryReviewRequest(user_id=user_id, action="edit", domain="project", sensitivity="public"), authorization=None) assert response.status_code == 200 payload = json.loads(response.body) - assert payload["label_floor_applied"] is True - assert payload["memory"]["domain"] == "health" - assert payload["memory"]["sensitivity"] == "confidential" with user_connection(url, user_id) as conn: store = PostgresVNextStore(conn) stored = store.get_memory(str(memory["id"])) - assert stored["domain"] == "health" and stored["sensitivity"] == "confidential" assert stored["metadata_json"]["source_id"] == str(source["id"]) event = conn.execute("SELECT payload_json FROM event_log WHERE event_type='memory.labels_raised' AND target_id=%s", (str(memory["id"]),)).fetchone() - assert event["payload_json"]["cause"] == "floor_clamped" + observed = { + "response_flag": payload.get("label_floor_applied", False), + "response_domain": payload["memory"]["domain"], + "response_sensitivity": payload["memory"]["sensitivity"], + "stored_domain": stored["domain"], + "stored_sensitivity": stored["sensitivity"], + "event_cause": event["payload_json"].get("cause") if event else None, + } + assert observed == { + "response_flag": True, + "response_domain": "health", + "response_sensitivity": "confidential", + "stored_domain": "health", + "stored_sensitivity": "confidential", + "event_cause": "floor_clamped", + } encoded = json.dumps(event["payload_json"]) assert "Synthetic private input" not in encoded assert "Synthetic private observation" not in encoded From 9eaee88da4814be7916ac39ad51a855ae0873be5 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Tue, 6 Oct 2026 00:53:05 +0200 Subject: [PATCH 71/90] Measure committed derived-label work in disposable PostgreSQL fixtures --- scripts/measure_derived_label_budgets.py | 211 +++++++++++++++++++++++ 1 file changed, 211 insertions(+) create mode 100644 scripts/measure_derived_label_budgets.py diff --git a/scripts/measure_derived_label_budgets.py b/scripts/measure_derived_label_budgets.py new file mode 100644 index 000000000..02ab97704 --- /dev/null +++ b/scripts/measure_derived_label_budgets.py @@ -0,0 +1,211 @@ +#!/usr/bin/env python3 +"""Measure actual capture and relabel work in fresh disposable PostgreSQL databases.""" + +from __future__ import annotations + +import argparse +import cProfile +import json +import os +from pathlib import Path +import pstats +import statistics +import time +from uuid import uuid4 + +from alembic import command + +from alicebot_api.db import user_connection +from alicebot_api.migrations import make_alembic_config +from alicebot_api.store import ContinuityStore +from alicebot_api.vnext_capture import SourceCaptureInput, VNextCaptureService +from alicebot_api.vnext_store import PostgresVNextStore +from tests.integration.conftest import _create_role_separated_database, _drop_database + + +def capture(url, repetitions): + samples = [] + for repeat in range(repetitions + 1): + user_id = uuid4() + with user_connection(url, user_id) as conn: + ContinuityStore(conn).create_user(user_id, f"capture-{user_id}@example.invalid", "Capture") + store = PostgresVNextStore(conn) + text = "\n".join( + f"Decision: Synthetic item {index} is assigned to synthetic route {index}." for index in range(200) + ) + start = time.perf_counter() + profile = cProfile.Profile() if os.getenv("ALICE_BUDGET_PROFILE") else None + if profile: + profile.enable() + result = VNextCaptureService(store).capture_source( + SourceCaptureInput( + source_type="manual_text", + title="Synthetic capture", + raw_text=text, + domain="project", + sensitivity="public", + ) + ) + elapsed = time.perf_counter() - start + if profile: + profile.disable() + profile.dump_stats(os.environ["ALICE_BUDGET_PROFILE"]) + pstats.Stats(profile).sort_stats("cumulative").print_stats(45) + assert result.candidate_memory_count == 200, result.to_record() + with user_connection(url, user_id) as conn: + assert conn.execute("SELECT count(*) AS n FROM memories").fetchone()["n"] == 200 + if repeat: + samples.append(elapsed) + return { + "facts": 200, + "warmup_runs": 1, + "repetitions": repetitions, + "seconds": samples, + "median_seconds": statistics.median(samples), + "timing_includes_commit": True, + "fixture_validation_timed": False, + } + + +def relabel(url, repetitions, dependants, *, state=None, prepare=False): + from alicebot_api.vnext_derived_labels import with_derived_from + + user_id = uuid4() if state is None else state["user_id"] + if state is None: + with user_connection(url, user_id) as conn: + ContinuityStore(conn).create_user(user_id, f"relabel-{user_id}@example.invalid", "Relabel") + store = PostgresVNextStore(conn) + source = store.create_source( + { + "source_type": "manual_text", + "title": "Synthetic input", + "content_hash": str(uuid4()), + "domain": "project", + "sensitivity": "public", + } + ) + source_id = str(source["id"]) + conn.execute( + """INSERT INTO memories(id,user_id,memory_key,value,title,canonical_text,status,domain,sensitivity,source_event_ids,metadata_json) + SELECT gen_random_uuid(),app.current_user_id(),'scale.'||n,'{}'::jsonb,'Synthetic row '||n,'Synthetic scale row '||n, + 'active','project','public','[]'::jsonb,CASE WHEN n<=%s THEN jsonb_build_object('source_id',%s::text) ELSE '{}'::jsonb END + FROM generate_series(1,50000) n""", + (dependants, source_id), + ) + record = with_derived_from({"workflow": "daily_brief"}, {"sources": [source]}) + conn.execute( + """INSERT INTO generated_artifacts(id,user_id,artifact_type,title,content_markdown,status,domain,sensitivity,generated_by,metadata_json) + SELECT gen_random_uuid(),app.current_user_id(),'daily_brief','Synthetic artifact '||n,'Synthetic scale artifact '||n, + 'needs_review','project','public','system',CASE WHEN n<=%s THEN %s::jsonb ELSE %s::jsonb END + FROM generate_series(1,2000) n""", + (dependants, json.dumps(record), json.dumps(with_derived_from({"workflow": "daily_brief"}, {}))), + ) + else: + source_id = state["source_id"] + if prepare: + return {"user_id": str(user_id), "source_id": source_id, "dependants": dependants} + samples = [] + # Commit is timed. Restore only this synthetic fixture outside each timed operation. + for repeat in range(repetitions + 1): + with user_connection(url, user_id) as conn: + store = PostgresVNextStore(conn) + before_events = conn.execute( + "SELECT count(*) AS n FROM event_log WHERE event_type LIKE '%%.labels_raised'" + ).fetchone()["n"] + start = time.perf_counter() + profile = cProfile.Profile() if os.getenv("ALICE_BUDGET_PROFILE") else None + if profile: + profile.enable() + store.update_source(source_id=source_id, patch={"sensitivity": "confidential"}) + elapsed = time.perf_counter() - start + if profile: + profile.disable() + profile.dump_stats(os.environ["ALICE_BUDGET_PROFILE"]) + pstats.Stats(profile).sort_stats("cumulative").print_stats(45) + with user_connection(url, user_id) as conn: + assert conn.execute("SELECT count(*) AS n FROM memories").fetchone()["n"] == 50000 + assert conn.execute("SELECT count(*) AS n FROM generated_artifacts").fetchone()["n"] == 2000 + assert ( + conn.execute("SELECT count(*) AS n FROM memories WHERE sensitivity='confidential'").fetchone()["n"] + == dependants + ) + assert ( + conn.execute( + "SELECT count(*) AS n FROM generated_artifacts WHERE sensitivity='confidential'" + ).fetchone()["n"] + == dependants + ) + assert ( + conn.execute("SELECT sensitivity FROM sources WHERE id=%s", (source_id,)).fetchone()["sensitivity"] + == "confidential" + ) + assert ( + conn.execute("SELECT count(*) AS n FROM event_log WHERE event_type LIKE '%%.labels_raised'").fetchone()[ + "n" + ] + - before_events + == 2 * dependants + ) + conn.execute("UPDATE sources SET sensitivity='public' WHERE id=%s", (source_id,)) + conn.execute("UPDATE memories SET sensitivity='public' WHERE sensitivity='confidential'") + conn.execute("UPDATE generated_artifacts SET sensitivity='public' WHERE sensitivity='confidential'") + if repeat: + samples.append(elapsed) + return { + "sources": 1, + "memories": 50000, + "artifacts": 2000, + "dependent_memories": dependants, + "dependent_artifacts": dependants, + "warmup_runs": 1, + "repetitions": repetitions, + "seconds": samples, + "median_seconds": statistics.median(samples), + "budget_seconds": 3, + "label_events_per_run": 2 * dependants, + "timing_includes_commit": True, + "fixture_validation_and_reset_timed": False, + } + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--mode", choices=("capture", "relabel"), required=True) + parser.add_argument("--repetitions", type=int, default=5) + parser.add_argument("--dependants", type=int, default=100) + parser.add_argument("--output", type=Path, required=True) + parser.add_argument("--prepare", type=Path) + parser.add_argument("--prepared", type=Path) + args = parser.parse_args() + prepared = json.loads(args.prepared.read_text()) if args.prepared else None + name = prepared["database_name"] if prepared else "alice_label_perf_" + uuid4().hex[:12] + urls = prepared["urls"] if prepared else _create_role_separated_database(name) + retained = False + try: + if prepared is None: + command.upgrade(make_alembic_config(urls["admin"]), "head") + if args.prepare: + state = relabel(urls["app"], 0, args.dependants, prepare=True) if args.mode == "relabel" else None + args.prepare.write_text(json.dumps({"database_name": name, "urls": urls, "state": state}, indent=2) + "\n") + retained = True + print("Synthetic scale fixture prepared") + return + record = ( + capture(urls["app"], args.repetitions) + if args.mode == "capture" + else relabel( + urls["app"], + args.repetitions, + prepared["state"]["dependants"] if prepared else args.dependants, + state=prepared["state"] if prepared else None, + ) + ) + args.output.write_text(json.dumps(record, indent=2) + "\n") + print(json.dumps(record)) + finally: + if not retained: + _drop_database(name) + + +if __name__ == "__main__": + main() From 797c859e4b399238281eebc32f33590d3267b21f Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Tue, 6 Oct 2026 01:00:06 +0200 Subject: [PATCH 72/90] Track producer-stage artifact project-floor query argument --- tests/unit/test_vnext_store.py | 1 + 1 file changed, 1 insertion(+) diff --git a/tests/unit/test_vnext_store.py b/tests/unit/test_vnext_store.py index ec0c424b9..3ff885712 100644 --- a/tests/unit/test_vnext_store.py +++ b/tests/unit/test_vnext_store.py @@ -739,6 +739,7 @@ def test_list_artifacts_applies_type_domain_sensitivity_and_limit_filters() -> N ["public", "private"], None, None, + None, 5, ) From f6d23e9273c036e804f6f70903620a16df6e269d Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Tue, 6 Oct 2026 01:02:39 +0200 Subject: [PATCH 73/90] Apply effective loop reference admission at exact reader stage --- apps/api/src/alicebot_api/vnext_open_loop_references.py | 8 +++++++- tests/unit/test_main.py | 2 +- 2 files changed, 8 insertions(+), 2 deletions(-) diff --git a/apps/api/src/alicebot_api/vnext_open_loop_references.py b/apps/api/src/alicebot_api/vnext_open_loop_references.py index 92b51c8ff..958b595ed 100644 --- a/apps/api/src/alicebot_api/vnext_open_loop_references.py +++ b/apps/api/src/alicebot_api/vnext_open_loop_references.py @@ -135,7 +135,13 @@ def withhold_unreadable_references( source_rows = _rows_by_id(store, sorted(source_ids | metadata_ids), bulk="get_sources_by_ids", single="get_source") memory_rows = _rows_by_id(store, sorted(memory_ids | metadata_ids), bulk="get_memories_by_ids", single="get_memory") admitted_sources = frozenset(key for key, row in source_rows.items() if fence.admits(row)) - admitted_memories = frozenset(key for key, row in memory_rows.items() if fence.admits_memory(row)) + from alicebot_api.vnext_label_guard import LabelGuard + + guard = LabelGuard.for_fence(store, fence) + admitted_memories = frozenset( + key for key, row in memory_rows.items() + if isinstance(effective := guard.effective_row("memory", row), Mapping) and fence.admits_memory(effective) + ) refused = (set(source_rows) - admitted_sources) | (set(memory_rows) - admitted_memories) # Every id of a row the fence refuses, and every id at a reference position that names no admitted row (a refused, # a deleted, a removed or a missing one), is withheld at every position of every row of the response, in every diff --git a/tests/unit/test_main.py b/tests/unit/test_main.py index 563513a30..6c8783108 100644 --- a/tests/unit/test_main.py +++ b/tests/unit/test_main.py @@ -645,7 +645,7 @@ def test_openapi_helper_backed_contracts_track_authoritative_response_types() -> def test_openapi_store_row_contracts_track_authoritative_column_sets() -> None: def column_fields(columns: str) -> set[str]: - return {column.strip() for column in columns.split(",") if column.strip()} + return {column.strip().rsplit(" AS ", 1)[-1] for column in columns.split(",") if column.strip()} row_contracts = { ("GET", "/v0/vnext/sources/{source_id}"): SOURCE_COLUMNS, From aa191488823f5c1c7247f7aa75a92311904d7cd6 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Tue, 6 Oct 2026 01:14:02 +0200 Subject: [PATCH 74/90] Enable strict locks for original PostgreSQL behavior proofs --- ...t_derived_labels_main_behavior_postgres.py | 71 +++++++++++++++++++ 1 file changed, 71 insertions(+) create mode 100644 tests/integration/test_derived_labels_main_behavior_postgres.py diff --git a/tests/integration/test_derived_labels_main_behavior_postgres.py b/tests/integration/test_derived_labels_main_behavior_postgres.py new file mode 100644 index 000000000..c7f5cdddd --- /dev/null +++ b/tests/integration/test_derived_labels_main_behavior_postgres.py @@ -0,0 +1,71 @@ +"""Execute the original stale-label behaviors even before the new kernel exists.""" + +from importlib.util import find_spec +from uuid import uuid4 + +import pytest + +from alicebot_api.db import user_connection +from alicebot_api.store import ContinuityStore +from alicebot_api.vnext_store import PostgresVNextStore + + +@pytest.fixture(autouse=True) +def strict_when_the_label_writer_exists(monkeypatch): + # Archived main has no writer module; its real stale-row behavior still runs. + if find_spec("alicebot_api.vnext_label_writes") is not None: + from alicebot_api import vnext_label_writes + + monkeypatch.setattr(vnext_label_writes, "STRICT_LOCK_ORDER", True) + + +def test_source_relabel_reaches_an_existing_report_on_main(migrated_database_urls): + user_id = uuid4() + with user_connection(migrated_database_urls["app"], user_id) as conn: + ContinuityStore(conn).create_user( + user_id, f"main-behavior-{user_id}@example.invalid", "Synthetic main behavior" + ) + store = PostgresVNextStore(conn) + store.lock_graph_mutation() + source = store.create_source( + {"source_type": "note", "content_hash": str(uuid4()), "domain": "project", "sensitivity": "public"} + ) + report = store.create_artifact( + { + "artifact_type": "daily_brief", + "title": "Synthetic report", + "content_markdown": "Synthetic report", + "domain": "project", + "sensitivity": "public", + "metadata_json": {"workflow": "daily_brief", "source_refs": [str(source["id"])]}, + } + ) + lock = getattr(store, "lock_label_writes", None) + if callable(lock): + lock(exclusive=True) + store.update_source(source_id=str(source["id"]), patch={"domain": "health", "sensitivity": "confidential"}) + updated = store.get_artifact(str(report["id"])) + assert updated["domain"] == "health" + assert updated["sensitivity"] == "confidential" + + +def test_insert_floor_reads_current_source_labels_on_main(migrated_database_urls): + user_id = uuid4() + with user_connection(migrated_database_urls["app"], user_id) as conn: + ContinuityStore(conn).create_user(user_id, f"main-floor-{user_id}@example.invalid", "Synthetic main floor") + store = PostgresVNextStore(conn) + source = store.create_source( + {"source_type": "note", "content_hash": str(uuid4()), "domain": "health", "sensitivity": "confidential"} + ) + report = store.create_artifact( + { + "artifact_type": "daily_brief", + "title": "Synthetic report", + "content_markdown": "Synthetic report", + "domain": "project", + "sensitivity": "public", + "metadata_json": {"workflow": "daily_brief", "source_refs": [str(source["id"])]}, + } + ) + assert report["domain"] == "health" + assert report["sensitivity"] == "confidential" From ad719ad2f67e60c47bd889f35138f89074458bb5 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Tue, 6 Oct 2026 01:19:05 +0200 Subject: [PATCH 75/90] Provide current ancestry labels in dashboard test stores --- tests/unit/test_cli.py | 38 +++++++++++++++++++++++++++++++ tests/unit/test_vnext_projects.py | 37 ++++++++++++++++++++++++++++++ 2 files changed, 75 insertions(+) diff --git a/tests/unit/test_cli.py b/tests/unit/test_cli.py index 776760fc1..b7f0a4b1b 100644 --- a/tests/unit/test_cli.py +++ b/tests/unit/test_cli.py @@ -422,6 +422,25 @@ def append_event(self, event: dict[str, object]) -> dict[str, object]: self.events.append(event) return event + def read_label_rows(self, kind: str, ids: list[str]) -> list[dict[str, object]]: + """Mirror narrow dependency reads for this fixture's stored rows.""" + + from alicebot_api.vnext_derived_labels import identifier + + rows = { + "source": self.sources, + "memory": self.memories, + "open_loop": self.open_loops, + "artifact": list(self.artifacts.values()), + "project": list(self.projects.values()), + }.get(kind, []) + wanted = {identifier(value) for value in ids} + fields = ("id", "user_id", "domain", "sensitivity", "metadata_json", "value", "project_id", "source_id", "memory_id", "status", "memory_type", "artifact_type") + return [ + {field: row[field] for field in fields if field in row} + for row in rows if identifier(row.get("id")) in wanted + ] + def upsert_agent_identity(self, identity: dict[str, object], **_kwargs) -> dict[str, object]: row = { **identity, @@ -1685,6 +1704,16 @@ def test_vnext_contradiction_and_belief_cli(monkeypatch) -> None: "memory_type": "belief", } + store.memories.append({ + "id": "memory-belief-1", + "canonical_text": "Alice should auto-promote generated artifacts into memory.", + "memory_type": "belief", + "status": "active", + "domain": "project", + "sensitivity": "private", + "metadata_json": {}, + }) + @contextmanager def fake_vnext_store_context(_ctx): yield store @@ -1794,6 +1823,15 @@ def fake_vnext_store_context(_ctx): assert review_loop_payload["due_at"] == "2026-05-12T09:00:00Z" assert dashboard_payload["counts"]["open_loops"] == 1 + # The CLI must keep the dashboard's current-input admission checks. + store.sources[0]["domain"] = "health" + store.sources[0]["sensitivity"] = "regulated" + restricted = json.loads(dashboard_args.handler(ctx, dashboard_args)) + assert restricted["counts"]["open_loops"] == 0 + store.sources.clear() + missing = json.loads(dashboard_args.handler(ctx, dashboard_args)) + assert missing["counts"]["open_loops"] == 0 + def test_vnext_queue_cli_add_process_review_and_export(monkeypatch, tmp_path: Path) -> None: store = FakeVNextCliStore() diff --git a/tests/unit/test_vnext_projects.py b/tests/unit/test_vnext_projects.py index 01f164afb..eec4242b1 100644 --- a/tests/unit/test_vnext_projects.py +++ b/tests/unit/test_vnext_projects.py @@ -77,6 +77,25 @@ def list_project_update_events( rows.append(event) return rows + def read_label_rows(self, kind: str, ids: list[str]) -> list[dict[str, object]]: + """Mirror narrow dependency reads for this fixture's stored rows.""" + + from alicebot_api.vnext_derived_labels import identifier + + rows = { + "source": self.sources, + "memory": list(self.memories.values()), + "open_loop": list(self.open_loops.values()), + "artifact": list(self.artifacts.values()), + "project": list(self.projects.values()), + }.get(kind, []) + wanted = {identifier(value) for value in ids} + fields = ("id", "user_id", "domain", "sensitivity", "metadata_json", "value", "project_id", "source_id", "memory_id", "status", "memory_type", "artifact_type") + return [ + {field: row[field] for field in fields if field in row} + for row in rows if identifier(row.get("id")) in wanted + ] + def create_artifact(self, artifact: dict[str, object], **_kwargs) -> dict[str, object]: row = {**artifact, "id": f"artifact-{len(self.artifacts) + 1}"} self.artifacts[str(row["id"])] = row @@ -1562,6 +1581,24 @@ def test_open_loop_extraction_and_review_support_source_owner_and_filters() -> N assert dashboard["counts"]["open_loops"] == 1 + +@pytest.mark.parametrize("parent_change", ["missing", "restricted"]) +def test_dashboard_withholds_loops_after_their_source_becomes_unreadable(parent_change) -> None: + store = _seed_store() + service = VNextProjectService(store) + loops = service.extract_open_loops(ProjectAutomationRequest(project_id="project-1", domains=("project",))) + assert service.project_dashboard(project_id="project-1")["counts"]["open_loops"] == 2 + if parent_change == "missing": + store.sources.clear() + else: + store.sources[0]["domain"] = "health" + store.sources[0]["sensitivity"] = "regulated" + dashboard = service.project_dashboard(project_id="project-1") + assert dashboard["counts"]["open_loops"] == 0 + assert dashboard["open_loops"] == [] + assert all(loop["source_id"] == "source-1" for loop in loops) + + def test_project_service_validation_errors() -> None: service = VNextProjectService(InMemoryVNextProjectStore()) From c2369835fce09020bde8c99547f7754e46fe41de Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Tue, 6 Oct 2026 01:20:51 +0200 Subject: [PATCH 76/90] Stage dashboard reader controls with the read filters --- tests/unit/test_cli.py | 9 --------- tests/unit/test_vnext_projects.py | 15 --------------- 2 files changed, 24 deletions(-) diff --git a/tests/unit/test_cli.py b/tests/unit/test_cli.py index b7f0a4b1b..0a9fdf937 100644 --- a/tests/unit/test_cli.py +++ b/tests/unit/test_cli.py @@ -1823,15 +1823,6 @@ def fake_vnext_store_context(_ctx): assert review_loop_payload["due_at"] == "2026-05-12T09:00:00Z" assert dashboard_payload["counts"]["open_loops"] == 1 - # The CLI must keep the dashboard's current-input admission checks. - store.sources[0]["domain"] = "health" - store.sources[0]["sensitivity"] = "regulated" - restricted = json.loads(dashboard_args.handler(ctx, dashboard_args)) - assert restricted["counts"]["open_loops"] == 0 - store.sources.clear() - missing = json.loads(dashboard_args.handler(ctx, dashboard_args)) - assert missing["counts"]["open_loops"] == 0 - def test_vnext_queue_cli_add_process_review_and_export(monkeypatch, tmp_path: Path) -> None: store = FakeVNextCliStore() diff --git a/tests/unit/test_vnext_projects.py b/tests/unit/test_vnext_projects.py index eec4242b1..a4cbef901 100644 --- a/tests/unit/test_vnext_projects.py +++ b/tests/unit/test_vnext_projects.py @@ -1582,21 +1582,6 @@ def test_open_loop_extraction_and_review_support_source_owner_and_filters() -> N -@pytest.mark.parametrize("parent_change", ["missing", "restricted"]) -def test_dashboard_withholds_loops_after_their_source_becomes_unreadable(parent_change) -> None: - store = _seed_store() - service = VNextProjectService(store) - loops = service.extract_open_loops(ProjectAutomationRequest(project_id="project-1", domains=("project",))) - assert service.project_dashboard(project_id="project-1")["counts"]["open_loops"] == 2 - if parent_change == "missing": - store.sources.clear() - else: - store.sources[0]["domain"] = "health" - store.sources[0]["sensitivity"] = "regulated" - dashboard = service.project_dashboard(project_id="project-1") - assert dashboard["counts"]["open_loops"] == 0 - assert dashboard["open_loops"] == [] - assert all(loop["source_id"] == "source-1" for loop in loops) def test_project_service_validation_errors() -> None: From e466ce7e45df30b3f1d8ea471bb35d4037b90f7c Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Tue, 6 Oct 2026 01:15:43 +0200 Subject: [PATCH 77/90] Provide recorded label inputs in the semantic rollup fixture --- tests/unit/test_vnext_rollups_semantic.py | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/tests/unit/test_vnext_rollups_semantic.py b/tests/unit/test_vnext_rollups_semantic.py index 85ad78280..4e2ba5d45 100644 --- a/tests/unit/test_vnext_rollups_semantic.py +++ b/tests/unit/test_vnext_rollups_semantic.py @@ -22,6 +22,7 @@ from alicebot_api.sqlite_schema import bootstrap_sqlite_schema from alicebot_api.sqlite_store import SQLiteVNextStore, ensure_sqlite_user +from alicebot_api.vnext_derived_labels import identifier from alicebot_api.vnext_embeddings import memory_embedding_text from alicebot_api.vnext_repositories import JsonObject from alicebot_api.vnext_rollups import ( @@ -61,6 +62,13 @@ def create_memory(self, memory: JsonObject, *, actor_type: str = "system") -> Js def list_memories(self, *, status: str | None = None) -> list[JsonObject]: return [dict(row) for row in self.memories if status is None or row.get("status") == status] + def read_label_rows(self, kind: str, ids) -> list[JsonObject]: + """Expose recorded inputs so an existing card's label can be verified.""" + if kind != "memory": + return [] + wanted = {identifier(item) for item in ids} + return [dict(row) for row in self.memories if identifier(row.get("id")) in wanted] + @staticmethod def _in_scope( row: JsonObject, From 7c8b7fd11701c409c8cc93879559cf2f7b358d83 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Tue, 6 Oct 2026 01:21:25 +0200 Subject: [PATCH 78/90] Take graph and label locks before project review fence --- apps/api/src/alicebot_api/routers/vnext_review.py | 4 ++++ tests/unit/test_vnext_main.py | 6 ++++++ 2 files changed, 10 insertions(+) diff --git a/apps/api/src/alicebot_api/routers/vnext_review.py b/apps/api/src/alicebot_api/routers/vnext_review.py index ea4004c89..7c7cc14fb 100644 --- a/apps/api/src/alicebot_api/routers/vnext_review.py +++ b/apps/api/src/alicebot_api/routers/vnext_review.py @@ -983,6 +983,10 @@ def review_vnext_project_update_candidate( identity = _vnext_authenticated_agent_identity( store, request, user_id=request.user_id, authorization=authorization ) + store.lock_graph_mutation() + from alicebot_api.vnext_label_writes import acquire_exclusive_label_lock + + acquire_exclusive_label_lock(store) _artifact, decision = _vnext_authorized_artifact( store=store, identity=identity, diff --git a/tests/unit/test_vnext_main.py b/tests/unit/test_vnext_main.py index 59cbcdb3a..ebc4564c2 100644 --- a/tests/unit/test_vnext_main.py +++ b/tests/unit/test_vnext_main.py @@ -2677,6 +2677,12 @@ def test_vnext_project_and_open_loop_endpoints(monkeypatch) -> None: update_response = vnext_review_router.generate_vnext_project_update_candidate(request) update_payload = json.loads(update_response.body) extract_response = vnext_projects_router.extract_vnext_open_loops(request) + original_row_locker = store.get_artifact_for_update + def checked_row_locker(artifact_id): + assert store.graph_locked is True + assert store.labels_exclusive is True + return original_row_locker(artifact_id) + monkeypatch.setattr(store, "get_artifact_for_update", checked_row_locker) review_update_response = vnext_review_router.review_vnext_project_update_candidate( update_payload["id"], vnext_review_router.VNextProjectUpdateReviewRequest( From 4c179e759f3624647acb241620413d33ce2ab59f Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Tue, 6 Oct 2026 01:27:49 +0200 Subject: [PATCH 79/90] Count source regeneration in the default surface smoke --- tests/integration/test_default_surface_integration.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/integration/test_default_surface_integration.py b/tests/integration/test_default_surface_integration.py index 5b4549320..d3c48329a 100644 --- a/tests/integration/test_default_surface_integration.py +++ b/tests/integration/test_default_surface_integration.py @@ -20,7 +20,7 @@ REPO_ROOT = Path(__file__).resolve().parents[2] -DEFAULT_HTTP_OPERATION_COUNT = 183 +DEFAULT_HTTP_OPERATION_COUNT = 184 DEFAULT_MCP_TOOL_NAMES = [ "alice_memory_commit", "alice_recall", From 7e9cc413bbecb90e6295353438222b377b290ea7 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Tue, 6 Oct 2026 01:33:00 +0200 Subject: [PATCH 80/90] Provide ordered lock protocol in deferred review fixture --- tests/unit/test_main.py | 48 +++++++++++++++++++++++++++++++++++++++-- 1 file changed, 46 insertions(+), 2 deletions(-) diff --git a/tests/unit/test_main.py b/tests/unit/test_main.py index 6c8783108..476f198c7 100644 --- a/tests/unit/test_main.py +++ b/tests/unit/test_main.py @@ -5416,6 +5416,7 @@ def test_vnext_project_review_defers_embedding_and_preserves_human_attribution(m user_id = uuid4() transaction_depth = 0 calls: list[str] = [] + lock_calls: list[str] = [] review_kwargs: dict[str, object] = {} deferred_input = object() decision = main_module.PolicyDecision( @@ -5434,6 +5435,48 @@ def fake_user_connection(_database_url: str, _user_id): finally: transaction_depth -= 1 + class FakeLockCursor: + def execute(self, query: str, params=None) -> None: + assert transaction_depth == 1 + assert query in { + "SELECT current_setting('lock_timeout') AS lock_timeout", + "SET LOCAL lock_timeout = '3s'", + "SELECT set_config('lock_timeout', %s, true)", + } + if query.startswith("SELECT set_config"): + assert params == ("0",) + + def fetchone(self): + return {"lock_timeout": "0"} + + class FakeLockConnection: + @contextmanager + def cursor(self): + yield FakeLockCursor() + + class FakeStore: + conn = FakeLockConnection() + + def lock_graph_mutation(self) -> None: + assert transaction_depth == 1 + assert lock_calls == [] + lock_calls.append("graph") + + def lock_label_writes(self, *, exclusive: bool = False) -> None: + assert transaction_depth == 1 + assert exclusive is True + assert lock_calls == ["graph"] + lock_calls.append("exclusive_labels") + + store = FakeStore() + + def fake_authorized_artifact(**_kwargs): + assert transaction_depth == 1 + assert lock_calls == ["graph", "exclusive_labels"] + assert _kwargs["store"] is store + lock_calls.append("authorize") + return {"id": "artifact-1", "status": "needs_review"}, decision + class FakeProjectService: def __init__(self, _store, *, defer_embeddings: bool = False) -> None: assert transaction_depth == 1 @@ -5456,12 +5499,12 @@ def fake_persist(**kwargs) -> None: monkeypatch.setattr(vnext_review_router, "get_settings", lambda: Settings(database_url="postgresql://db")) monkeypatch.setattr(vnext_review_router, "user_connection", fake_user_connection) - monkeypatch.setattr(vnext_review_router, "PostgresVNextStore", lambda _conn: object()) + monkeypatch.setattr(vnext_review_router, "PostgresVNextStore", lambda _conn: store) monkeypatch.setattr(vnext_review_router, "_vnext_authenticated_agent_identity", lambda *_args, **_kwargs: None) monkeypatch.setattr( vnext_review_router, "_vnext_authorized_artifact", - lambda **_kwargs: ({"id": "artifact-1", "status": "needs_review"}, decision), + fake_authorized_artifact, ) monkeypatch.setattr(vnext_review_router, "VNextProjectService", FakeProjectService) monkeypatch.setattr(vnext_review_router, "_persist_vnext_deferred_embeddings", fake_persist) @@ -5477,6 +5520,7 @@ def fake_persist(**kwargs) -> None: assert response.status_code == 200 assert calls == ["review", "embedding"] + assert lock_calls == ["graph", "exclusive_labels", "authorize"] assert review_kwargs["actor_type"] == "user" assert review_kwargs["actor_id"] == str(user_id) assert review_kwargs["trace_id"] == "request-trace-1" From cffdb0a0abb9bfc5b2658509e405ecda32f95205 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Tue, 6 Oct 2026 01:34:00 +0200 Subject: [PATCH 81/90] Take exclusive labels before memory review row locks --- apps/api/src/alicebot_api/mcp/memories.py | 3 +++ apps/api/src/alicebot_api/mcp/review.py | 3 +++ apps/api/src/alicebot_api/routers/vnext_memories.py | 9 +++++---- 3 files changed, 11 insertions(+), 4 deletions(-) diff --git a/apps/api/src/alicebot_api/mcp/memories.py b/apps/api/src/alicebot_api/mcp/memories.py index 6165fe47c..1e8757c13 100644 --- a/apps/api/src/alicebot_api/mcp/memories.py +++ b/apps/api/src/alicebot_api/mcp/memories.py @@ -412,6 +412,9 @@ def redact_memory_flow( raise VNextMemoryCommitValidationError("reason is required to redact a memory") memory_service = VNextMemoryCommitService(store) memory_service.lock_supersession_graph() + from alicebot_api.vnext_label_writes import acquire_exclusive_label_lock + + acquire_exclusive_label_lock(store) memory = store.get_memory_for_redaction(memory_id) if memory is None: raise MemoryNotFoundError("memory was not found") diff --git a/apps/api/src/alicebot_api/mcp/review.py b/apps/api/src/alicebot_api/mcp/review.py index 297df7896..515a746de 100644 --- a/apps/api/src/alicebot_api/mcp/review.py +++ b/apps/api/src/alicebot_api/mcp/review.py @@ -542,6 +542,9 @@ def _vnext_memory_correct(context: MCPRuntimeContext, arguments: Mapping[str, ob # approval activates a memory too, so it must not be a row-first # exception to the lifecycle mutation boundary. memory_service.lock_supersession_graph() + from alicebot_api.vnext_label_writes import acquire_exclusive_label_lock + + acquire_exclusive_label_lock(store) get_memory_for_update = getattr(store, "get_memory_for_update", None) memory = get_memory_for_update(memory_id) if callable(get_memory_for_update) else store.get_memory(memory_id) if memory is None: diff --git a/apps/api/src/alicebot_api/routers/vnext_memories.py b/apps/api/src/alicebot_api/routers/vnext_memories.py index f8d44a3ca..f312817b1 100644 --- a/apps/api/src/alicebot_api/routers/vnext_memories.py +++ b/apps/api/src/alicebot_api/routers/vnext_memories.py @@ -1040,16 +1040,17 @@ def review_vnext_memory( # graph boundary before the route takes any candidate/member row lock; # delegated service calls may safely reacquire the transaction lock. memory_service.lock_supersession_graph() + # A status-only review can also raise stale derived labels at the + # owner floor, so acquire the label lock before reading for update. + from alicebot_api.vnext_label_writes import acquire_exclusive_label_lock + + acquire_exclusive_label_lock(store) label_change = ( request.domain is not None or request.sensitivity is not None or request.project_id is not None or action in {"private", "assign_project"} ) - if label_change: - from alicebot_api.vnext_label_writes import acquire_exclusive_label_lock - - acquire_exclusive_label_lock(store) preview = store.get_memory(str(memory_id)) if preview is None: return _vnext_public_error_response(status_code=404, detail="vNext memory was not found") From 34b7477d7a06e4b6d796a6bc9c80cd14b765bbb1 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Tue, 6 Oct 2026 01:45:25 +0200 Subject: [PATCH 82/90] Align existing PostgreSQL fixtures with ordered label writes --- tests/integration/conftest.py | 21 ++++++++++++ .../integration/test_calendar_accounts_api.py | 3 +- .../test_capture_dedupe_postgres.py | 3 ++ tests/integration/test_context_compile.py | 13 +++++--- tests/integration/test_continuity_api.py | 15 +++++++-- .../integration/test_continuity_recall_api.py | 21 ++++++++++-- .../test_continuity_resumption_api.py | 21 ++++++++++-- tests/integration/test_continuity_store.py | 24 ++++++++------ .../test_credential_floor_every_door_api.py | 2 ++ .../test_derived_domain_postgres.py | 2 ++ tests/integration/test_gmail_accounts_api.py | 8 ++++- .../test_mcp_refusal_before_state_postgres.py | 2 ++ tests/integration/test_memory_admission.py | 3 +- .../test_memory_review_labels_api.py | 29 +++++++++-------- ...t_pending_project_update_guard_postgres.py | 2 ++ .../integration/test_provider_runtime_api.py | 32 +++++++++++++++++-- tests/integration/test_proxy_execution_api.py | 8 ++++- .../integration/test_saved_quotes_postgres.py | 4 +++ tests/integration/test_task_artifacts_api.py | 5 ++- tests/integration/test_temporal_state_api.py | 7 ++-- .../test_temporal_state_mcp_cli.py | 7 ++-- .../test_vnext_consolidation_postgres.py | 2 ++ .../test_vnext_live_workspace_api.py | 4 +++ 23 files changed, 192 insertions(+), 46 deletions(-) diff --git a/tests/integration/conftest.py b/tests/integration/conftest.py index 404554c6e..5d8d3667f 100644 --- a/tests/integration/conftest.py +++ b/tests/integration/conftest.py @@ -171,3 +171,24 @@ def migrated_database_urls(migrated_template_database: str) -> Iterator[dict[str yield urls finally: _drop_database(database_name) + + +def lock_label_fixture(store) -> None: + """Seed and mutate one transaction with the production lock order.""" + from alicebot_api.vnext_label_writes import acquire_exclusive_label_lock + + store.lock_graph_mutation() + acquire_exclusive_label_lock(store) + + +def assert_append_only_mutation_refused(conn, *, snapshot_sql, mutation_sql, params) -> None: + """Prove refusal under both forced RLS and the append-only trigger.""" + before = conn.execute(snapshot_sql, params).fetchall() + assert before, "the synthetic target must be visible before attempting its mutation" + try: + with conn.transaction(): + changed = conn.execute(mutation_sql, params).rowcount + assert changed == 0, "an append-only row was changed" + except psycopg.Error as error: + assert "append-only" in str(error) + assert conn.execute(snapshot_sql, params).fetchall() == before diff --git a/tests/integration/test_calendar_accounts_api.py b/tests/integration/test_calendar_accounts_api.py index 708abcc89..f221aac06 100644 --- a/tests/integration/test_calendar_accounts_api.py +++ b/tests/integration/test_calendar_accounts_api.py @@ -13,7 +13,7 @@ from alicebot_api.routers import legacy_gated as legacy_gated_router from alicebot_api.config import Settings import alicebot_api.calendar as calendar_module -from alicebot_api.db import user_connection +from alicebot_api.db import set_current_user, user_connection from alicebot_api.store import ContinuityStore @@ -251,6 +251,7 @@ def test_calendar_account_endpoints_connect_list_detail_and_isolate( assert '"access_token":' not in json.dumps(detail_payload) with psycopg.connect(migrated_database_urls["admin"]) as conn: + set_current_user(conn, owner["user_id"]) with conn.cursor() as cur: cur.execute( """ diff --git a/tests/integration/test_capture_dedupe_postgres.py b/tests/integration/test_capture_dedupe_postgres.py index e54f73ec6..475419ce3 100644 --- a/tests/integration/test_capture_dedupe_postgres.py +++ b/tests/integration/test_capture_dedupe_postgres.py @@ -15,6 +15,7 @@ content_hash_for_text, ) from alicebot_api.vnext_store import PostgresVNextStore +from tests.integration.conftest import lock_label_fixture def test_two_connections_claim_one_source_dedupe_identity( @@ -144,6 +145,7 @@ def test_source_scope_mutation_rotates_postgres_identity_and_releases_old_captur "Capture mutation", ) store = PostgresVNextStore(conn) + lock_label_fixture(store) service = VNextCaptureService(store) text = "Fact: Reviewed source scope changes must rotate identity atomically." first = service.capture_text( @@ -209,6 +211,7 @@ def test_source_scope_mutation_collision_rolls_back_postgres_row( "Capture collision", ) store = PostgresVNextStore(conn) + lock_label_fixture(store) service = VNextCaptureService(store) text = "Fact: Collision rollback leaves both source identities intact." alpha = service.capture_text( diff --git a/tests/integration/test_context_compile.py b/tests/integration/test_context_compile.py index 905b26da4..012996646 100644 --- a/tests/integration/test_context_compile.py +++ b/tests/integration/test_context_compile.py @@ -12,8 +12,9 @@ import alicebot_api.main as main_module from alicebot_api.routers import memories_legacy as memories_legacy_router from alicebot_api.config import Settings -from alicebot_api.db import user_connection +from alicebot_api.db import set_current_user, user_connection from alicebot_api.store import ContinuityStore +from tests.integration.conftest import assert_append_only_mutation_refused def invoke_compile_context(payload: dict[str, Any]) -> tuple[int, dict[str, Any]]: @@ -662,9 +663,13 @@ def test_compile_context_endpoint_persists_trace_and_trace_events(migrated_datab assert trace_events[-1]["payload"]["excluded_entity_edge_limit_count"] == 1 with psycopg.connect(migrated_database_urls["admin"]) as conn: - with conn.cursor() as cur: - with pytest.raises(psycopg.Error, match="append-only"): - cur.execute("UPDATE trace_events SET kind = 'mutated' WHERE trace_id = %s", (trace_id,)) + set_current_user(conn, user_id) + assert_append_only_mutation_refused( + conn, + snapshot_sql="SELECT * FROM trace_events WHERE trace_id = %s ORDER BY id", + mutation_sql="UPDATE trace_events SET kind = 'mutated' WHERE trace_id = %s", + params=(trace_id,), + ) def test_compile_context_prefers_updated_active_memory_within_same_transaction( diff --git a/tests/integration/test_continuity_api.py b/tests/integration/test_continuity_api.py index 1502722ba..e3d45d5a7 100644 --- a/tests/integration/test_continuity_api.py +++ b/tests/integration/test_continuity_api.py @@ -12,7 +12,7 @@ import alicebot_api.main as main_module from alicebot_api.routers import memories_legacy as memories_legacy_router from alicebot_api.config import Settings -from alicebot_api.db import user_connection +from alicebot_api.db import set_current_user, user_connection from alicebot_api.store import ContinuityStore @@ -103,12 +103,14 @@ def seed_user_with_continuity(database_url: str, *, email: str) -> dict[str, obj def set_thread_timestamps( admin_database_url: str, + user_id: UUID, *, thread_id: UUID, created_at: datetime, updated_at: datetime, ) -> None: - with psycopg.connect(admin_database_url, autocommit=True) as conn: + with psycopg.connect(admin_database_url) as conn: + set_current_user(conn, user_id) with conn.cursor() as cur: cur.execute( "UPDATE threads SET created_at = %s, updated_at = %s WHERE id = %s", @@ -118,13 +120,15 @@ def set_thread_timestamps( def set_session_timestamps( admin_database_url: str, + user_id: UUID, *, session_id: UUID, started_at: datetime, ended_at: datetime | None, created_at: datetime, ) -> None: - with psycopg.connect(admin_database_url, autocommit=True) as conn: + with psycopg.connect(admin_database_url) as conn: + set_current_user(conn, user_id) with conn.cursor() as cur: cur.execute( "UPDATE sessions SET started_at = %s, ended_at = %s, created_at = %s WHERE id = %s", @@ -219,24 +223,28 @@ def test_thread_continuity_endpoints_create_list_detail_sessions_and_events( set_thread_timestamps( migrated_database_urls["admin"], + user_id=seeded["user_id"], thread_id=seeded["first_thread"]["id"], created_at=shared_created_at, updated_at=shared_created_at, ) set_thread_timestamps( migrated_database_urls["admin"], + user_id=seeded["user_id"], thread_id=seeded["second_thread"]["id"], created_at=shared_created_at, updated_at=shared_created_at, ) set_thread_timestamps( migrated_database_urls["admin"], + user_id=seeded["user_id"], thread_id=api_thread_id, created_at=newer_created_at, updated_at=newer_created_at, ) set_session_timestamps( migrated_database_urls["admin"], + user_id=seeded["user_id"], session_id=seeded["first_session"]["id"], started_at=first_session_start, ended_at=first_session_end, @@ -244,6 +252,7 @@ def test_thread_continuity_endpoints_create_list_detail_sessions_and_events( ) set_session_timestamps( migrated_database_urls["admin"], + user_id=seeded["user_id"], session_id=seeded["second_session"]["id"], started_at=second_session_start, ended_at=None, diff --git a/tests/integration/test_continuity_recall_api.py b/tests/integration/test_continuity_recall_api.py index d30bc10e6..7d71d206c 100644 --- a/tests/integration/test_continuity_recall_api.py +++ b/tests/integration/test_continuity_recall_api.py @@ -12,7 +12,7 @@ import alicebot_api.main as main_module from alicebot_api.routers import continuity as continuity_router from alicebot_api.config import Settings -from alicebot_api.db import user_connection +from alicebot_api.db import set_current_user, user_connection from alicebot_api.store import ContinuityStore @@ -70,11 +70,13 @@ def seed_user(database_url: str, *, email: str) -> UUID: def set_continuity_timestamps( admin_database_url: str, + user_id: UUID, *, continuity_object_id: UUID, created_at: datetime, ) -> None: - with psycopg.connect(admin_database_url, autocommit=True) as conn: + with psycopg.connect(admin_database_url) as conn: + set_current_user(conn, user_id) with conn.cursor() as cur: cur.execute( "UPDATE continuity_objects SET created_at = %s, updated_at = %s WHERE id = %s", @@ -84,6 +86,7 @@ def set_continuity_timestamps( def set_continuity_lifecycle_flags( admin_database_url: str, + user_id: UUID, *, continuity_object_id: UUID, is_searchable: bool | None = None, @@ -101,7 +104,8 @@ def set_continuity_lifecycle_flags( return values.append(continuity_object_id) - with psycopg.connect(admin_database_url, autocommit=True) as conn: + with psycopg.connect(admin_database_url) as conn: + set_current_user(conn, user_id) with conn.cursor() as cur: cur.execute( f"UPDATE continuity_objects SET {', '.join(assignments)} WHERE id = %s", @@ -170,11 +174,13 @@ def test_continuity_recall_api_returns_provenance_backed_scoped_results( set_continuity_timestamps( migrated_database_urls["admin"], + user_id=user_id, continuity_object_id=primary_object["id"], created_at=datetime(2026, 3, 29, 10, 0, tzinfo=UTC), ) set_continuity_timestamps( migrated_database_urls["admin"], + user_id=user_id, continuity_object_id=other_object["id"], created_at=datetime(2026, 3, 29, 9, 0, tzinfo=UTC), ) @@ -316,6 +322,7 @@ def test_continuity_recall_debug_api_persists_and_exposes_trace( set_continuity_timestamps( migrated_database_urls["admin"], + user_id=user_id, continuity_object_id=continuity_object["id"], created_at=datetime(2026, 3, 29, 10, 0, tzinfo=UTC), ) @@ -393,6 +400,7 @@ def test_continuity_resumption_debug_api_includes_underlying_retrieval_trace( set_continuity_timestamps( migrated_database_urls["admin"], + user_id=user_id, continuity_object_id=decision["id"], created_at=datetime(2026, 3, 29, 10, 0, tzinfo=UTC), ) @@ -486,16 +494,19 @@ def test_continuity_recall_api_prefers_confirmed_fresh_active_truth_over_superse set_continuity_timestamps( migrated_database_urls["admin"], + user_id=user_id, continuity_object_id=current_object["id"], created_at=datetime(2026, 3, 29, 10, 0, tzinfo=UTC), ) set_continuity_timestamps( migrated_database_urls["admin"], + user_id=user_id, continuity_object_id=stale_object["id"], created_at=datetime(2026, 3, 20, 9, 0, tzinfo=UTC), ) set_continuity_timestamps( migrated_database_urls["admin"], + user_id=user_id, continuity_object_id=superseded_object["id"], created_at=datetime(2026, 3, 10, 8, 0, tzinfo=UTC), ) @@ -573,16 +584,19 @@ def test_continuity_recall_api_excludes_preserved_but_non_searchable_objects( set_continuity_lifecycle_flags( migrated_database_urls["admin"], + user_id=user_id, continuity_object_id=hidden_object["id"], is_searchable=False, ) set_continuity_timestamps( migrated_database_urls["admin"], + user_id=user_id, continuity_object_id=hidden_object["id"], created_at=datetime(2026, 3, 29, 10, 0, tzinfo=UTC), ) set_continuity_timestamps( migrated_database_urls["admin"], + user_id=user_id, continuity_object_id=visible_object["id"], created_at=datetime(2026, 3, 29, 10, 5, tzinfo=UTC), ) @@ -637,6 +651,7 @@ def test_continuity_lifecycle_debug_endpoints_expose_flags( set_continuity_lifecycle_flags( migrated_database_urls["admin"], + user_id=user_id, continuity_object_id=continuity_object["id"], is_promotable=False, ) diff --git a/tests/integration/test_continuity_resumption_api.py b/tests/integration/test_continuity_resumption_api.py index 9397924ff..665d96d70 100644 --- a/tests/integration/test_continuity_resumption_api.py +++ b/tests/integration/test_continuity_resumption_api.py @@ -12,7 +12,7 @@ import alicebot_api.main as main_module from alicebot_api.routers import continuity as continuity_router from alicebot_api.config import Settings -from alicebot_api.db import user_connection +from alicebot_api.db import set_current_user, user_connection from alicebot_api.store import ContinuityStore @@ -74,11 +74,13 @@ def seed_user(database_url: str, *, email: str) -> UUID: def set_continuity_timestamps( admin_database_url: str, + user_id: UUID, *, continuity_object_id: UUID, created_at: datetime, ) -> None: - with psycopg.connect(admin_database_url, autocommit=True) as conn: + with psycopg.connect(admin_database_url) as conn: + set_current_user(conn, user_id) with conn.cursor() as cur: cur.execute( "UPDATE continuity_objects SET created_at = %s, updated_at = %s WHERE id = %s", @@ -88,11 +90,13 @@ def set_continuity_timestamps( def set_continuity_lifecycle_flags( admin_database_url: str, + user_id: UUID, *, continuity_object_id: UUID, is_promotable: bool, ) -> None: - with psycopg.connect(admin_database_url, autocommit=True) as conn: + with psycopg.connect(admin_database_url) as conn: + set_current_user(conn, user_id) with conn.cursor() as cur: cur.execute( "UPDATE continuity_objects SET is_promotable = %s WHERE id = %s", @@ -176,21 +180,25 @@ def test_continuity_resumption_api_returns_required_sections( set_continuity_timestamps( migrated_database_urls["admin"], + user_id=user_id, continuity_object_id=decision_object["id"], created_at=datetime(2026, 3, 29, 9, 0, tzinfo=UTC), ) set_continuity_timestamps( migrated_database_urls["admin"], + user_id=user_id, continuity_object_id=waiting_object["id"], created_at=datetime(2026, 3, 29, 10, 0, tzinfo=UTC), ) set_continuity_timestamps( migrated_database_urls["admin"], + user_id=user_id, continuity_object_id=next_object["id"], created_at=datetime(2026, 3, 29, 10, 5, tzinfo=UTC), ) set_continuity_timestamps( migrated_database_urls["admin"], + user_id=user_id, continuity_object_id=latest_decision_object["id"], created_at=datetime(2026, 3, 29, 10, 10, tzinfo=UTC), ) @@ -263,6 +271,7 @@ def test_continuity_resumption_api_returns_explicit_empty_states( set_continuity_timestamps( migrated_database_urls["admin"], + user_id=user_id, continuity_object_id=continuity_object["id"], created_at=datetime(2026, 3, 29, 9, 0, tzinfo=UTC), ) @@ -376,17 +385,20 @@ def test_continuity_resumption_api_selects_latest_sections_beyond_recall_limit( for index, continuity_object_id in enumerate(historical_object_ids): set_continuity_timestamps( migrated_database_urls["admin"], + user_id=user_id, continuity_object_id=continuity_object_id, created_at=base_time + timedelta(minutes=index), ) set_continuity_timestamps( migrated_database_urls["admin"], + user_id=user_id, continuity_object_id=latest_decision_object["id"], created_at=base_time + timedelta(minutes=200), ) set_continuity_timestamps( migrated_database_urls["admin"], + user_id=user_id, continuity_object_id=latest_next_action_object["id"], created_at=base_time + timedelta(minutes=201), ) @@ -465,16 +477,19 @@ def test_continuity_resumption_api_uses_promotable_facts_by_default_with_overrid set_continuity_lifecycle_flags( migrated_database_urls["admin"], + user_id=user_id, continuity_object_id=fact_object["id"], is_promotable=False, ) set_continuity_timestamps( migrated_database_urls["admin"], + user_id=user_id, continuity_object_id=fact_object["id"], created_at=datetime(2026, 3, 29, 10, 0, tzinfo=UTC), ) set_continuity_timestamps( migrated_database_urls["admin"], + user_id=user_id, continuity_object_id=decision_object["id"], created_at=datetime(2026, 3, 29, 10, 5, tzinfo=UTC), ) diff --git a/tests/integration/test_continuity_store.py b/tests/integration/test_continuity_store.py index 956156388..fd91100fc 100644 --- a/tests/integration/test_continuity_store.py +++ b/tests/integration/test_continuity_store.py @@ -9,6 +9,7 @@ from alicebot_api.db import set_current_user, user_connection from alicebot_api.store import ContinuityStore +from tests.integration.conftest import assert_append_only_mutation_refused def test_thread_session_and_event_persistence(migrated_database_urls): @@ -40,12 +41,13 @@ def test_thread_session_and_event_persistence(migrated_database_urls): assert events[0]["payload"]["text"] == "hello" with psycopg.connect(migrated_database_urls["admin"]) as conn: - with pytest.raises(psycopg.Error, match="append-only"): - with conn.cursor() as cur: - cur.execute( - "UPDATE events SET kind = 'message.mutated' WHERE id = %s", - (first_event["id"],), - ) + set_current_user(conn, user_id) + assert_append_only_mutation_refused( + conn, + snapshot_sql="SELECT * FROM events WHERE id = %s", + mutation_sql="UPDATE events SET kind = 'message.mutated' WHERE id = %s", + params=(first_event['id'],), + ) def test_event_deletes_are_rejected_at_database_level(migrated_database_urls): @@ -59,9 +61,13 @@ def test_event_deletes_are_rejected_at_database_level(migrated_database_urls): event = store.append_event(thread["id"], session["id"], "message.user", {"text": "keep"}) with psycopg.connect(migrated_database_urls["admin"]) as conn: - with pytest.raises(psycopg.Error, match="append-only"): - with conn.cursor() as cur: - cur.execute("DELETE FROM events WHERE id = %s", (event["id"],)) + set_current_user(conn, user_id) + assert_append_only_mutation_refused( + conn, + snapshot_sql="SELECT * FROM events WHERE id = %s", + mutation_sql='DELETE FROM events WHERE id = %s', + params=(event['id'],), + ) def test_continuity_rls_blocks_cross_user_access(migrated_database_urls): diff --git a/tests/integration/test_credential_floor_every_door_api.py b/tests/integration/test_credential_floor_every_door_api.py index 04b39f349..fe3d782ee 100644 --- a/tests/integration/test_credential_floor_every_door_api.py +++ b/tests/integration/test_credential_floor_every_door_api.py @@ -40,6 +40,7 @@ from alicebot_api.vnext_store import PostgresVNextStore from tests.integration.test_memory_mutations_api import identity_header, invoke_request, seed_user +from tests.integration.conftest import lock_label_fixture # Built rather than written out so the source carries no scanner-shaped token. @@ -638,6 +639,7 @@ def test_round2_c2_project_update_accept_refuses_a_credential_state(migrated_dat user_id = seed_user(app_url, email="floor-c2-project@example.com") with user_connection(app_url, user_id) as conn: store = PostgresVNextStore(conn) + lock_label_fixture(store) project = store.create_project( { "name": "Deploy pipeline", diff --git a/tests/integration/test_derived_domain_postgres.py b/tests/integration/test_derived_domain_postgres.py index 3c68bc2dc..accabda86 100644 --- a/tests/integration/test_derived_domain_postgres.py +++ b/tests/integration/test_derived_domain_postgres.py @@ -297,6 +297,8 @@ def test_promoted_artifact_uuid_alias_repaired(database_urls): with without_insert_floor(), user_connection(database_urls["app"], user) as conn: ContinuityStore(conn).create_user(user, "alias@example.invalid", "Alias fixture") store = PostgresVNextStore(conn) + # Promotion takes the graph lock before any label-table writes. + store.lock_graph_mutation() memory = store.create_memory({"memory_key": "health", "canonical_text": "Private observation", "domain": "health", "sensitivity": "public", "status": "active"}) artifact = store.create_artifact({"artifact_type": "daily_brief", "title": "Fixture brief", diff --git a/tests/integration/test_gmail_accounts_api.py b/tests/integration/test_gmail_accounts_api.py index 0869c5f61..73d407c70 100644 --- a/tests/integration/test_gmail_accounts_api.py +++ b/tests/integration/test_gmail_accounts_api.py @@ -14,7 +14,7 @@ from alicebot_api.routers import legacy_gated as legacy_gated_router from alicebot_api.config import Settings import alicebot_api.gmail as gmail_module -from alicebot_api.db import user_connection +from alicebot_api.db import set_current_user, user_connection from alicebot_api.store import ContinuityStore @@ -273,6 +273,7 @@ def test_gmail_account_endpoints_connect_list_detail_and_isolate( assert '"client_secret":' not in json.dumps(create_payload) with psycopg.connect(migrated_database_urls["admin"]) as conn: + set_current_user(conn, owner["user_id"]) with conn.cursor() as cur: cur.execute( """ @@ -495,6 +496,7 @@ def fake_fetch_gmail_message_raw_bytes(*, access_token: str, **_kwargs) -> bytes assert '"client_secret":' not in json.dumps(ingest_payload) with psycopg.connect(migrated_database_urls["admin"]) as conn: + set_current_user(conn, owner["user_id"]) with conn.cursor() as cur: cur.execute( """ @@ -636,6 +638,7 @@ def fake_fetch_gmail_message_raw_bytes(*, access_token: str, **_kwargs) -> bytes assert '"client_secret":' not in json.dumps(ingest_payload) with psycopg.connect(migrated_database_urls["admin"]) as conn: + set_current_user(conn, owner["user_id"]) with conn.cursor() as cur: cur.execute( """ @@ -757,6 +760,7 @@ def fail_fetch(**_kwargs): assert store.list_task_artifacts_for_task(owner["task_id"]) == [] with psycopg.connect(migrated_database_urls["admin"]) as conn: + set_current_user(conn, owner["user_id"]) with conn.cursor() as cur: cur.execute( """ @@ -891,6 +895,7 @@ def fail_fetch(**_kwargs): ) with psycopg.connect(migrated_database_urls["admin"]) as conn: + set_current_user(conn, owner["user_id"]) with conn.cursor() as cur: cur.execute( "DELETE FROM gmail_account_credentials WHERE gmail_account_id = %s", @@ -964,6 +969,7 @@ def test_gmail_message_ingestion_endpoint_rejects_missing_external_secret_withou ) with psycopg.connect(migrated_database_urls["admin"]) as conn: + set_current_user(conn, owner["user_id"]) with conn.cursor() as cur: cur.execute( """ diff --git a/tests/integration/test_mcp_refusal_before_state_postgres.py b/tests/integration/test_mcp_refusal_before_state_postgres.py index 8a8a0b455..8d8f43c1f 100644 --- a/tests/integration/test_mcp_refusal_before_state_postgres.py +++ b/tests/integration/test_mcp_refusal_before_state_postgres.py @@ -30,6 +30,7 @@ from alicebot_api.vnext_store import PostgresVNextStore from tests.integration.test_vnext_live_workspace_api import invoke_request, seed_user +from tests.integration.conftest import lock_label_fixture _FIXED_MESSAGE = "The tool request could not be processed" _OWN_PROJECT = "alicebot" @@ -63,6 +64,7 @@ def _states(app_url: str, user_id: UUID) -> dict[str, str]: with user_connection(app_url, user_id) as conn: store = PostgresVNextStore(conn) + lock_label_fixture(store) ordinary = _memory(store) pending = _memory( store, diff --git a/tests/integration/test_memory_admission.py b/tests/integration/test_memory_admission.py index 31a04e735..202f597a1 100644 --- a/tests/integration/test_memory_admission.py +++ b/tests/integration/test_memory_admission.py @@ -11,7 +11,7 @@ import alicebot_api.main as main_module from alicebot_api.routers import memories_legacy as memories_legacy_router from alicebot_api.config import Settings -from alicebot_api.db import user_connection +from alicebot_api.db import set_current_user, user_connection from alicebot_api.store import ContinuityStore @@ -240,6 +240,7 @@ def test_admit_memory_endpoint_persists_add_update_and_delete_revisions( assert revisions[2]["new_value"] is None with psycopg.connect(migrated_database_urls["admin"]) as conn: + set_current_user(conn, user_id) with conn.cursor() as cur: with pytest.raises(psycopg.Error, match="append-only"): cur.execute( diff --git a/tests/integration/test_memory_review_labels_api.py b/tests/integration/test_memory_review_labels_api.py index 288fc03b6..ab80e4ad8 100644 --- a/tests/integration/test_memory_review_labels_api.py +++ b/tests/integration/test_memory_review_labels_api.py @@ -13,9 +13,10 @@ from alicebot_api.routers import memories_legacy as memories_legacy_router from alicebot_api.config import Settings from alicebot_api.contracts import MemoryCandidateInput -from alicebot_api.db import user_connection +from alicebot_api.db import set_current_user, user_connection from alicebot_api.memory import admit_memory_candidate from alicebot_api.store import ContinuityStore +from tests.integration.conftest import assert_append_only_mutation_refused def invoke_request( @@ -296,20 +297,22 @@ def test_memory_review_labels_reject_update_and_delete_at_database_level(migrate ) with psycopg.connect(migrated_database_urls["admin"]) as conn: - with pytest.raises(psycopg.Error, match="append-only"): - with conn.cursor() as cur: - cur.execute( - "UPDATE memory_review_labels SET label = 'incorrect' WHERE id = %s", - (label["id"],), - ) + set_current_user(conn, UUID(seeded["user_id"])) + assert_append_only_mutation_refused( + conn, + snapshot_sql="SELECT * FROM memory_review_labels WHERE id = %s", + mutation_sql="UPDATE memory_review_labels SET label = 'incorrect' WHERE id = %s", + params=(label['id'],), + ) with psycopg.connect(migrated_database_urls["admin"]) as conn: - with pytest.raises(psycopg.Error, match="append-only"): - with conn.cursor() as cur: - cur.execute( - "DELETE FROM memory_review_labels WHERE id = %s", - (label["id"],), - ) + set_current_user(conn, UUID(seeded["user_id"])) + assert_append_only_mutation_refused( + conn, + snapshot_sql="SELECT * FROM memory_review_labels WHERE id = %s", + mutation_sql='DELETE FROM memory_review_labels WHERE id = %s', + params=(label['id'],), + ) def test_memory_review_label_endpoints_enforce_per_user_isolation_and_not_found_behavior( diff --git a/tests/integration/test_pending_project_update_guard_postgres.py b/tests/integration/test_pending_project_update_guard_postgres.py index 70849733a..c9e4dc6b7 100644 --- a/tests/integration/test_pending_project_update_guard_postgres.py +++ b/tests/integration/test_pending_project_update_guard_postgres.py @@ -12,6 +12,7 @@ from alicebot_api.vnext_memory_commit import VNextMemoryCommitService, VNextMemoryCommitValidationError from alicebot_api.vnext_project_update_guard import PENDING_PROJECT_UPDATE_MEMORY_MUTATION_MESSAGE from alicebot_api.vnext_store import PostgresVNextStore +from tests.integration.conftest import lock_label_fixture @pytest.mark.parametrize("marker", ["workflow", "memory_key"]) @@ -29,6 +30,7 @@ def test_postgres_pending_project_update_candidate_blocks_generic_memory_mutatio "Pending project guard", ) store = PostgresVNextStore(conn) + lock_label_fixture(store) metadata: dict[str, object] = {"candidate": True} memory_key = f"ordinary.pending.{uuid4().hex}" if marker == "workflow": diff --git a/tests/integration/test_provider_runtime_api.py b/tests/integration/test_provider_runtime_api.py index 27b5dc408..63adf3ce3 100644 --- a/tests/integration/test_provider_runtime_api.py +++ b/tests/integration/test_provider_runtime_api.py @@ -15,7 +15,7 @@ import alicebot_api.main as main_module from alicebot_api.config import Settings, WorkspaceProviderConfig -from alicebot_api.db import user_connection +from alicebot_api.db import set_current_user, set_current_user_account, user_connection from alicebot_api.provider_configuration import provider_config_fingerprint from alicebot_api.public_errors import UPSTREAM_FAILURE from alicebot_api.provider_secrets import decode_provider_secret_ref, resolve_provider_api_key @@ -148,6 +148,8 @@ def _bootstrap_local_workspace(email: str) -> tuple[str, str, str]: def _seed_thread_for_user(*, admin_db_url: str, user_id: str, email: str) -> str: thread_id = str(uuid4()) with psycopg.connect(admin_db_url) as conn: + set_current_user(conn, UUID(user_id)) + set_current_user_account(conn, UUID(user_id)) with conn.cursor() as cur: cur.execute( """ @@ -842,6 +844,8 @@ def test_openai_compatible_registration_still_works(migrated_database_urls, monk assert any(record["url"] == "https://provider.example/v1/models" for record in captured_requests) with psycopg.connect(migrated_database_urls["admin"]) as conn: + set_current_user(conn, UUID(user_id)) + set_current_user_account(conn, UUID(user_id)) with conn.cursor() as cur: cur.execute( """ @@ -929,6 +933,8 @@ def test_openai_compatible_no_auth_update_omits_auth_for_test_and_runtime( assert len(captured_requests) == 4 assert all("authorization" not in {str(key).lower() for key in record["headers"]} for record in captured_requests) with psycopg.connect(migrated_database_urls["admin"]) as conn: + set_current_user(conn, UUID(user_id)) + set_current_user_account(conn, UUID(user_id)) with conn.cursor() as cur: cur.execute( "SELECT auth_mode, api_key FROM model_providers WHERE id = %s AND workspace_id = %s", @@ -965,6 +971,8 @@ def test_provider_update_uses_atomic_cas_and_hides_stale_capability( migrated_database_urls["admin"], row_factory=psycopg.rows.dict_row, ) as conn: + set_current_user(conn, UUID(user_id)) + set_current_user_account(conn, UUID(user_id)) store = ContinuityStore(conn) original = store.get_model_provider_for_workspace_optional( provider_id=provider_id, @@ -1017,6 +1025,8 @@ def test_provider_update_uses_atomic_cas_and_hides_stale_capability( migrated_database_urls["admin"], row_factory=psycopg.rows.dict_row, ) as conn: + set_current_user(conn, UUID(user_id)) + set_current_user_account(conn, UUID(user_id)) lost_update = ContinuityStore(conn).update_model_provider( provider_id=provider_id, workspace_id=UUID(workspace_id), @@ -1460,6 +1470,8 @@ def test_provider_invocation_telemetry_persists_for_test_and_runtime( assert provider_secret not in caplog.text with psycopg.connect(migrated_database_urls["admin"]) as conn: + set_current_user(conn, UUID(user_id)) + set_current_user_account(conn, UUID(user_id)) with conn.cursor() as cur: cur.execute( """ @@ -1680,6 +1692,8 @@ def fake_urlopen(request, timeout, enforce_public_peer): assert test_payload["result"]["usage"]["total_tokens"] == 14 with psycopg.connect(migrated_database_urls["admin"]) as conn: + set_current_user(conn, UUID(user_id)) + set_current_user_account(conn, UUID(user_id)) with conn.cursor() as cur: cur.execute( """ @@ -1738,6 +1752,8 @@ def fake_urlopen(request, timeout, enforce_public_peer): provider_id = register_payload["provider"]["id"] with psycopg.connect(migrated_database_urls["admin"]) as conn: + set_current_user(conn, UUID(user_id)) + set_current_user_account(conn, UUID(user_id)) with conn.cursor() as cur: cur.execute( """ @@ -1777,6 +1793,8 @@ def fake_urlopen(request, timeout, enforce_public_peer): assert updated_payload["capabilities"]["snapshot"]["azure_auth_mode"] == ("azure_ad_token") with psycopg.connect(migrated_database_urls["admin"]) as conn: + set_current_user(conn, UUID(user_id)) + set_current_user_account(conn, UUID(user_id)) with conn.cursor() as cur: cur.execute( """ @@ -1997,6 +2015,8 @@ def private_dns_getaddrinfo(hostname: str, port, type=0, proto=0): } with psycopg.connect(migrated_database_urls["admin"]) as conn: + set_current_user(conn, UUID(user_id)) + set_current_user_account(conn, UUID(user_id)) with conn.cursor() as cur: cur.execute( """ @@ -2019,7 +2039,7 @@ def test_provider_test_and_runtime_reject_disallowed_target_without_outbound( user_id, workspace_id, user_account_id = _bootstrap_local_workspace("provider-security-blocked-runtime@example.com") urlopen_call_count = 0 - def fake_urlopen(_request, _timeout): + def fake_urlopen(_request, timeout, enforce_public_peer): nonlocal urlopen_call_count urlopen_call_count += 1 raise AssertionError("outbound request should not be attempted for blocked targets") @@ -2042,6 +2062,8 @@ def fake_urlopen(_request, _timeout): provider_id = register_payload["provider"]["id"] with psycopg.connect(migrated_database_urls["admin"]) as conn: + set_current_user(conn, UUID(user_id)) + set_current_user_account(conn, UUID(user_id)) with conn.cursor() as cur: cur.execute( """ @@ -2117,6 +2139,8 @@ def test_provider_rejects_userinfo_and_redacts_legacy_rows( legacy_provider_id: str with psycopg.connect(migrated_database_urls["admin"]) as conn: + set_current_user(conn, UUID(user_id)) + set_current_user_account(conn, UUID(user_id)) with conn.cursor() as cur: cur.execute( """ @@ -2235,6 +2259,8 @@ def fake_urlopen(request, timeout, enforce_public_peer): assert sensitive_detail not in json.dumps(test_payload) with psycopg.connect(migrated_database_urls["admin"]) as conn: + set_current_user(conn, UUID(user_id)) + set_current_user_account(conn, UUID(user_id)) with conn.cursor() as cur: cur.execute( """ @@ -2272,6 +2298,8 @@ def fake_urlopen(request, timeout, enforce_public_peer): assert sensitive_detail not in json.dumps(runtime_payload) with psycopg.connect(migrated_database_urls["admin"]) as conn: + set_current_user(conn, UUID(user_id)) + set_current_user_account(conn, UUID(user_id)) with conn.cursor() as cur: cur.execute( """ diff --git a/tests/integration/test_proxy_execution_api.py b/tests/integration/test_proxy_execution_api.py index 24ca106e8..68909d9d9 100644 --- a/tests/integration/test_proxy_execution_api.py +++ b/tests/integration/test_proxy_execution_api.py @@ -11,7 +11,7 @@ import alicebot_api.main as main_module from alicebot_api.routers import legacy_gated as legacy_gated_router from alicebot_api.config import Settings -from alicebot_api.db import user_connection +from alicebot_api.db import set_current_user, user_connection from alicebot_api.store import ContinuityStore @@ -167,11 +167,13 @@ def create_execution_budget( def set_execution_executed_at( admin_database_url: str, + user_id: UUID, *, execution_id: UUID, executed_at_sql: str, ) -> None: with psycopg.connect(admin_database_url) as conn: + set_current_user(conn, user_id) conn.execute( f"UPDATE tool_executions SET executed_at = {executed_at_sql} WHERE id = %s", (execution_id,), @@ -181,11 +183,13 @@ def set_execution_executed_at( def set_approval_request_thread_id( admin_database_url: str, + user_id: UUID, *, approval_id: UUID, request_thread_id: str, ) -> None: with psycopg.connect(admin_database_url) as conn: + set_current_user(conn, user_id) conn.execute( """ UPDATE approvals @@ -851,6 +855,7 @@ def test_execute_approved_proxy_endpoint_fail_closes_when_runtime_context_is_inv set_approval_request_thread_id( migrated_database_urls["admin"], + user_id=owner["user_id"], approval_id=UUID(create_payload["approval"]["id"]), request_thread_id="not-a-uuid", ) @@ -1317,6 +1322,7 @@ def test_execute_approved_proxy_endpoint_excludes_old_window_history_and_keeps_c set_execution_executed_at( migrated_database_urls["admin"], + user_id=owner["user_id"], execution_id=owner_first_execution_id, executed_at_sql="clock_timestamp() - interval '2 hours'", ) diff --git a/tests/integration/test_saved_quotes_postgres.py b/tests/integration/test_saved_quotes_postgres.py index b08fe6d74..8b7e65569 100644 --- a/tests/integration/test_saved_quotes_postgres.py +++ b/tests/integration/test_saved_quotes_postgres.py @@ -38,6 +38,7 @@ from alicebot_api.vnext_retrieval import VNextRetrievalRequest, VNextRetrievalService from alicebot_api.vnext_source_fence import SavedProvenanceReader, SourceReadFence from alicebot_api.vnext_store import PostgresVNextStore +from tests.integration.conftest import lock_label_fixture _QUOTE = "zinnwald-quote-8841 cone ten firing kiln log marlin-oxide-5520" @@ -84,6 +85,7 @@ def _read(store: PostgresVNextStore, memory_id: str, fence: SourceReadFence) -> def run_lifecycle(store: PostgresVNextStore) -> None: """The lifecycle over one store: save a quote, make the source confidential, archive it, read it three ways.""" + lock_label_fixture(store) source = store.create_source( { "source_type": "document", @@ -198,6 +200,7 @@ def _pack_for(store: PostgresVNextStore, fence: SourceReadFence) -> dict[str, ob def run_linkless_and_sibling_lifecycle(store: PostgresVNextStore) -> None: """A memory with no link and a memory with two links of the same quote, read before and after a source changes.""" + lock_label_fixture(store) refused = _make_source(store, "Alpha held log") kept = _make_source(store, "Alpha second log") linkless = store.create_memory( @@ -309,6 +312,7 @@ def test_a_memory_with_no_link_and_a_sibling_quote_are_withheld_on_postgres(migr def run_nested_reference_lifecycle(store: PostgresVNextStore) -> None: """Two memories that cite a readable source and one that is reclassified, in one ref, read before and after.""" + lock_label_fixture(store) readable = _make_source(store, "Alpha second log") refused = _make_source(store, "Alpha nested log") shapes = { diff --git a/tests/integration/test_task_artifacts_api.py b/tests/integration/test_task_artifacts_api.py index e5092e9e0..fdf78c113 100644 --- a/tests/integration/test_task_artifacts_api.py +++ b/tests/integration/test_task_artifacts_api.py @@ -18,7 +18,7 @@ from alicebot_api.routers import memories_legacy as memories_legacy_router from alicebot_api.config import Settings from alicebot_api.artifacts import TASK_ARTIFACT_CHUNK_RETRIEVAL_MATCHING_RULE -from alicebot_api.db import user_connection +from alicebot_api.db import set_current_user, user_connection from alicebot_api.store import ContinuityStore @@ -1742,6 +1742,7 @@ def test_task_artifact_ingestion_enforces_rooted_workspace_paths( assert register_status == 201 with psycopg.connect(migrated_database_urls["admin"]) as conn: + set_current_user(conn, owner["user_id"]) with conn.cursor() as cur: cur.execute( """ @@ -1821,6 +1822,7 @@ def test_task_artifact_docx_ingestion_enforces_rooted_workspace_paths( assert register_status == 201 with psycopg.connect(migrated_database_urls["admin"]) as conn: + set_current_user(conn, owner["user_id"]) with conn.cursor() as cur: cur.execute( """ @@ -1900,6 +1902,7 @@ def test_task_artifact_rfc822_ingestion_enforces_rooted_workspace_paths( assert register_status == 201 with psycopg.connect(migrated_database_urls["admin"]) as conn: + set_current_user(conn, owner["user_id"]) with conn.cursor() as cur: cur.execute( """ diff --git a/tests/integration/test_temporal_state_api.py b/tests/integration/test_temporal_state_api.py index 0a7bc76f0..fd81b2c3a 100644 --- a/tests/integration/test_temporal_state_api.py +++ b/tests/integration/test_temporal_state_api.py @@ -14,7 +14,7 @@ from alicebot_api.routers import continuity as continuity_router from alicebot_api.config import Settings from alicebot_api.contracts import MemoryCandidateInput -from alicebot_api.db import user_connection +from alicebot_api.db import set_current_user, user_connection from alicebot_api.memory import admit_memory_candidate from alicebot_api.store import ContinuityStore @@ -69,13 +69,15 @@ async def send(message: dict[str, object]) -> None: def _set_temporal_timestamps( admin_database_url: str, + user_id: UUID, *, entity_id: UUID, edge_id: UUID, entity_created_at: datetime, edge_created_at: datetime, ) -> None: - with psycopg.connect(admin_database_url, autocommit=True) as conn: + with psycopg.connect(admin_database_url) as conn: + set_current_user(conn, user_id) with conn.cursor() as cur: cur.execute( "UPDATE entities SET created_at = %s WHERE id = %s", @@ -150,6 +152,7 @@ def seed_temporal_entity_graph(database_url: str, admin_database_url: str) -> di midpoint = add_at + (update_at - add_at) / 2 _set_temporal_timestamps( admin_database_url, + user_id=user_id, entity_id=entity["id"], edge_id=edge["id"], entity_created_at=add_at - timedelta(seconds=1), diff --git a/tests/integration/test_temporal_state_mcp_cli.py b/tests/integration/test_temporal_state_mcp_cli.py index 5550497f4..1e434a508 100644 --- a/tests/integration/test_temporal_state_mcp_cli.py +++ b/tests/integration/test_temporal_state_mcp_cli.py @@ -12,7 +12,7 @@ import psycopg from alicebot_api.contracts import MemoryCandidateInput -from alicebot_api.db import user_connection +from alicebot_api.db import set_current_user, user_connection from alicebot_api.memory import admit_memory_candidate from alicebot_api.store import ContinuityStore @@ -137,13 +137,15 @@ def _call_tool(client: MCPClient, *, name: str, arguments: dict[str, object]) -> def _set_temporal_timestamps( admin_database_url: str, + user_id: UUID, *, entity_id: UUID, edge_id: UUID, entity_created_at: datetime, edge_created_at: datetime, ) -> None: - with psycopg.connect(admin_database_url, autocommit=True) as conn: + with psycopg.connect(admin_database_url) as conn: + set_current_user(conn, user_id) with conn.cursor() as cur: cur.execute("UPDATE entities SET created_at = %s WHERE id = %s", (entity_created_at, entity_id)) cur.execute( @@ -214,6 +216,7 @@ def seed_temporal_entity_graph(database_url: str, admin_database_url: str) -> di midpoint = add_at + (update_at - add_at) / 2 _set_temporal_timestamps( admin_database_url, + user_id=user_id, entity_id=entity["id"], edge_id=edge["id"], entity_created_at=add_at - timedelta(seconds=1), diff --git a/tests/integration/test_vnext_consolidation_postgres.py b/tests/integration/test_vnext_consolidation_postgres.py index 63c08eec5..1b125b2aa 100644 --- a/tests/integration/test_vnext_consolidation_postgres.py +++ b/tests/integration/test_vnext_consolidation_postgres.py @@ -22,6 +22,7 @@ from alicebot_api.vnext_embeddings import pad_embedding_vector from alicebot_api.vnext_memory_commit import VNextMemoryCommitService from alicebot_api.vnext_store import PostgresVNextStore +from tests.integration.conftest import lock_label_fixture def test_generate_memory_consolidation_persists_its_artifact(migrated_database_urls) -> None: @@ -67,6 +68,7 @@ def test_rollup_candidate_round_trips_and_acceptance_promotes_it( with user_connection(migrated_database_urls["app"], user_id) as conn: ContinuityStore(conn).create_user(user_id, "rollups@example.invalid", "Rollups") store = PostgresVNextStore(conn) + lock_label_fixture(store) members = [] for index, (text, day) in enumerate( ( diff --git a/tests/integration/test_vnext_live_workspace_api.py b/tests/integration/test_vnext_live_workspace_api.py index 504648ec4..73653f62d 100644 --- a/tests/integration/test_vnext_live_workspace_api.py +++ b/tests/integration/test_vnext_live_workspace_api.py @@ -30,6 +30,7 @@ VNextProjectTerminalConsistencyError, ) from alicebot_api.vnext_store import PostgresVNextStore +from tests.integration.conftest import lock_label_fixture def invoke_request( @@ -282,6 +283,7 @@ def test_project_update_true_redaction_scrubs_the_role_separated_coupled_graph( with user_connection(migrated_database_urls["app"], user_id) as conn: store = PostgresVNextStore(conn) + lock_label_fixture(store) project = store.create_project( { "name": f"Option A {action} {sentinel}", @@ -785,6 +787,7 @@ def test_project_update_terminal_replay_survives_authorized_true_redaction( ) with user_connection(migrated_database_urls["app"], user_id) as conn: store = PostgresVNextStore(conn) + lock_label_fixture(store) project = store.create_project( { "name": f"{terminal_status.title()} redaction replay", @@ -1006,6 +1009,7 @@ def test_project_update_terminal_replay_rejects_competing_postgres_decision_with ) with user_connection(migrated_database_urls["app"], user_id) as conn: store = PostgresVNextStore(conn) + lock_label_fixture(store) project = store.create_project( { "name": f"{action.title()} competing decision", From dcea6ecbb3148c646eb0c938a55c658ccc937946 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Tue, 6 Oct 2026 01:21:25 +0200 Subject: [PATCH 83/90] Take graph and label locks before project review fence --- apps/api/src/alicebot_api/routers/vnext_review.py | 4 ++++ tests/unit/test_vnext_main.py | 6 ++++++ 2 files changed, 10 insertions(+) diff --git a/apps/api/src/alicebot_api/routers/vnext_review.py b/apps/api/src/alicebot_api/routers/vnext_review.py index 6ebc447e0..0369d0663 100644 --- a/apps/api/src/alicebot_api/routers/vnext_review.py +++ b/apps/api/src/alicebot_api/routers/vnext_review.py @@ -973,6 +973,10 @@ def review_vnext_project_update_candidate( identity = _vnext_authenticated_agent_identity( store, request, user_id=request.user_id, authorization=authorization ) + store.lock_graph_mutation() + from alicebot_api.vnext_label_writes import acquire_exclusive_label_lock + + acquire_exclusive_label_lock(store) _artifact, decision = _vnext_authorized_artifact( store=store, identity=identity, diff --git a/tests/unit/test_vnext_main.py b/tests/unit/test_vnext_main.py index 90335d3dd..de77572fe 100644 --- a/tests/unit/test_vnext_main.py +++ b/tests/unit/test_vnext_main.py @@ -2675,6 +2675,12 @@ def test_vnext_project_and_open_loop_endpoints(monkeypatch) -> None: update_response = vnext_review_router.generate_vnext_project_update_candidate(request) update_payload = json.loads(update_response.body) extract_response = vnext_projects_router.extract_vnext_open_loops(request) + original_row_locker = store.get_artifact_for_update + def checked_row_locker(artifact_id): + assert store.graph_locked is True + assert store.labels_exclusive is True + return original_row_locker(artifact_id) + monkeypatch.setattr(store, "get_artifact_for_update", checked_row_locker) review_update_response = vnext_review_router.review_vnext_project_update_candidate( update_payload["id"], vnext_review_router.VNextProjectUpdateReviewRequest( From 0a3d0c03319205de1ee7560766ca9923dd51591a Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Tue, 6 Oct 2026 01:33:00 +0200 Subject: [PATCH 84/90] Provide ordered lock protocol in deferred review fixture --- tests/unit/test_main.py | 48 +++++++++++++++++++++++++++++++++++++++-- 1 file changed, 46 insertions(+), 2 deletions(-) diff --git a/tests/unit/test_main.py b/tests/unit/test_main.py index 563513a30..1e1c559c3 100644 --- a/tests/unit/test_main.py +++ b/tests/unit/test_main.py @@ -5416,6 +5416,7 @@ def test_vnext_project_review_defers_embedding_and_preserves_human_attribution(m user_id = uuid4() transaction_depth = 0 calls: list[str] = [] + lock_calls: list[str] = [] review_kwargs: dict[str, object] = {} deferred_input = object() decision = main_module.PolicyDecision( @@ -5434,6 +5435,48 @@ def fake_user_connection(_database_url: str, _user_id): finally: transaction_depth -= 1 + class FakeLockCursor: + def execute(self, query: str, params=None) -> None: + assert transaction_depth == 1 + assert query in { + "SELECT current_setting('lock_timeout') AS lock_timeout", + "SET LOCAL lock_timeout = '3s'", + "SELECT set_config('lock_timeout', %s, true)", + } + if query.startswith("SELECT set_config"): + assert params == ("0",) + + def fetchone(self): + return {"lock_timeout": "0"} + + class FakeLockConnection: + @contextmanager + def cursor(self): + yield FakeLockCursor() + + class FakeStore: + conn = FakeLockConnection() + + def lock_graph_mutation(self) -> None: + assert transaction_depth == 1 + assert lock_calls == [] + lock_calls.append("graph") + + def lock_label_writes(self, *, exclusive: bool = False) -> None: + assert transaction_depth == 1 + assert exclusive is True + assert lock_calls == ["graph"] + lock_calls.append("exclusive_labels") + + store = FakeStore() + + def fake_authorized_artifact(**_kwargs): + assert transaction_depth == 1 + assert lock_calls == ["graph", "exclusive_labels"] + assert _kwargs["store"] is store + lock_calls.append("authorize") + return {"id": "artifact-1", "status": "needs_review"}, decision + class FakeProjectService: def __init__(self, _store, *, defer_embeddings: bool = False) -> None: assert transaction_depth == 1 @@ -5456,12 +5499,12 @@ def fake_persist(**kwargs) -> None: monkeypatch.setattr(vnext_review_router, "get_settings", lambda: Settings(database_url="postgresql://db")) monkeypatch.setattr(vnext_review_router, "user_connection", fake_user_connection) - monkeypatch.setattr(vnext_review_router, "PostgresVNextStore", lambda _conn: object()) + monkeypatch.setattr(vnext_review_router, "PostgresVNextStore", lambda _conn: store) monkeypatch.setattr(vnext_review_router, "_vnext_authenticated_agent_identity", lambda *_args, **_kwargs: None) monkeypatch.setattr( vnext_review_router, "_vnext_authorized_artifact", - lambda **_kwargs: ({"id": "artifact-1", "status": "needs_review"}, decision), + fake_authorized_artifact, ) monkeypatch.setattr(vnext_review_router, "VNextProjectService", FakeProjectService) monkeypatch.setattr(vnext_review_router, "_persist_vnext_deferred_embeddings", fake_persist) @@ -5477,6 +5520,7 @@ def fake_persist(**kwargs) -> None: assert response.status_code == 200 assert calls == ["review", "embedding"] + assert lock_calls == ["graph", "exclusive_labels", "authorize"] assert review_kwargs["actor_type"] == "user" assert review_kwargs["actor_id"] == str(user_id) assert review_kwargs["trace_id"] == "request-trace-1" From e100914530e83097a86e6f22b45f6f545b5d8ef4 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Tue, 6 Oct 2026 01:41:32 +0200 Subject: [PATCH 85/90] Model exclusive labels before deferred memory review row locks --- tests/unit/test_main.py | 36 +++++++++++++++++++++++++++++++++++- 1 file changed, 35 insertions(+), 1 deletion(-) diff --git a/tests/unit/test_main.py b/tests/unit/test_main.py index 1e1c559c3..8d431d216 100644 --- a/tests/unit/test_main.py +++ b/tests/unit/test_main.py @@ -5279,6 +5279,7 @@ def test_vnext_memory_review_defers_embedding_until_primary_transaction_closes(m memory_id = uuid4() transaction_depth = 0 calls: list[str] = [] + lock_calls: list[str] = [] deferred_input = object() memory = { "id": str(memory_id), @@ -5300,11 +5301,41 @@ def fake_user_connection(_database_url: str, _user_id): finally: transaction_depth -= 1 + class FakeLockCursor: + def execute(self, query: str, params=None) -> None: + assert transaction_depth == 1 + assert query in { + "SELECT current_setting('lock_timeout') AS lock_timeout", + "SET LOCAL lock_timeout = '3s'", + "SELECT set_config('lock_timeout', %s, true)", + } + if query.startswith("SELECT set_config"): + assert params == ("0",) + + def fetchone(self): + return {"lock_timeout": "0"} + + class FakeLockConnection: + @contextmanager + def cursor(self): + yield FakeLockCursor() + class FakeStore: + conn = FakeLockConnection() + + def lock_label_writes(self, *, exclusive: bool = False) -> None: + assert transaction_depth == 1 + assert exclusive is True + assert lock_calls == ["graph"] + lock_calls.append("exclusive_labels") + def get_memory(self, _memory_id: str): return memory def get_memory_for_update(self, _memory_id: str): + assert transaction_depth == 1 + assert lock_calls == ["graph", "exclusive_labels"] + lock_calls.append("row") return memory def update_memory(self, *, memory_id: str, patch: dict[str, object], **_kwargs): @@ -5329,7 +5360,9 @@ def __init__(self, _store, *, defer_embeddings: bool = False) -> None: self.deferred_embedding_inputs = (deferred_input,) def lock_supersession_graph(self) -> None: - pass + assert transaction_depth == 1 + assert lock_calls == [] + lock_calls.append("graph") def refresh_memory_derived_state(self, _memory, **_kwargs) -> None: assert transaction_depth == 1 @@ -5356,6 +5389,7 @@ def fake_persist(**kwargs) -> None: assert response.status_code == 200 assert calls == ["refresh", "embedding"] + assert lock_calls == ["graph", "exclusive_labels", "row"] def test_vnext_consolidation_defers_embedding_until_primary_transaction_closes(monkeypatch) -> None: From fd66925f546d6c7451afc2c3e20f8f1d46a00090 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Tue, 6 Oct 2026 01:45:52 +0200 Subject: [PATCH 86/90] Import protected identity helpers for the staged memory audit route --- apps/api/src/alicebot_api/routers/vnext_memories.py | 2 ++ 1 file changed, 2 insertions(+) diff --git a/apps/api/src/alicebot_api/routers/vnext_memories.py b/apps/api/src/alicebot_api/routers/vnext_memories.py index abb13be2c..6af8f8f74 100644 --- a/apps/api/src/alicebot_api/routers/vnext_memories.py +++ b/apps/api/src/alicebot_api/routers/vnext_memories.py @@ -61,6 +61,8 @@ ) from alicebot_api.vnext_agent_keys import ( AgentKeyAuthenticationError, + agent_key_from_authorization, + resolve_protected_agent_identity, ) from alicebot_api.vnext_capture import ( VNextCaptureService, From f9d26b0177742c83bbda4d04d28c9164ddf108ca Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Tue, 6 Oct 2026 01:50:20 +0200 Subject: [PATCH 87/90] Model PostgreSQL lock order in CLI redaction fixtures --- tests/unit/test_cli.py | 54 ++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 54 insertions(+) diff --git a/tests/unit/test_cli.py b/tests/unit/test_cli.py index 0a9fdf937..dd8c72013 100644 --- a/tests/unit/test_cli.py +++ b/tests/unit/test_cli.py @@ -398,8 +398,40 @@ def test_parser_preserves_explicit_vnext_sensitivity_filter() -> None: assert cli_module._vnext_sensitivity_allowed(omitted) == ("public", "internal", "private", "unknown") +class FakeVNextCliLockCursor: + def __init__(self, conn) -> None: + self.conn = conn + + def execute(self, query: str, params=None) -> None: + if query == "SELECT current_setting('lock_timeout') AS lock_timeout": + assert params is None + elif query == "SET LOCAL lock_timeout = '3s'": + assert params is None + self.conn.lock_timeout = "3s" + else: + assert query == "SELECT set_config('lock_timeout', %s, true)" + assert params == ("0",) + self.conn.lock_timeout = params[0] + + def fetchone(self): + return {"lock_timeout": self.conn.lock_timeout} + + +class FakeVNextCliLockConnection: + def __init__(self) -> None: + self.lock_timeout = "0" + + @contextmanager + def cursor(self): + yield FakeVNextCliLockCursor(self) + + class FakeVNextCliStore: def __init__(self) -> None: + self.conn = FakeVNextCliLockConnection() + self.graph_locked = False + self.labels_exclusive = False + self.lock_calls: list[str] = [] self.sources: list[dict[str, object]] = [] self.chunks: list[dict[str, object]] = [] self.memories: list[dict[str, object]] = [] @@ -418,6 +450,17 @@ def __init__(self) -> None: self.scheduler_workflows: dict[str, dict[str, object]] = {} self.scheduler_runs: list[dict[str, object]] = [] + def lock_graph_mutation(self) -> None: + self.graph_locked = True + self.lock_calls.append("graph") + + def lock_label_writes(self, *, exclusive: bool = False) -> None: + assert self.graph_locked + if exclusive: + assert self.conn.lock_timeout == "3s" + self.labels_exclusive |= exclusive + self.lock_calls.append("exclusive_labels" if exclusive else "shared_labels") + def append_event(self, event: dict[str, object]) -> dict[str, object]: self.events.append(event) return event @@ -512,6 +555,9 @@ def get_memory_for_update(self, memory_id: str) -> dict[str, object] | None: return self.get_memory(memory_id) def get_memory_for_redaction(self, memory_id: str) -> dict[str, object] | None: + assert self.graph_locked + assert self.labels_exclusive + self.lock_calls.append("redaction_row") return self.get_memory(memory_id) def lock_project_update_artifacts_for_redaction(self, memory_id: str) -> list[dict[str, object]]: @@ -2497,12 +2543,17 @@ def fake_vnext_store_context(_ctx): ) first = json.loads(cli_module._run_vnext_memory_redact(context, args)) + assert store.lock_calls[:3] == ["graph", "exclusive_labels", "redaction_row"] + assert store.conn.lock_timeout == "0" assert first["status"] == "redacted" assert first["forgotten_first"] is True assert first["idempotent_replay"] is False frozen = deepcopy((store.memories, store.artifacts, store.revisions, store.events)) + previous_lock_count = len(store.lock_calls) second = json.loads(cli_module._run_vnext_memory_redact(context, args)) + assert store.lock_calls[previous_lock_count:][:3] == ["graph", "exclusive_labels", "redaction_row"] + assert store.conn.lock_timeout == "0" assert second["status"] == "redacted" assert second["forgotten_first"] is False assert second["idempotent_replay"] is True @@ -2716,6 +2767,9 @@ def test_cli_generic_memory_mutations_cannot_strand_pending_project_update_candi ) assert (store.projects, store.memories, store.artifacts, store.revisions) == state_before assert [event.get("event_type") for event in store.events] == event_types_before + if operation == "redact": + assert store.lock_calls[-3:] == ["graph", "exclusive_labels", "redaction_row"] + assert store.conn.lock_timeout == "0" def _apply_supported_cli_memory_lifecycle( From 98949374745b3b808328f2974b4c810603d8351d Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Tue, 6 Oct 2026 01:57:14 +0200 Subject: [PATCH 88/90] test: acquire label locks before producer fixture writes --- tests/integration/test_derived_labels_group_scope_postgres.py | 2 ++ 1 file changed, 2 insertions(+) diff --git a/tests/integration/test_derived_labels_group_scope_postgres.py b/tests/integration/test_derived_labels_group_scope_postgres.py index 39c093d19..8397a4ed2 100644 --- a/tests/integration/test_derived_labels_group_scope_postgres.py +++ b/tests/integration/test_derived_labels_group_scope_postgres.py @@ -10,6 +10,7 @@ from alicebot_api.vnext_memory_commit import VNextMemoryCommitService from alicebot_api.vnext_rollups import VNextRollupService from alicebot_api.vnext_store import PostgresVNextStore +from tests.integration.conftest import lock_label_fixture from tests.unit.test_group_scope_sqlite import ALPHA, seed_members @@ -19,6 +20,7 @@ def test_a_second_scoped_rollup_run_over_the_same_group_finds_its_card_and_inser with user_connection(migrated_database_urls["app"], user_id) as conn: ContinuityStore(conn).create_user(user_id, f"group-{user_id}@example.invalid", "Group") store = PostgresVNextStore(conn) + lock_label_fixture(store) members = seed_members(store) service = VNextRollupService(store) first = service.propose_rollups(projects=(ALPHA,)) From 6297861fa98d47c918deb7a6b99b6838014054ba Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Tue, 6 Oct 2026 02:16:21 +0200 Subject: [PATCH 89/90] Align saved quote and counting fixtures with exact source guards --- ...st_saved_quotes_follow_the_source_fence.py | 17 ++++++++++-- tests/unit/test_vnext_retrieval.py | 26 ++++++++++++++++--- 2 files changed, 38 insertions(+), 5 deletions(-) diff --git a/tests/unit/test_saved_quotes_follow_the_source_fence.py b/tests/unit/test_saved_quotes_follow_the_source_fence.py index c1d8a66b5..8b238b645 100644 --- a/tests/unit/test_saved_quotes_follow_the_source_fence.py +++ b/tests/unit/test_saved_quotes_follow_the_source_fence.py @@ -1161,8 +1161,8 @@ def test_a_source_with_no_project_is_outside_the_fence_of_a_key_bound_to_a_proje what ``alice_explain`` has always done for such a key. A key bound to no project reads it. On v0.20.0 the review returned all three to a key bound to a project. The release notes say this in one sentence. - The memory is planted with the two writes a review makes (the metadata copy and a link), as the lifecycle tests of the - review door do. + The memory is an original row with the two saved provenance copies that a review makes, as the lifecycle tests of + this projection do. The derived-copy control below separately verifies refusal of the whole captured copy. Mutation: pass ``require_explicit_project_scope=False`` in ``SourceReadFence._admits``: the keys bound to ``alpha`` read the link, the quote and the id again, and disagree with explain. @@ -1173,6 +1173,7 @@ def test_a_source_with_no_project_is_outside_the_fence_of_a_key_bound_to_a_proje assert json.loads(row["metadata_json"])["project_scope"] == [], "the owner's capture belongs to no project" candidate = vault._candidate("Projectlessnote: the alpha kiln shelf is cleaned on Fridays") _plant_saved_quote(vault, candidate, source_id) + vault._original_quote_fixture(candidate) for who in _KEY_SPECS: answer = vault.review(who, candidate) assert answer["is_error"] is False, (who, answer) @@ -1184,6 +1185,18 @@ def test_a_source_with_no_project_is_outside_the_fence_of_a_key_bound_to_a_proje +def test_a_derived_copy_of_a_projectless_source_is_denied_to_bound_keys(vault: _Vault) -> None: + source_id = vault.capture_source(as_owner=True) + candidate = vault._candidate("Projectlesscopy: the alpha kiln shelf is cleaned on Fridays") + _plant_saved_quote(vault, candidate, source_id) + for who in _KEY_SPECS: + answer = vault.review(who, candidate) + readable = who == "unbound" + assert answer["is_error"] is not readable, (who, answer) + assert _holds_quote(answer) is readable, who + assert _holds_source_id(answer, source_id) is readable, who + + def test_current_derived_copy_is_denied_as_a_whole_after_source_relabel(vault: _Vault) -> None: source_id = vault.capture_source() memory_id, _query = vault.edit_and_approve(source_id) diff --git a/tests/unit/test_vnext_retrieval.py b/tests/unit/test_vnext_retrieval.py index 773b819cb..70454a4a5 100644 --- a/tests/unit/test_vnext_retrieval.py +++ b/tests/unit/test_vnext_retrieval.py @@ -1040,10 +1040,30 @@ def test_keyword_query_that_and_matches_does_not_use_the_fallback_on_sqlite() -> def test_count_candidate_statistic_uses_real_sqlite_fts_mode_and_provenance_dedup() -> None: store = _sqlite_retrieval_store() + source_a = store.create_source( + { + "source_type": "note", + "title": "Bike service source A", + "content_hash": "sha256:bike-a", + "domain": "personal", + "sensitivity": "private", + "captured_at": "2026-01-05T00:00:00Z", + } + ) + source_b = store.create_source( + { + "source_type": "note", + "title": "Bike service source B", + "content_hash": "sha256:bike-b", + "domain": "personal", + "sensitivity": "private", + "captured_at": "2026-01-06T00:00:00Z", + } + ) provenance = ( - ("record-1", "source-a", "chunk-a"), - ("record-2", "source-a", "chunk-a"), # restatement of the same captured turn - ("record-3", "source-b", "chunk-b"), + ("record-1", source_a["id"], "chunk-a"), + ("record-2", source_a["id"], "chunk-a"), # restatement of the same captured turn + ("record-3", source_b["id"], "chunk-b"), ) for memory_key, source_id, chunk_id in provenance: store.create_memory( From 87cd9d0b2ba9d1ccd9617211035eeab2b4d6fb07 Mon Sep 17 00:00:00 2001 From: Sami Rusani Date: Tue, 6 Oct 2026 02:18:46 +0200 Subject: [PATCH 90/90] Seed real source parents for SQLite count-candidate fixture --- tests/unit/test_vnext_retrieval.py | 26 +++++++++++++++++++++++--- 1 file changed, 23 insertions(+), 3 deletions(-) diff --git a/tests/unit/test_vnext_retrieval.py b/tests/unit/test_vnext_retrieval.py index 773b819cb..70454a4a5 100644 --- a/tests/unit/test_vnext_retrieval.py +++ b/tests/unit/test_vnext_retrieval.py @@ -1040,10 +1040,30 @@ def test_keyword_query_that_and_matches_does_not_use_the_fallback_on_sqlite() -> def test_count_candidate_statistic_uses_real_sqlite_fts_mode_and_provenance_dedup() -> None: store = _sqlite_retrieval_store() + source_a = store.create_source( + { + "source_type": "note", + "title": "Bike service source A", + "content_hash": "sha256:bike-a", + "domain": "personal", + "sensitivity": "private", + "captured_at": "2026-01-05T00:00:00Z", + } + ) + source_b = store.create_source( + { + "source_type": "note", + "title": "Bike service source B", + "content_hash": "sha256:bike-b", + "domain": "personal", + "sensitivity": "private", + "captured_at": "2026-01-06T00:00:00Z", + } + ) provenance = ( - ("record-1", "source-a", "chunk-a"), - ("record-2", "source-a", "chunk-a"), # restatement of the same captured turn - ("record-3", "source-b", "chunk-b"), + ("record-1", source_a["id"], "chunk-a"), + ("record-2", source_a["id"], "chunk-a"), # restatement of the same captured turn + ("record-3", source_b["id"], "chunk-b"), ) for memory_key, source_id, chunk_id in provenance: store.create_memory(