feat(drive): start implementation of drive

- add storage.drives
    - prepare migration phase
    - add created_by and updated_by on storage.folders
This commit is contained in:
Edouard Vanbelle
2026-06-18 13:29:41 +02:00
parent 77545aee05
commit eab7a609b9
43 changed files with 2434 additions and 154 deletions
@@ -0,0 +1,155 @@
-- ════════════════════════════════════════════════════════════════════════════
-- D0 / M1 — Drive foundation: additive schema only
-- ════════════════════════════════════════════════════════════════════════════
-- First of three D0 migrations.
-- M1 (this file) additive — creates storage.drives + nullable columns.
-- M2 (next) backfill — promotes wrapper folders, fills drive_id.
-- M3 (final) constraints — NOT NULL + FKs + drive_id indexes.
--
-- This file is **safe to run on a populated database without an outage**.
-- It only ADDs structure (new table, new nullable columns, extended CHECK
-- constraints, new FK targets). No row is modified; no existing query
-- needs to be aware of the new columns yet.
--
-- The migration is reversible at this stage: dropping the new table and
-- the new columns leaves the database identical to its pre-D0 state. The
-- dual-write / data-movement phase (M2) is where rollback becomes
-- progressively harder.
-- ── 1. storage.drives — the central drive entity ────────────────────────────
CREATE TABLE IF NOT EXISTS storage.drives (
id UUID PRIMARY KEY DEFAULT gen_random_uuid(),
name TEXT NOT NULL,
-- Discriminant. Two kinds today; extending the set is a DROP + ADD
-- CHECK constraint pair (no separate lookup table).
-- 'personal' = single-owner, no membership API
-- 'shared' = multi-member, full role roster, group-aware
kind TEXT NOT NULL
CHECK (kind IN ('personal', 'shared')),
-- Set iff this is the user's default personal drive. The partial
-- unique index below enforces "one default drive per user" without
-- blocking secondaries (NULL means "not the default"); shared drives
-- always have NULL here.
default_for_user UUID
REFERENCES auth.users(id) ON DELETE CASCADE,
-- Storage quota in bytes. NULL = no quota (admin override / system
-- drives). Initial value on personal-drive creation is taken from
-- the owner's `auth.users.storage_quota_bytes` at the application
-- layer.
quota_bytes BIGINT,
-- Running total of bytes consumed. Maintained by D4's incremental
-- counters; on D0 backfilled from the per-user counters as a
-- starting baseline.
used_bytes BIGINT NOT NULL DEFAULT 0,
-- Capability flags / feature toggles bag (see docs/plan/drive.md §8
-- and §15 for the known keys: forbid_public_links,
-- forbid_external_sharing, include_in_photo_index, forbid_music_index,
-- etc.). Unknown keys preserved verbatim — the schema is
-- intentionally permissive so future flags land without migration.
policies JSONB NOT NULL DEFAULT '{}'::jsonb,
created_at TIMESTAMPTZ NOT NULL DEFAULT now(),
updated_at TIMESTAMPTZ NOT NULL DEFAULT now()
);
COMMENT ON TABLE storage.drives IS
'Drive entity. Top-level container that owns a tree of folders/files; '
'membership lives in storage.role_grants with resource_type=''drive''. '
'Replaced the per-user My Folder wrapper at D0 (see docs/plan/drive.md).';
COMMENT ON COLUMN storage.drives.kind IS
'personal = single-owner (no add_member); shared = multi-member with full role roster.';
COMMENT ON COLUMN storage.drives.default_for_user IS
'Set iff this is the user''s default personal drive. NULL on secondaries and shared drives.';
COMMENT ON COLUMN storage.drives.policies IS
'JSONB capability-flag bag; see docs/plan/drive.md §8 §15 for known keys.';
-- "One default drive per user." Partial unique index — NULLs (every
-- non-default row) are excluded from the constraint surface.
CREATE UNIQUE INDEX IF NOT EXISTS idx_drives_default_for_user_unique
ON storage.drives (default_for_user)
WHERE default_for_user IS NOT NULL;
-- Hot-path "what's this drive?" lookup by kind for admin / D3 flows
-- ("list every shared drive").
CREATE INDEX IF NOT EXISTS idx_drives_kind ON storage.drives (kind);
-- ── 2. drive_id columns on folders + files (NULL during M1) ────────────────
-- M2 fills these in for every existing row; M3 promotes them to NOT NULL
-- and adds the FK + index. Keep nullable here so the migration runs on a
-- populated DB without violating a constraint.
ALTER TABLE storage.folders ADD COLUMN IF NOT EXISTS drive_id UUID;
ALTER TABLE storage.files ADD COLUMN IF NOT EXISTS drive_id UUID;
-- ── 3. Provenance columns: created_by / updated_by ─────────────────────────
-- D0 adds these on folders + files (see docs/plan/drive.md §14). FKs use
-- ON DELETE SET NULL so deleting a user nulls these out instead of
-- cascading the resource away. M2 backfills from the existing `user_id`
-- column so pre-Drive content carries authentic provenance from day one.
ALTER TABLE storage.folders
ADD COLUMN IF NOT EXISTS created_by UUID
REFERENCES auth.users(id) ON DELETE SET NULL;
ALTER TABLE storage.folders
ADD COLUMN IF NOT EXISTS updated_by UUID
REFERENCES auth.users(id) ON DELETE SET NULL;
ALTER TABLE storage.files
ADD COLUMN IF NOT EXISTS created_by UUID
REFERENCES auth.users(id) ON DELETE SET NULL;
ALTER TABLE storage.files
ADD COLUMN IF NOT EXISTS updated_by UUID
REFERENCES auth.users(id) ON DELETE SET NULL;
COMMENT ON COLUMN storage.folders.created_by IS
'Who originally created the folder. NULL when the original creator''s '
'auth.users row has since been deleted.';
COMMENT ON COLUMN storage.folders.updated_by IS
'Who last touched the folder (rename, move, metadata change). Same '
'write-path discipline as updated_at.';
COMMENT ON COLUMN storage.files.created_by IS
'Who originally uploaded the file. NULL when the original uploader''s '
'auth.users row has since been deleted.';
COMMENT ON COLUMN storage.files.updated_by IS
'Who last touched the file (rename, move, overwrite, restore). Same '
'write-path discipline as updated_at.';
-- ── 4. role_grants resource_type CHECK — admit 'drive' ─────────────────────
-- The D-Prep migration's CHECK only listed 'folder' and 'file'. Drives
-- need to be a valid resource_type so the lifecycle hook (D0-9) and the
-- membership API (D2) can write `role_grants` rows with
-- resource_type='drive'.
ALTER TABLE storage.role_grants
DROP CONSTRAINT IF EXISTS role_grants_resource_type_check;
ALTER TABLE storage.role_grants
ADD CONSTRAINT role_grants_resource_type_check
CHECK (resource_type IN ('folder', 'file', 'drive'));
-- ── 5. updated_at trigger for storage.drives ───────────────────────────────
-- Mirror the convention from auth.users / storage.folders / storage.files
-- so rename / quota-change / policy-toggle bumps updated_at automatically.
-- Drive owners shouldn't have to remember to maintain this.
CREATE OR REPLACE FUNCTION storage.drives_touch_updated_at()
RETURNS trigger AS $$
BEGIN
NEW.updated_at := now();
RETURN NEW;
END;
$$ LANGUAGE plpgsql;
DROP TRIGGER IF EXISTS trg_drives_touch_updated_at ON storage.drives;
CREATE TRIGGER trg_drives_touch_updated_at
BEFORE UPDATE ON storage.drives
FOR EACH ROW EXECUTE FUNCTION storage.drives_touch_updated_at();
@@ -0,0 +1,355 @@
-- ════════════════════════════════════════════════════════════════════════════
-- D0 / M2 — Drive backfill: create drives + stamp drive_id + provenance
-- ════════════════════════════════════════════════════════════════════════════
-- Second of the D0 migration trio. Half (1) of the §A backfill — the safe,
-- focused half:
--
-- * For every internal user with a root folder, create a Personal drive.
-- * The folder literally named `My Folder - <username>` becomes the
-- user's default Personal drive (`default_for_user = <uid>`).
-- * Any sibling root folders become secondary Personal drives
-- (`default_for_user = NULL`, name carried over verbatim).
-- * One owner role_grants row per new drive.
-- * Every existing folder/file row gets a `drive_id` (cascaded down the
-- ltree from the wrapper).
-- * Every existing folder/file row gets `created_by` and `updated_by`
-- backfilled from the existing `user_id` column.
--
-- The aggressive half (drop the wrapper folder, rewrite path columns,
-- strip the `My Folder - <username>/` prefix from every path/lpath value)
-- lands in M2b — kept separate so the tree-shape rewrite can be reviewed
-- in isolation.
--
-- External users (`auth.users.is_external = TRUE`) are intentionally
-- skipped — they have no root folder of their own, only role_grants
-- against other users' resources.
-- ── Pre-flight 1: refuse on sibling root literally named 'drives' ──────────
-- 'drives' is a reserved URL segment on the native WebDAV surface
-- (`/webdav/drives/<uuid>/`). A folder named 'drives' would shadow the
-- drive-listing route once D1 ships. Surface the conflict now — operator
-- renames before retrying.
DO $BODY$
DECLARE
bad_count BIGINT;
BEGIN
SELECT count(*) INTO bad_count
FROM storage.folders f
JOIN auth.users u ON u.id = f.user_id
WHERE f.parent_id IS NULL
AND NOT f.is_trashed
AND lower(f.name) = 'drives'
AND NOT u.is_external;
IF bad_count > 0 THEN
RAISE EXCEPTION
'D0 backfill refused: % root folder(s) literally named ''drives'' '
'would collide with the reserved /webdav/drives/<uuid>/ URL segment '
'once D1 ships. Rename the offending folders, then retry the '
'migration. Query to inspect: SELECT f.id, f.user_id, f.name '
'FROM storage.folders f JOIN auth.users u ON u.id = f.user_id '
'WHERE f.parent_id IS NULL AND NOT f.is_trashed AND lower(f.name) '
'= ''drives'' AND NOT u.is_external;',
bad_count;
END IF;
END $BODY$;
-- ── Pre-flight 2: report sibling-root distribution (informational) ─────────
-- Most users have exactly one root (`My Folder - <username>`). Some may
-- have SQL-added siblings — those become secondary drives. Surface the
-- count so operators can sanity-check before the migration commits.
DO $BODY$
DECLARE
extras BIGINT;
BEGIN
WITH counts AS (
SELECT u.id AS user_id, count(*) AS root_count
FROM auth.users u
JOIN storage.folders f ON f.user_id = u.id
WHERE f.parent_id IS NULL
AND NOT f.is_trashed
AND NOT u.is_external
GROUP BY u.id
)
SELECT count(*) INTO extras FROM counts WHERE root_count > 1;
IF extras > 0 THEN
RAISE NOTICE
'D0 backfill: % user(s) have more than one root folder. Their '
'siblings will be promoted to secondary Personal drives. Inspect '
'with: WITH c AS (SELECT u.id, u.username, count(*) cnt FROM '
'auth.users u JOIN storage.folders f ON f.user_id=u.id WHERE '
'f.parent_id IS NULL AND NOT f.is_trashed AND NOT u.is_external '
'GROUP BY u.id, u.username) SELECT * FROM c WHERE cnt > 1;',
extras;
END IF;
END $BODY$;
-- ── 1. Plan every drive that needs to be created ──────────────────────────
-- Temp table is the cleanest way to pre-compute the new UUIDs once and
-- reuse them across the INSERT-drives, INSERT-grants, and UPDATE-folders
-- steps below. `gen_random_uuid()` in a CTE would re-evaluate on every
-- branch.
--
-- A row joins each existing root folder to its future drive_id. The
-- `is_default` flag is computed per-user as a window function so EVERY
-- internal user with at least one root folder ends up with exactly one
-- default drive — even if the user's wrapper was renamed away from
-- `My Folder - <username>` at some point. Preference order:
-- 1. The folder literally named `My Folder - <username>` if it exists.
-- 2. Otherwise the oldest root by `created_at`, tiebroken by `id`.
-- `ON COMMIT DROP` would race with the `[init-schema]` CI flow that
-- runs migrations via `psql \i` in autocommit mode: the CREATE statement
-- commits, the table drops, and the next statement (the DO block) can't
-- see it. The plain temp table survives until session end in autocommit
-- mode and until our explicit DROP at the bottom under `sqlx migrate`'s
-- single-tx mode. Works under both.
CREATE TEMPORARY TABLE _drive_plan AS
WITH root_folders AS (
SELECT
u.id AS user_id,
u.username AS username,
u.storage_quota_bytes AS quota,
f.id AS wrapper_id,
f.name AS wrapper_name,
f.created_at AS created_at,
(f.name = 'My Folder - ' || u.username) AS name_matches_default
FROM auth.users u
JOIN storage.folders f
ON f.user_id = u.id
AND f.parent_id IS NULL
AND NOT f.is_trashed
WHERE NOT u.is_external
)
SELECT
user_id,
username,
quota,
wrapper_id,
wrapper_name,
-- Rank candidates per user: name-matched root wins; otherwise oldest
-- by created_at then by id (stable, deterministic). ROW_NUMBER() = 1
-- becomes the default drive for that user.
(ROW_NUMBER() OVER (
PARTITION BY user_id
ORDER BY name_matches_default DESC,
created_at ASC,
wrapper_id ASC
) = 1) AS is_default,
gen_random_uuid() AS new_drive_id
FROM root_folders;
-- ── 1b. Log which users got an auto-picked default (no name match) ────────
-- Operational nicety: if a user's default came from oldest-root fallback
-- rather than the canonical `My Folder - <username>`, surface it so an
-- operator can DM the user and confirm the migration picked the right
-- root. Not a failure — just visibility.
DO $BODY$
DECLARE
auto_picked BIGINT;
BEGIN
SELECT count(*) INTO auto_picked
FROM _drive_plan p
WHERE p.is_default
AND p.wrapper_name <> 'My Folder - ' || p.username;
IF auto_picked > 0 THEN
RAISE NOTICE
'D0 backfill: % user(s) had no `My Folder - <username>` root; '
'the oldest sibling root was auto-picked as their default '
'Personal drive. Inspect with: SELECT user_id, username, '
'wrapper_name FROM _drive_plan WHERE is_default AND '
'wrapper_name <> ''My Folder - '' || username; '
'(temp table only exists during the migration transaction.)',
auto_picked;
END IF;
END $BODY$;
-- ── 2. Insert the drive rows ───────────────────────────────────────────────
-- Default drives carry the i18n-neutral name 'Personal' (renameable
-- later via the drive settings panel). Secondary drives carry their
-- original folder name verbatim.
INSERT INTO storage.drives
(id, name, kind, default_for_user, quota_bytes)
SELECT
p.new_drive_id,
CASE WHEN p.is_default THEN 'Personal' ELSE p.wrapper_name END,
'personal',
CASE WHEN p.is_default THEN p.user_id ELSE NULL END,
p.quota
FROM _drive_plan p;
-- ── 3. Insert one owner role_grants row per drive ─────────────────────────
-- Each user is the sole owner of every drive their wrappers produced.
-- The lifecycle hook (D0-9) will do the same for users created post-D0.
INSERT INTO storage.role_grants
(subject_type, subject_id, resource_type, resource_id, role, granted_by)
SELECT 'user', p.user_id, 'drive', p.new_drive_id, 'owner', p.user_id
FROM _drive_plan p;
-- ── 4. Stamp drive_id on each wrapper folder ──────────────────────────────
-- The wrapper still exists as a folder during M2 (the wrapper-drop lives
-- in M2b). Setting drive_id on the wrapper lets the cascade in §5 walk
-- the ltree subtree without needing a separate index.
UPDATE storage.folders f
SET drive_id = p.new_drive_id
FROM _drive_plan p
WHERE f.id = p.wrapper_id;
-- ── 5. Cascade drive_id down the folder tree ──────────────────────────────
-- For every folder descended from a wrapper, set drive_id to that
-- wrapper's. Uses the existing GiST index `idx_folders_lpath` for the
-- @> (ancestor-of) lookup. Trashed descendants get a drive_id too —
-- soft-deleted folders need a drive_id once M3 makes the column NOT NULL.
UPDATE storage.folders sub
SET drive_id = wrapper.drive_id
FROM storage.folders wrapper
WHERE wrapper.id IN (SELECT wrapper_id FROM _drive_plan)
AND sub.lpath <@ wrapper.lpath
AND sub.id != wrapper.id
AND sub.drive_id IS NULL;
-- ── 6. Cascade drive_id to files (via their folder) ───────────────────────
-- Files inherit drive_id from their containing folder. A NULL folder_id
-- file is an orphan — left with NULL drive_id here; M3's NOT NULL
-- constraint will refuse the migration if any such orphans remain,
-- which is the right outcome (forces operator inspection).
UPDATE storage.files fi
SET drive_id = fo.drive_id
FROM storage.folders fo
WHERE fi.folder_id = fo.id
AND fi.drive_id IS NULL
AND fo.drive_id IS NOT NULL;
-- ── 7. Provenance backfill ────────────────────────────────────────────────
-- Every pre-Drive row carries authentic provenance from day one: created_by
-- and updated_by both default to the user_id that we know created the
-- resource (that's exactly what user_id meant pre-D0). New writes during
-- the dual-write window populate both columns explicitly.
UPDATE storage.folders
SET created_by = user_id,
updated_by = user_id
WHERE created_by IS NULL;
UPDATE storage.files
SET created_by = user_id,
updated_by = user_id
WHERE created_by IS NULL;
-- ── 7b. Drop the planning temp table ──────────────────────────────────────
-- Explicit drop since we removed `ON COMMIT DROP` above. Idempotent
-- (`IF EXISTS`) so a partial re-run during development doesn't error.
DROP TABLE IF EXISTS _drive_plan;
-- ── 8. Post-flight consistency check ──────────────────────────────────────
-- The checks here REFUSE to commit if any invariant is violated, so a
-- successful migration is a verifiable migration.
--
-- 8a. Every internal user with a root folder has exactly one
-- default drive.
-- 8b. Every drive has at least one owner role_grants row.
-- 8c. No NULL drive_id remains on a folder/file row whose owner is
-- a non-external user with a root folder (i.e. every row that
-- belongs to a drive must now declare which one).
--
-- M3 turns drive_id NOT NULL; the check below is a stricter pre-flight
-- so the failure mode is "migration refuses" rather than "M3 errors
-- with a NOT NULL violation halfway through".
DO $BODY$
DECLARE
missing_default BIGINT;
grantless_drives BIGINT;
null_folder_drive_id BIGINT;
null_file_drive_id BIGINT;
BEGIN
SELECT count(*) INTO missing_default
FROM auth.users u
WHERE NOT u.is_external
AND EXISTS (
SELECT 1 FROM storage.folders f
WHERE f.user_id = u.id AND f.parent_id IS NULL AND NOT f.is_trashed
)
AND NOT EXISTS (
SELECT 1 FROM storage.drives d
WHERE d.default_for_user = u.id
);
IF missing_default > 0 THEN
RAISE EXCEPTION
'D0 backfill consistency check failed: % internal user(s) with a '
'root folder have no default Personal drive. Investigate before '
'declaring the migration successful.',
missing_default;
END IF;
SELECT count(*) INTO grantless_drives
FROM storage.drives d
WHERE NOT EXISTS (
SELECT 1 FROM storage.role_grants g
WHERE g.resource_type = 'drive'
AND g.resource_id = d.id
AND g.role = 'owner'
);
IF grantless_drives > 0 THEN
RAISE EXCEPTION
'D0 backfill consistency check failed: % drive(s) have no owner '
'role_grants row. Investigate before declaring the migration '
'successful.',
grantless_drives;
END IF;
SELECT count(*) INTO null_folder_drive_id
FROM storage.folders f
WHERE f.drive_id IS NULL
AND EXISTS (
SELECT 1 FROM auth.users u
WHERE u.id = f.user_id AND NOT u.is_external
);
IF null_folder_drive_id > 0 THEN
RAISE EXCEPTION
'D0 backfill consistency check failed: % folder(s) belonging to '
'an internal user still have NULL drive_id. M3 will refuse to '
'add NOT NULL until these are resolved.',
null_folder_drive_id;
END IF;
SELECT count(*) INTO null_file_drive_id
FROM storage.files fi
WHERE fi.drive_id IS NULL
AND EXISTS (
SELECT 1 FROM auth.users u
WHERE u.id = fi.user_id AND NOT u.is_external
);
IF null_file_drive_id > 0 THEN
RAISE EXCEPTION
'D0 backfill consistency check failed: % file(s) belonging to an '
'internal user still have NULL drive_id (likely orphans — '
'folder_id pointing at a missing folder). Inspect with: SELECT '
'fi.id, fi.user_id, fi.folder_id FROM storage.files fi JOIN '
'auth.users u ON u.id = fi.user_id WHERE fi.drive_id IS NULL '
'AND NOT u.is_external;',
null_file_drive_id;
END IF;
END $BODY$;
@@ -0,0 +1,108 @@
-- ════════════════════════════════════════════════════════════════════════════
-- D0 / M3 — Drive constraints: NOT NULL, FK, indexes
-- ════════════════════════════════════════════════════════════════════════════
-- Third of the D0 migration trio. Runs only after M2 (the backfill) has
-- populated every folder/file row with a `drive_id`. This migration is
-- the point of no easy rollback — once `drive_id` is NOT NULL and
-- foreign-keyed to `storage.drives`, dropping the column requires the
-- application code to first stop reading it.
--
-- What lands here:
-- * NOT NULL on `storage.folders.drive_id` and `storage.files.drive_id`.
-- * Foreign keys from both to `storage.drives(id)` with ON DELETE
-- CASCADE (deleting a drive removes its tree — matches the
-- post-D2 lifecycle plan).
-- * Indexes on `drive_id` for both tables (hot path: list-by-drive,
-- drive-aware Tantivy reseed, drive-quota counters).
--
-- The `user_id` column is intentionally left in place: dual-write during
-- the D0 release cycle is the rollback safety net. D7 drops user_id once
-- the new model has baked.
-- ── 1. NOT NULL on drive_id ────────────────────────────────────────────────
-- M2's post-flight check refused to commit if any row was missing
-- drive_id, so this should never fail. The check at column promotion
-- time is the belt; M2's pre-commit assertion was the suspenders.
ALTER TABLE storage.folders
ALTER COLUMN drive_id SET NOT NULL;
ALTER TABLE storage.files
ALTER COLUMN drive_id SET NOT NULL;
-- ── 2. Foreign keys to storage.drives ──────────────────────────────────────
-- ON DELETE CASCADE: when a drive is deleted (D3 ships the delete-drive
-- flow), every folder and file row carrying that drive_id is removed in
-- the same transaction. Trash retention does not apply — drive deletion
-- is the explicit "I'm done with this storage" gesture.
ALTER TABLE storage.folders
ADD CONSTRAINT folders_drive_id_fkey
FOREIGN KEY (drive_id) REFERENCES storage.drives(id) ON DELETE CASCADE;
ALTER TABLE storage.files
ADD CONSTRAINT files_drive_id_fkey
FOREIGN KEY (drive_id) REFERENCES storage.drives(id) ON DELETE CASCADE;
-- ── 3. Indexes on drive_id ─────────────────────────────────────────────────
-- The hot path that ranks every drive-aware query: "list folders in
-- drive X", "files in drive X for Tantivy reindex", "per-drive quota
-- aggregation". The existing `user_id` indexes are kept during dual-
-- write and dropped in D7 alongside the column.
CREATE INDEX IF NOT EXISTS idx_folders_drive_id ON storage.folders (drive_id);
CREATE INDEX IF NOT EXISTS idx_files_drive_id ON storage.files (drive_id);
-- ── 4. Post-flight: confirm constraints landed ────────────────────────────
-- Belt-and-suspenders verification that the NOT NULL + FK actually
-- exist after the ALTERs above. Any failure here means PostgreSQL
-- silently no-op'd one of the constraint changes, which would be a
-- bug worth surfacing immediately.
DO $BODY$
DECLARE
folder_not_null BOOLEAN;
file_not_null BOOLEAN;
folder_fk_exists BOOLEAN;
file_fk_exists BOOLEAN;
BEGIN
SELECT NOT is_nullable::boolean INTO folder_not_null
FROM information_schema.columns
WHERE table_schema = 'storage'
AND table_name = 'folders'
AND column_name = 'drive_id';
SELECT NOT is_nullable::boolean INTO file_not_null
FROM information_schema.columns
WHERE table_schema = 'storage'
AND table_name = 'files'
AND column_name = 'drive_id';
SELECT EXISTS (
SELECT 1 FROM information_schema.table_constraints
WHERE table_schema = 'storage'
AND table_name = 'folders'
AND constraint_name = 'folders_drive_id_fkey'
AND constraint_type = 'FOREIGN KEY'
) INTO folder_fk_exists;
SELECT EXISTS (
SELECT 1 FROM information_schema.table_constraints
WHERE table_schema = 'storage'
AND table_name = 'files'
AND constraint_name = 'files_drive_id_fkey'
AND constraint_type = 'FOREIGN KEY'
) INTO file_fk_exists;
IF NOT folder_not_null OR NOT file_not_null
OR NOT folder_fk_exists OR NOT file_fk_exists THEN
RAISE EXCEPTION
'D0 M3 post-flight failed: '
'folder NOT NULL=%, file NOT NULL=%, folder FK=%, file FK=%. '
'All four must be true after this migration commits.',
folder_not_null, file_not_null, folder_fk_exists, file_fk_exists;
END IF;
END $BODY$;
@@ -0,0 +1,148 @@
-- ════════════════════════════════════════════════════════════════════════════
-- D0 / M4 — tree_etag_dirty drive_id awareness
-- ════════════════════════════════════════════════════════════════════════════
-- Fourth D0 migration. The async tree-ETag queue (introduced in
-- `20260627000000_async_tree_etag_queue.sql`) walks `f.lpath @> t.lpath`
-- to bump every ancestor of a changed folder/file. Without a drive_id
-- filter, that walk can match folders in OTHER drives whose ltree
-- prefixes happen to align numerically — a silent cross-drive ETag
-- bump that nobody would notice until D2 ships shared drives.
--
-- This migration is preventive: it adds the column, teaches the
-- triggers to carry drive_id into the queue, and the Rust flush
-- service (`tree_etag_flush_service.rs`) gets the matching
-- `AND f.drive_id = t.drive_id` predicate so the cross-drive case is
-- closed end-to-end before drives can collide.
-- ── 1. drive_id column on the queue table ──────────────────────────────────
-- NULL-tolerant during the rollover: existing queue entries enqueued by
-- the old triggers have no drive_id. They drain on the next flush tick
-- with the old (no drive_id) semantics — which for D0 is still correct
-- because every personal drive's lpath is structurally disjoint from
-- every other user's. New entries enqueued by the updated triggers
-- below carry a non-NULL value.
ALTER TABLE storage.tree_etag_dirty ADD COLUMN IF NOT EXISTS drive_id UUID;
-- ── 2. File-side INSERT/DELETE trigger ─────────────────────────────────────
-- Source rows live in `storage.files` (changed_rows). Each file row
-- carries `drive_id` directly (D0-8 dual-write). Pull from the joined
-- folder row so the (lpath, folder_id, drive_id) triple is internally
-- consistent — a single source of truth per enqueued row.
CREATE OR REPLACE FUNCTION storage.bump_tree_from_files_stmt()
RETURNS TRIGGER LANGUAGE plpgsql AS $$
BEGIN
IF pg_trigger_depth() > 1 THEN
RETURN NULL;
END IF;
INSERT INTO storage.tree_etag_dirty (lpath, folder_id, drive_id)
SELECT DISTINCT fo.lpath, fo.id, fo.drive_id
FROM (SELECT DISTINCT folder_id
FROM changed_rows
WHERE folder_id IS NOT NULL) c
JOIN storage.folders fo ON fo.id = c.folder_id;
RETURN NULL;
END;
$$;
-- ── 3. File-side UPDATE trigger ────────────────────────────────────────────
-- Move case: the file changed parent. Bump both the old and the new
-- parent chains (each in its own drive — D0 has them equal, D2 can see
-- them diverge once cross-drive moves land).
CREATE OR REPLACE FUNCTION storage.bump_tree_from_files_stmt_upd()
RETURNS TRIGGER LANGUAGE plpgsql AS $$
BEGIN
IF pg_trigger_depth() > 1 THEN
RETURN NULL;
END IF;
WITH changed AS (
SELECT o.folder_id AS old_folder_id, n.folder_id AS new_folder_id
FROM old_rows o
JOIN new_rows n USING (id)
WHERE (o.name, o.folder_id, o.blob_hash, o.size,
o.mime_type, o.is_trashed, o.updated_at)
IS DISTINCT FROM
(n.name, n.folder_id, n.blob_hash, n.size,
n.mime_type, n.is_trashed, n.updated_at)
)
INSERT INTO storage.tree_etag_dirty (lpath, folder_id, drive_id)
SELECT DISTINCT fo.lpath, fo.id, fo.drive_id
FROM (SELECT old_folder_id AS folder_id
FROM changed WHERE old_folder_id IS NOT NULL
UNION
SELECT new_folder_id
FROM changed WHERE new_folder_id IS NOT NULL) c
JOIN storage.folders fo ON fo.id = c.folder_id;
RETURN NULL;
END;
$$;
-- ── 4. Folder-side INSERT/DELETE trigger ──────────────────────────────────
-- changed_rows are storage.folders rows, which carry drive_id directly
-- post-D0-7.
CREATE OR REPLACE FUNCTION storage.bump_tree_from_folders_stmt()
RETURNS TRIGGER LANGUAGE plpgsql AS $$
BEGIN
IF pg_trigger_depth() > 1 THEN
RETURN NULL;
END IF;
INSERT INTO storage.tree_etag_dirty (lpath, folder_id, drive_id)
SELECT DISTINCT subpath(lpath, 0, nlevel(lpath) - 1), parent_id, drive_id
FROM changed_rows
WHERE lpath IS NOT NULL
AND nlevel(lpath) > 1;
RETURN NULL;
END;
$$;
-- ── 5. Folder-side UPDATE trigger ──────────────────────────────────────────
-- Same shape as the INSERT/DELETE case; we union OLD and NEW parents,
-- carrying each chain's drive_id from the matching row side.
CREATE OR REPLACE FUNCTION storage.bump_tree_from_folders_stmt_upd()
RETURNS TRIGGER LANGUAGE plpgsql AS $$
BEGIN
IF pg_trigger_depth() > 1 THEN
RETURN NULL;
END IF;
WITH changed AS (
SELECT o.lpath AS old_lpath,
o.parent_id AS old_parent_id,
o.drive_id AS old_drive_id,
n.lpath AS new_lpath,
n.parent_id AS new_parent_id,
n.drive_id AS new_drive_id
FROM old_rows o
JOIN new_rows n USING (id)
WHERE (o.name, o.parent_id, o.is_trashed, o.updated_at)
IS DISTINCT FROM
(n.name, n.parent_id, n.is_trashed, n.updated_at)
)
INSERT INTO storage.tree_etag_dirty (lpath, folder_id, drive_id)
SELECT DISTINCT subpath(c.lpath, 0, nlevel(c.lpath) - 1), c.parent_id, c.drive_id
FROM (SELECT old_lpath AS lpath,
old_parent_id AS parent_id,
old_drive_id AS drive_id
FROM changed WHERE old_lpath IS NOT NULL
UNION
SELECT new_lpath, new_parent_id, new_drive_id
FROM changed WHERE new_lpath IS NOT NULL) c
WHERE nlevel(c.lpath) > 1;
RETURN NULL;
END;
$$;
@@ -0,0 +1,133 @@
-- ════════════════════════════════════════════════════════════════════════════
-- D0 / M5 — storage.copy_folder_tree drive_id + provenance
-- ════════════════════════════════════════════════════════════════════════════
-- The `copy_folder_tree` SQL function (initial_schema.sql) batches a
-- recursive folder copy in PL/pgSQL — its own INSERTs into
-- `storage.folders` and `storage.files`. D0's M3 made `drive_id` NOT
-- NULL on both tables; the function's pre-D0 body doesn't write it,
-- so any `/api/batch/folders/copy` call errors with "null value in
-- column drive_id" until this migration lands.
--
-- The replacement preserves every other semantic of the original:
-- - level-by-level INSERTs so the BEFORE INSERT trigger
-- (`trg_folders_path`) can resolve the parent's path/lpath from
-- rows inserted in the previous level.
-- - One batched file INSERT (zero-copy via blob hash) at the end.
-- - Returns the same shape: (new_root_id::text, folders_copied,
-- files_copied).
--
-- New columns written:
-- - drive_id: pulled from the source row (intra-drive copy — cross-
-- drive copies are a D2+ feature; the function preserves the
-- source's drive_id for both folders and files).
-- - created_by / updated_by: set to the source row's user_id,
-- matching the dual-write convention used by the Rust repos.
CREATE OR REPLACE FUNCTION storage.copy_folder_tree(
p_source_id UUID,
p_target_parent_id UUID, -- NULL = copy to root
p_dest_name TEXT DEFAULT NULL -- NULL = keep source folder name
) RETURNS TABLE(new_root_id TEXT, folders_copied BIGINT, files_copied BIGINT) AS $$
DECLARE
v_root_lpath ltree;
v_root_depth INT;
v_max_depth INT;
v_level INT;
v_folders BIGINT := 0;
v_files BIGINT := 0;
v_inserted BIGINT;
v_new_root UUID;
BEGIN
-- Validate source exists
SELECT fo.lpath, nlevel(fo.lpath)
INTO v_root_lpath, v_root_depth
FROM storage.folders fo
WHERE fo.id = p_source_id AND NOT fo.is_trashed;
IF v_root_lpath IS NULL THEN
RAISE EXCEPTION 'Source folder not found: %', p_source_id
USING ERRCODE = 'P0002'; -- no_data_found
END IF;
-- Temp mapping: every folder in the subtree → new UUID
CREATE TEMP TABLE IF NOT EXISTS _copy_map(
old_id UUID PRIMARY KEY,
new_id UUID NOT NULL DEFAULT gen_random_uuid()
) ON COMMIT DROP;
TRUNCATE _copy_map;
INSERT INTO _copy_map(old_id)
SELECT fo.id
FROM storage.folders fo
WHERE NOT fo.is_trashed
AND fo.lpath <@ v_root_lpath;
-- Remember new root ID
SELECT cm.new_id INTO v_new_root
FROM _copy_map cm WHERE cm.old_id = p_source_id;
-- Max depth for level iteration
SELECT MAX(nlevel(fo.lpath))
INTO v_max_depth
FROM storage.folders fo
JOIN _copy_map cm ON fo.id = cm.old_id;
-- ── Insert folders level by level ──
-- Each level is a separate INSERT so the BEFORE INSERT trigger
-- (trg_folders_path) can resolve the parent's path/lpath from rows
-- inserted in the previous level. drive_id + provenance threaded
-- through from the source row at each level.
FOR v_level IN v_root_depth .. v_max_depth LOOP
INSERT INTO storage.folders(
id, name, parent_id, user_id,
drive_id, created_by, updated_by
)
SELECT cm.new_id,
CASE WHEN fo.id = p_source_id AND p_dest_name IS NOT NULL
THEN p_dest_name ELSE fo.name END,
CASE WHEN fo.id = p_source_id THEN p_target_parent_id
ELSE pm.new_id END,
fo.user_id,
fo.drive_id,
fo.user_id,
fo.user_id
FROM storage.folders fo
JOIN _copy_map cm ON fo.id = cm.old_id
LEFT JOIN _copy_map pm ON fo.parent_id = pm.old_id
WHERE NOT fo.is_trashed
AND nlevel(fo.lpath) = v_level;
GET DIAGNOSTICS v_inserted = ROW_COUNT;
v_folders := v_folders + v_inserted;
END LOOP;
-- ── Batch copy all files (zero-copy: same blob_hash) ──
INSERT INTO storage.files(
name, folder_id, user_id, blob_hash, size, mime_type,
media_sort_date, drive_id, created_by, updated_by
)
SELECT f.name, cm.new_id, f.user_id, f.blob_hash, f.size, f.mime_type,
f.media_sort_date, f.drive_id, f.user_id, f.user_id
FROM storage.files f
JOIN _copy_map cm ON f.folder_id = cm.old_id
WHERE NOT f.is_trashed;
GET DIAGNOSTICS v_files = ROW_COUNT;
-- ── Batch increment blob ref_counts ──
IF v_files > 0 THEN
UPDATE storage.blobs b
SET ref_count = ref_count + hc.cnt
FROM (
SELECT f.blob_hash, COUNT(*)::int AS cnt
FROM storage.files f
JOIN _copy_map cm ON f.folder_id = cm.new_id
WHERE NOT f.is_trashed
GROUP BY f.blob_hash
) hc
WHERE b.hash = hc.blob_hash;
END IF;
RETURN QUERY SELECT v_new_root::text, v_folders, v_files;
END;
$$ LANGUAGE plpgsql;