From c8ef9c10291859ac86e93894ae1627c500c2a71e Mon Sep 17 00:00:00 2001 From: alice Date: Fri, 28 Aug 2026 22:20:53 +0000 Subject: [PATCH 1/7] =?UTF-8?q?vms-6ca:=20DLM=20H5=20executive=20core=20?= =?UTF-8?q?=E2=80=94=20requester-side=20GRANT=20RECEIVE=20+=20deferred-gra?= =?UTF-8?q?nt=20report?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The cross-node contention semantics (#873) proved block-then-grant on a single real /dev/vms. This adds the two executive halves the async-reply SCS wire needs: - MASTER side: vms_lock_dlm_xnode_deq now REPORTS the queued cross-node waiter that a release flipped to granted (its requester CSID + original req_lkid + master handle + granted mode), through the fields a DEQ otherwise leaves 0, so the daemon can WIRE a deferred GRANT to the requester without a second probe. The master lock entry now carries the requester's own lock handle (req_lkid). - REQUESTER side: VMS_DLM_OP_GRANT is implemented (was SS$_UNSUPPORTED). A GRANT / queued-reply the master sent back completes an executive-resident ORIGIN record — the requester-side proxy of the outstanding request, on its own list, never touched by the local lock manager. Its granted mode is set ONLY from what the master genuinely sent over SCS: a queued-reply (mode NL) leaves it pending, a deferred GRANT (mode EX) flips it NL->EX. GETLKI falls through to it, so the status flip is observable on the REQUESTER node. INV-6: genuine executive state, no fabricated grant. BLKAST as a receive op stays SS$_UNSUPPORTED — the BLKAST wire is deferred honestly on this rung (the block-then-grant round-trip is proven without it). test_syssvc_dlm_xnode.c extended to prove both halves against a real /dev/vms (x86_64 + Alpha LP64): the deferred-grant report on the block-then-grant $DEQ, and the requester-side queued->granted origin flip. vms.ko builds clean. Co-Authored-By: Claude Opus 4.8 (1M context) --- src/kernel-core/vms_lock.c | 245 ++++++++++++++++++++++++++++- src/kernel-netbsd/vms_internal.h | 7 + src/kernel/vms_internal.h | 8 + src/libvmssys/vms_kif.c | 11 +- src/libvmssys/vms_kif.h | 3 +- tests/qemu/test_syssvc_dlm_xnode.c | 105 ++++++++++--- 6 files changed, 354 insertions(+), 25 deletions(-) diff --git a/src/kernel-core/vms_lock.c b/src/kernel-core/vms_lock.c index c5c5d4a86..5e1cb4581 100644 --- a/src/kernel-core/vms_lock.c +++ b/src/kernel-core/vms_lock.c @@ -80,12 +80,52 @@ uint32_t vms_next_lock_id = 1; EXEC_DEFINE_HASHTABLE(vms_res_hash, VMS_RES_HASH_BITS); exec_lock_t vms_res_hash_lock; +/* ================================================================ + * Cross-node REQUESTER-SIDE origin records (DLM epic vms-7fa rung H5, vms-6ca) + * ================================================================ + * + * When THIS node issues a cross-node $ENQ over SCS to a REMOTE master, the + * request is not a local lock (the local lock manager never grants it -- the + * remote master does). But the requester still needs a REAL, executive-resident + * record of the outstanding request so its completion is genuine state, not a + * per-process userspace fake (INV-6): the origin record's granted mode is set + * ONLY from GRANT/queued-reply messages the master genuinely sent back over the + * live SCS wire (VMS_DLM_OP_GRANT receive), never by a local grant decision. + * + * This is the requester-side mirror of the master's lock block. It lives on its + * OWN list, keyed by the requester's local lock handle (req_lkid), so it never + * touches the resource granted/waiting queues the local lock manager scans -- + * the executive cannot auto-grant an origin record, only the wire can complete + * it. GETLKI falls through to it so the NL->EX status flip driven by the remote + * master's deferred GRANT is observable on the REQUESTER node. + */ +struct vms_dlm_origin { + exec_list_node_t list; + uint32_t req_lkid; /* our own lock handle for the request */ + uint32_t req_csid; /* our node's CSID (the requester) */ + uint32_t master_lkid; /* the master's lock handle (0 until known) */ + uint32_t master_csid; /* the mastering node's CSID */ + uint32_t granted_mode; /* NL while pending/queued; the granted mode + * once the master's GRANT arrives */ + uint32_t requested_mode; /* the mode we asked for */ + char resnam[32]; +}; + +exec_list_head_t vms_dlm_origin_list; +exec_lock_t vms_dlm_origin_lock; + +static int vms_lock_dlm_origin_getlki(uint32_t lkid, uint32_t *granted_mode, + uint32_t *requested_mode, char *resnam, + size_t resnam_len); + int vms_lock_init(void) { exec_lock_init(&vms_lock_id_lock); exec_lock_init(&vms_res_hash_lock); + exec_lock_init(&vms_dlm_origin_lock); exec_rbtree_init(&vms_lock_id_tree); exec_hash_init(vms_res_hash); + exec_list_head_init(&vms_dlm_origin_list); return 0; } @@ -141,9 +181,21 @@ void vms_lock_cleanup(void) } exec_unlock(&vms_res_hash_lock); + /* Free any cross-node requester-side origin records (vms-6ca, H5). */ + { + struct vms_dlm_origin *org, *otmp; + exec_lock(&vms_dlm_origin_lock); + exec_list_for_each_entry_safe(org, otmp, &vms_dlm_origin_list, list) { + exec_list_del(&org->list); + exec_free(org); + } + exec_unlock(&vms_dlm_origin_lock); + } + /* Tear down the runtime-initialized locks (a no-op on Linux; a real * mutex_destroy on NetBSD -- paired with the exec_lock_init in * vms_lock_init). */ + exec_lock_destroy(&vms_dlm_origin_lock); exec_lock_destroy(&vms_res_hash_lock); exec_lock_destroy(&vms_lock_id_lock); } @@ -843,6 +895,9 @@ static int enq_wait_sync(struct vms_lock_resource *res, * that holder's master lock handle (blocking_master_lkid). */ struct dlm_xnode_enq_out { + uint32_t req_lkid; /* IN: the remote requester's own lock handle, + * stored on the master lock so a deferred + * GRANT can name the original request (H5). */ int queued; uint32_t blocking_csid; uint32_t blocking_master_lkid; @@ -932,6 +987,13 @@ static long vms_enq_core_ex(struct vms_proc *proc, struct vms_enq_args *io, * fabricated status (INV-6). */ lock->req_csid = args.owner_csid; + /* + * The remote requester's OWN lock handle (vms-6ca, H5). Non-zero only on the + * cross-node path (xn set), so that when this granted lock is later released + * and its release grants a queued cross-node waiter, the master can name the + * waiter's original request (req_lkid) in the deferred GRANT it wires back. + */ + lock->req_lkid = xn ? xn->req_lkid : 0; /* * Record the access mode $ENQ was issued from, so image rundown can tell @@ -1373,6 +1435,23 @@ long vms_ioctl_getlki(struct vms_proc *proc, unsigned long arg) lock = lock_find_by_id(args.lkid); if (!lock) { + /* + * Fall through to the requester-side cross-node ORIGIN records (vms-6ca, + * H5): a cross-node request THIS node issued is not a local lock (the + * remote master holds it), so it has no entry in the lock-ID tree, but + * its origin record carries the status the master's GRANT completed. This + * makes the NL->EX flip driven by the remote master observable here. + */ + uint32_t org_granted = 0, org_requested = 0; + if (vms_lock_dlm_origin_getlki(args.lkid, &org_granted, &org_requested, + args.resnam, sizeof(args.resnam))) { + args.granted_mode = org_granted; + args.requested_mode = org_requested; + args.parent_id = 0; + memset(args.valblk, 0, LCK_VALBLK_SIZE); + args.status = SS__NORMAL; + goto out; + } args.status = SS__IVLOCKID; goto out; } @@ -1506,10 +1585,60 @@ static uint32_t vms_lock_dlm_xnode_deq(struct vms_dlm_xnode_args *req) exec_list_del(&lock->res_waiting); else exec_list_del(&lock->res_granted); + + /* + * Deferred-GRANT discovery (vms-6ca, DLM epic vms-7fa rung H5). Snapshot the + * FIRST cross-node waiter (held FOR a remote CSID) BEFORE the release, so + * after try_grant_waiters we can tell whether THIS release flipped it from + * queued to granted -- the block-then-grant transition the master must now + * WIRE back to that requester as a deferred GRANT. We report the flipped + * waiter's identity (its requester CSID + req_lkid + master lock handle + + * granted mode) through the fields that are 0 on a DEQ, so the daemon that + * delivered this $DEQ can transport the GRANT without a second round-trip. + * First waiter only -- the same "first blocking holder" scope the contention + * rung uses (vms-904c); several flipped waiters is a refinement. INV-6: this + * REPORTS a grant the executive genuinely made; it fabricates nothing. + */ + uint32_t pend_lkid = 0, pend_csid = 0, pend_req_lkid = 0; + { + struct vms_lock_entry *w; + exec_list_for_each_entry(w, &res->waiting, res_waiting) { + if (w->req_csid != 0) { + pend_lkid = w->lkid; + pend_csid = w->req_csid; + pend_req_lkid = w->req_lkid; + break; + } + } + } + /* The release grants any waiter now compatible -- the blocked cross-node * request flips from queued (grant_state 0) to granted (grant_state * SS$_NORMAL, granted_mode = requested). */ try_grant_waiters(res); + + /* Did the snapshotted cross-node waiter just get granted? If so, name it in + * the deferred-GRANT outputs (reusing the fields that are otherwise 0 on a + * DEQ). We are still under res->lock, and the granted list is walked here. */ + req->queued = 0; + req->blocking_csid = 0; + req->blocking_master_lkid = 0; + if (pend_lkid != 0) { + struct vms_lock_entry *g; + exec_list_for_each_entry(g, &res->granted, res_granted) { + if (g->lkid == pend_lkid && !g->waiting && + g->grant_state == SS__NORMAL) { + /* DEFERRED GRANT: this queued cross-node request is now granted. */ + req->queued = 1; /* "a waiter flipped" */ + req->blocking_csid = pend_csid; /* -> deferred-grant CSID */ + req->blocking_master_lkid = g->lkid; /* -> its master handle */ + req->req_lkid = pend_req_lkid; /* -> requester's handle */ + req->lkmode = g->granted_mode; /* -> the mode granted */ + break; + } + } + (void)pend_req_lkid; + } exec_unlock(&res->lock); exec_lock(&lock->proc->lock_list_lock); @@ -1525,6 +1654,108 @@ static uint32_t vms_lock_dlm_xnode_deq(struct vms_dlm_xnode_args *req) return SS__NORMAL; } +/* + * vms_lock_dlm_xnode_grant_recv - the REQUESTER-SIDE GRANT RECEIVE (vms-6ca, DLM + * epic vms-7fa rung H5). Previously SS$_UNSUPPORTED. + * + * A GRANT / queued-reply message the MASTER sent back over SCS in answer to a + * cross-node $ENQ THIS node issued lands here. It completes the requester's + * origin record -- the executive-resident proxy of the outstanding request -- + * so the request's status is genuine executive state, not a per-process + * userspace flag (INV-6). The record's granted mode is set ONLY from what the + * master genuinely sent: + * - a queued-reply carries lkmode == NL -> the origin record stays pending + * (granted_mode NL): the requester genuinely sees "blocked", from the + * master's real waiting-queue decision transported over the wire. + * - a deferred GRANT carries lkmode == the granted mode (e.g. EX) and + * status SS$_NORMAL -> the origin record flips NL -> EX. THIS is the status + * flip observed on the REQUESTER node, driven by the master's real release. + * + * find-or-create keyed by the requester's own lock handle (req_lkid). The record + * never touches the resource granted/waiting queues (the local lock manager + * cannot auto-grant it); only this wire path completes it. GETLKI(req_lkid) + * reads it back, so the flip is independently observable. + * + * Returns SS$_NORMAL when the receive was accepted (the record now reflects the + * master's status), or SS$_INSFMEM if the record could not be allocated. It + * grants nothing itself -- it records the master's genuine decision. + */ +static uint32_t vms_lock_dlm_xnode_grant_recv(struct vms_dlm_xnode_args *req) +{ + struct vms_dlm_origin *org = NULL, *cur; + + if (req->req_lkid == 0) + return SS__BADPARAM; /* a GRANT must name the requester's own handle */ + + exec_lock(&vms_dlm_origin_lock); + exec_list_for_each_entry(cur, &vms_dlm_origin_list, list) { + if (cur->req_lkid == req->req_lkid) { + org = cur; + break; + } + } + if (org == NULL) { + org = exec_zalloc(sizeof(*org)); + if (org == NULL) { + exec_unlock(&vms_dlm_origin_lock); + return SS__INSFMEM; + } + org->req_lkid = req->req_lkid; + org->req_csid = req->req_csid; + org->requested_mode = LCK_K_EXMODE; /* refined below from the reply mode */ + org->granted_mode = LCK_K_NLMODE; /* pending until a real grant arrives */ + memcpy(org->resnam, req->resnam, sizeof(org->resnam)); + org->resnam[sizeof(org->resnam) - 1] = '\0'; + exec_list_add_tail(&org->list, &vms_dlm_origin_list); + } + + /* Record what the MASTER genuinely reported. The master's lock handle and + * CSID come from the reply; the granted mode is whatever the master said the + * request is granted at (NL == still pending/queued, non-NL == granted). */ + if (req->master_lkid != 0) + org->master_lkid = req->master_lkid; + if (req->master_csid != 0) + org->master_csid = req->master_csid; + if (req->lkmode > LCK_K_NLMODE) + org->requested_mode = req->lkmode; + org->granted_mode = req->lkmode; /* NL on a queued-reply; EX on a grant */ + + exec_unlock(&vms_dlm_origin_lock); + return SS__NORMAL; +} + +/* + * vms_lock_dlm_origin_getlki - GETLKI fall-through for a requester-side origin + * record (vms-6ca, H5). Returns 1 and fills the getlki fields if an origin + * record with lkid == req_lkid exists; 0 otherwise. Lets GETLKI observe the + * cross-node request's status flip on the REQUESTER node. + */ +static int vms_lock_dlm_origin_getlki(uint32_t lkid, uint32_t *granted_mode, + uint32_t *requested_mode, char *resnam, + size_t resnam_len) +{ + struct vms_dlm_origin *cur; + int found = 0; + + if (lkid == 0) + return 0; + exec_lock(&vms_dlm_origin_lock); + exec_list_for_each_entry(cur, &vms_dlm_origin_list, list) { + if (cur->req_lkid == lkid) { + if (granted_mode) + *granted_mode = cur->granted_mode; + if (requested_mode) + *requested_mode = cur->requested_mode; + if (resnam && resnam_len) + strscpy(resnam, cur->resnam, resnam_len); + found = 1; + break; + } + } + exec_unlock(&vms_dlm_origin_lock); + return found; +} + /* * vms_lock_dlm_xnode_dispatch - the cross-node DLM RECEIVE handler * (vms-94c transport; DLM epic vms-7fa). @@ -1607,6 +1838,8 @@ uint32_t vms_lock_dlm_xnode_dispatch(struct vms_proc *proc, a.owner_csid = req->req_csid; /* held FOR the remote requester */ memset(&xn, 0, sizeof(xn)); + xn.req_lkid = req->req_lkid; /* stamp the master lock with the + * requester's own handle (H5) */ vms_enq_core_ex(proc, &a, &xn); /* Hand the master's lock id back (the GRANT reply's master_lkid) and the @@ -1628,10 +1861,16 @@ uint32_t vms_lock_dlm_xnode_dispatch(struct vms_proc *proc, return SS__BADPARAM; return vms_lock_dlm_xnode_deq(req); case VMS_DLM_OP_GRANT: + /* REQUESTER-SIDE GRANT RECEIVE (vms-6ca, H5). A GRANT / queued-reply the + * MASTER sent back for a cross-node $ENQ THIS node issued completes the + * requester's origin record -- genuine executive state, the status flip + * observed on the requester node. Was SS$_UNSUPPORTED. */ + return vms_lock_dlm_xnode_grant_recv(req); case VMS_DLM_OP_BLKAST: - /* Responses carry no resource name. Completing/notifying the ORIGINATING - * node's pending request is the requester-side wiring, not this - * master-centric path; decline honestly. */ + /* BLKAST RECEIVE is the holder-side blocking-AST delivery. The BLKAST + * WIRE (master -> holder) is deferred honestly on this rung (vms-6ca): + * the block-then-grant round-trip is proven without it (the holder + * releases on its own). Still SS$_UNSUPPORTED -- never faked. */ return SS__UNSUPPORTED; default: return SS__BADPARAM; diff --git a/src/kernel-netbsd/vms_internal.h b/src/kernel-netbsd/vms_internal.h index bf9676b9f..ed3abbad3 100644 --- a/src/kernel-netbsd/vms_internal.h +++ b/src/kernel-netbsd/vms_internal.h @@ -394,6 +394,13 @@ struct vms_lock_entry { * a cross-node grant on behalf of a remote * node (vms-e8f1). Set from * vms_enq_args.owner_csid at creation. */ + uint32_t req_lkid; /* the REMOTE requester's own lock handle + * for a cross-node grant (vms-6ca, H5). + * 0 for a local lock. Set from the wire + * ENQ's req_lkid so a later deferred GRANT + * can name the requester's original + * request when this lock's release grants + * a queued cross-node waiter. */ }; /* Lock resource -- a named resource in the lock database. */ diff --git a/src/kernel/vms_internal.h b/src/kernel/vms_internal.h index a097d55a5..60571d0a6 100644 --- a/src/kernel/vms_internal.h +++ b/src/kernel/vms_internal.h @@ -460,6 +460,14 @@ struct vms_lock_entry { * vms-e8f1). Set from vms_enq_args * .owner_csid at creation; surfaced by * GET_RESMASTER.remote_holder_csid. */ + uint32_t req_lkid; /* the REMOTE requester's own lock handle + * for a cross-node grant (vms-6ca, DLM + * epic vms-7fa rung H5). 0 for a local + * lock. Set from the wire ENQ's req_lkid + * so a later deferred GRANT (sent when + * this lock's release grants a queued + * cross-node waiter) can name the + * requester's ORIGINAL request. */ }; /* Lock resource (named resource in the lock database) */ diff --git a/src/libvmssys/vms_kif.c b/src/libvmssys/vms_kif.c index 782855bb7..65a854bcc 100644 --- a/src/libvmssys/vms_kif.c +++ b/src/libvmssys/vms_kif.c @@ -759,7 +759,8 @@ uint32_t vms_kif_dlm_xnode(uint32_t op, uint32_t lkmode, uint32_t flags, const char *resnam, const uint8_t *valblk, uint32_t *out_master_lkid, uint32_t *out_queued, uint32_t *out_blocking_csid, - uint32_t *out_blocking_master_lkid) + uint32_t *out_blocking_master_lkid, + uint32_t *out_req_lkid, uint32_t *out_lkmode) { struct vms_dlm_xnode_args args; @@ -788,6 +789,14 @@ uint32_t vms_kif_dlm_xnode(uint32_t op, uint32_t lkmode, uint32_t flags, if (out_queued) *out_queued = args.queued; if (out_blocking_csid) *out_blocking_csid = args.blocking_csid; if (out_blocking_master_lkid) *out_blocking_master_lkid = args.blocking_master_lkid; + /* Deferred-GRANT readback (vms-6ca, H5): on a cross-node $DEQ that flipped a + * queued cross-node waiter to granted, the executive names that waiter in the + * fields a DEQ otherwise leaves 0 -- queued=1 (a waiter flipped), blocking_* + * = the granted waiter's CSID + master handle, req_lkid = its requester + * handle, lkmode = the mode it was granted at. The daemon reads these to wire + * the deferred GRANT back to the requester. */ + if (out_req_lkid) *out_req_lkid = args.req_lkid; + if (out_lkmode) *out_lkmode = args.lkmode; return args.status; } diff --git a/src/libvmssys/vms_kif.h b/src/libvmssys/vms_kif.h index 616613f4b..96bba98d2 100644 --- a/src/libvmssys/vms_kif.h +++ b/src/libvmssys/vms_kif.h @@ -308,7 +308,8 @@ uint32_t vms_kif_dlm_xnode(uint32_t op, uint32_t lkmode, uint32_t flags, const char *resnam, const uint8_t *valblk, uint32_t *out_master_lkid, uint32_t *out_queued, uint32_t *out_blocking_csid, - uint32_t *out_blocking_master_lkid); + uint32_t *out_blocking_master_lkid, + uint32_t *out_req_lkid, uint32_t *out_lkmode); /* ================================================================ * Device table (executive-resident I/O database) diff --git a/tests/qemu/test_syssvc_dlm_xnode.c b/tests/qemu/test_syssvc_dlm_xnode.c index b163706f8..ea3c048c1 100644 --- a/tests/qemu/test_syssvc_dlm_xnode.c +++ b/tests/qemu/test_syssvc_dlm_xnode.c @@ -39,10 +39,19 @@ * handle now shows granted_mode == EX -- the unfakeable status flip NL->EX, * driven by a real $DEQ -- and the master now holds FOR the second node's * CSID. + * (3b) ⭐ DEFERRED-GRANT REPORT + REQUESTER-SIDE RECEIVE (vms-6ca, rung H5). + * The block-then-grant $DEQ now REPORTS the flipped waiter (the requester + * CSID + original req_lkid + master handle + granted mode) so the daemon + * can WIRE a deferred GRANT to it. And the REQUESTER-SIDE GRANT RECEIVE + * (VMS_DLM_OP_GRANT, was SS$_UNSUPPORTED) completes an executive-resident + * ORIGIN record: a queued-reply (mode NL) leaves it PENDING, a deferred + * GRANT (mode EX) flips it NL->EX -- the status flip observed on the + * REQUESTER node, driven only by what the master sent over the wire, read + * back by GETLKI. INV-6: genuine executive state, no fabricated grant. * (4) SCOPE FENCE / INV-6: a wire NOQUEUE incompatible $ENQ still declines * SS$_NOTQUEUED (honest); a $DEQ of a lock NOT held for the releasing node - * is refused SS$_IVLOCKID; the RESPONSE ops (GRANT/BLKAST as receive) still - * return SS$_UNSUPPORTED -- requester-side completion is a later rung. + * is refused SS$_IVLOCKID; BLKAST as a receive op still returns + * SS$_UNSUPPORTED -- the BLKAST WIRE is deferred honestly on this rung. * (5) VALIDATION unchanged: a bad mode, a bad op, or an ENQ with an empty * resource name is refused SS$_BADPARAM, not silently dropped. * @@ -110,7 +119,7 @@ int main(void) st = vms_kif_dlm_xnode(VMS_DLM_OP_ENQ, LCK_K_EXMODE, 0, 0x00040011u /*req_lkid*/, 0 /*master_lkid*/, REQ_CSID_A, 0 /*master_csid: resolve*/, res, NULL, - &lkid_a, &queued, &blk_csid, &blk_lkid); + &lkid_a, &queued, &blk_csid, &blk_lkid, NULL, NULL); CHECK(st == SS_NORMAL, "cross-node ENQ (compatible) -> SS$_NORMAL (rung-2 grant)"); CHECK(queued == 0, "the granted request was NOT queued"); CHECK(lkid_a != 0, "the master returned a lock handle for the grant"); @@ -145,7 +154,7 @@ int main(void) queued = 0; blk_csid = 0; blk_lkid = 0; st = vms_kif_dlm_xnode(VMS_DLM_OP_ENQ, LCK_K_EXMODE, 0, 0x00050022u, 0, REQ_CSID_B, 0, res, NULL, - &lkid_b, &queued, &blk_csid, &blk_lkid); + &lkid_b, &queued, &blk_csid, &blk_lkid, NULL, NULL); CHECK(st == (uint32_t)VMS_DLM_STS_QUEUED, "second incompatible cross-node ENQ -> VMS_DLM_STS_QUEUED (blocked, not granted)"); CHECK(st != SS_NORMAL && st != SS_NOTQUEUED, @@ -171,22 +180,42 @@ int main(void) /* ---- 3. SCOPE FENCE: a NOQUEUE incompatible ENQ still DECLINES ---------- */ st = vms_kif_dlm_xnode(VMS_DLM_OP_ENQ, LCK_K_EXMODE, LCK_M_NOQUEUE, 0x00060033u, 0, REQ_CSID_C, 0, res, NULL, - NULL, NULL, NULL, NULL); + NULL, NULL, NULL, NULL, NULL, NULL); CHECK(st == SS_NOTQUEUED, "NOQUEUE incompatible cross-node ENQ -> SS$_NOTQUEUED (honest decline preserved)"); /* ---- 4. cross-node $DEQ authorization: wrong CSID is refused ------------ */ st = vms_kif_dlm_xnode(VMS_DLM_OP_DEQ, LCK_K_NLMODE, 0, 0, lkid_b /*B's queued lock*/, REQ_CSID_A /*wrong owner*/, - 0, res, NULL, NULL, NULL, NULL, NULL); + 0, res, NULL, NULL, NULL, NULL, NULL, NULL, NULL); CHECK(st == SS_IVLOCKID, "cross-node DEQ of a lock NOT held for the releasing node -> SS$_IVLOCKID"); - /* ---- 5. ⭐ BLOCK-THEN-GRANT: the holder DEQs, the blocked request GRANTS - */ + /* ---- 5. ⭐ BLOCK-THEN-GRANT + DEFERRED-GRANT REPORT (vms-6ca, H5) -------- + * The holder DEQs; the master releases and GRANTS the blocked request. The + * $DEQ dispatch now REPORTS the flipped waiter through the fields a DEQ + * otherwise leaves 0 -- queued=1 (a waiter flipped), blocking_csid = the + * requester to notify (B), blocking_master_lkid = its master handle (lkid_b), + * def_req_lkid = its ORIGINAL requester handle (0x00050022), def_mode = the + * mode it was granted at (EX). This is what the daemon reads to WIRE the + * deferred GRANT to the requester. INV-6: it reports a grant the executive + * genuinely made. */ + uint32_t def_flipped = 0, def_csid = 0, def_master_lkid = 0, + def_req_lkid = 0, def_mode = 0; st = vms_kif_dlm_xnode(VMS_DLM_OP_DEQ, LCK_K_NLMODE, 0, 0, lkid_a, REQ_CSID_A, 0, res, NULL, - NULL, NULL, NULL, NULL); + NULL, &def_flipped, &def_csid, &def_master_lkid, + &def_req_lkid, &def_mode); CHECK(st == SS_NORMAL, "holder's cross-node $DEQ -> SS$_NORMAL (released)"); + CHECK(def_flipped == 1, "the $DEQ REPORTED a deferred grant (a queued waiter flipped)"); + CHECK(def_csid == REQ_CSID_B, + "the deferred-grant report names the requester to notify (B)"); + CHECK(def_master_lkid == lkid_b, + "the deferred-grant report carries the flipped waiter's master handle"); + CHECK(def_req_lkid == 0x00050022u, + "the deferred-grant report carries the requester's ORIGINAL lock handle"); + CHECK(def_mode == LCK_K_EXMODE, + "the deferred-grant report carries the granted mode (EX)"); /* the previously-blocked request has FLIPPED to granted at EX */ gm = 99; rm = 99; @@ -201,38 +230,74 @@ int main(void) CHECK(n_granted == 1 && held_for == REQ_CSID_B, "the master now holds one lock, held FOR the second node (B)"); + /* ---- 5b. ⭐ REQUESTER-SIDE GRANT RECEIVE (vms-6ca, H5) ------------------- + * The OTHER half of the wire: on the REQUESTER node, a GRANT / queued-reply + * the master sent back completes an executive-resident ORIGIN record, so the + * request's status is genuine executive state (INV-6), not a userspace flag. + * A queued-reply (granted mode NL) leaves it PENDING; a deferred GRANT + * (granted mode EX) flips it NL->EX -- the status flip observed on the + * REQUESTER node, driven ONLY by what the master sent over the wire. The op + * was SS$_UNSUPPORTED before this rung. GETLKI reads the flip back + * independently. RREQ_LKID is the requester's own handle for the request. */ + const uint32_t RREQ_LKID = 0x00090077u; + /* (i) queued-reply arrives: mode NL, master handle 0x0808 on CSID B. */ + st = vms_kif_dlm_xnode(VMS_DLM_OP_GRANT, LCK_K_NLMODE, 0, + RREQ_LKID, 0x00000808u, REQ_CSID_A, REQ_CSID_B, + "RORIGIN1", NULL, + NULL, NULL, NULL, NULL, NULL, NULL); + CHECK(st == SS_NORMAL, + "requester-side queued-reply RECEIVE -> SS$_NORMAL (was SS$_UNSUPPORTED)"); + gm = 99; rm = 99; + st = vms_kif_getlki(RREQ_LKID, &gm, &rm, NULL, NULL); + CHECK(st == SS_NORMAL && gm == LCK_K_NLMODE, + "GETLKI on the requester: the origin record is PENDING (granted NL)"); + /* (ii) deferred GRANT arrives: mode EX -> the origin record FLIPS NL->EX. */ + st = vms_kif_dlm_xnode(VMS_DLM_OP_GRANT, LCK_K_EXMODE, 0, + RREQ_LKID, 0x00000808u, REQ_CSID_A, REQ_CSID_B, + "RORIGIN1", NULL, + NULL, NULL, NULL, NULL, NULL, NULL); + CHECK(st == SS_NORMAL, "requester-side deferred GRANT RECEIVE -> SS$_NORMAL"); + gm = 99; rm = 99; + st = vms_kif_getlki(RREQ_LKID, &gm, &rm, NULL, NULL); + CHECK(st == SS_NORMAL && gm == LCK_K_EXMODE, + "GETLKI on the requester: the origin record FLIPPED to GRANTED at EX (NL->EX)"); + /* a GRANT with no requester handle is refused, not silently dropped */ + st = vms_kif_dlm_xnode(VMS_DLM_OP_GRANT, LCK_K_EXMODE, 0, + 0 /*no req_lkid*/, 0x00000808u, REQ_CSID_A, REQ_CSID_B, + "RORIGIN1", NULL, + NULL, NULL, NULL, NULL, NULL, NULL); + CHECK(st == SS_BADPARAM, "requester-side GRANT with no req_lkid -> SS$_BADPARAM"); + /* ---- 6. release the second lock; the resource is torn down -------------- */ st = vms_kif_dlm_xnode(VMS_DLM_OP_DEQ, LCK_K_NLMODE, 0, 0, lkid_b, REQ_CSID_B, 0, res, NULL, - NULL, NULL, NULL, NULL); + NULL, NULL, NULL, NULL, NULL, NULL); CHECK(st == SS_NORMAL, "second node's cross-node $DEQ -> SS$_NORMAL"); - /* ---- 7. STILL FENCED: the RESPONSE ops (later rung) decline honestly ---- */ - st = vms_kif_dlm_xnode(VMS_DLM_OP_GRANT, LCK_K_EXMODE, 0, - 0x00040011u, 0x00080002u, REQ_CSID_A, REQ_CSID_B, "", NULL, - NULL, NULL, NULL, NULL); - CHECK(st == SS_UNSUPPORTED, "cross-node GRANT (receive, no resnam) -> SS$_UNSUPPORTED"); - + /* ---- 7. STILL FENCED: BLKAST receive (the BLKAST WIRE) is deferred ------ + * The block-then-grant round-trip is proven WITHOUT the BLKAST wire (vms-6ca + * defers it honestly). BLKAST as a RECEIVE op still declines SS$_UNSUPPORTED + * -- never faked. */ st = vms_kif_dlm_xnode(VMS_DLM_OP_BLKAST, LCK_K_EXMODE, 0, 0x00040011u, 0x00080002u, REQ_CSID_A, REQ_CSID_B, "", NULL, - NULL, NULL, NULL, NULL); - CHECK(st == SS_UNSUPPORTED, "cross-node BLKAST (receive, no resnam) -> SS$_UNSUPPORTED"); + NULL, NULL, NULL, NULL, NULL, NULL); + CHECK(st == SS_UNSUPPORTED, "cross-node BLKAST (receive) -> SS$_UNSUPPORTED (wire deferred)"); /* ---- 8. malformed requests are refused, not dropped ---- */ st = vms_kif_dlm_xnode(VMS_DLM_OP_ENQ, LCK_K_EXMODE + 1, 0, 1, 0, REQ_CSID_A, 0, res, NULL, - NULL, NULL, NULL, NULL); + NULL, NULL, NULL, NULL, NULL, NULL); /* negctl: dlm-xnode-mode-unvalidated */ CHECK(st == SS_BADPARAM, "bad lock mode -> SS$_BADPARAM"); st = vms_kif_dlm_xnode(99u /*bad op*/, LCK_K_EXMODE, 0, 1, 0, REQ_CSID_A, 0, res, NULL, - NULL, NULL, NULL, NULL); + NULL, NULL, NULL, NULL, NULL, NULL); CHECK(st == SS_BADPARAM, "unknown op -> SS$_BADPARAM"); st = vms_kif_dlm_xnode(VMS_DLM_OP_ENQ, LCK_K_EXMODE, 0, 1, 0, REQ_CSID_A, 0, "" /*empty name*/, NULL, - NULL, NULL, NULL, NULL); + NULL, NULL, NULL, NULL, NULL, NULL); CHECK(st == SS_BADPARAM, "ENQ with empty resource name -> SS$_BADPARAM"); printf("=== test_syssvc_dlm_xnode: %d passed, %d failed ===\n", pass, fail); From 5c270993060881fc59ca8783fadcb51a635bb857 Mon Sep 17 00:00:00 2001 From: alice Date: Fri, 28 Aug 2026 22:30:40 +0000 Subject: [PATCH 2/7] =?UTF-8?q?vms-6ca:=20DLM=20H5=20SCS=20wire=20?= =?UTF-8?q?=E2=80=94=20queued-reply=20+=20deferred-GRANT=20delivery=20over?= =?UTF-8?q?=20the=20live=20VC?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The scsd daemon now carries the async DLM replies over the real SCS wire, so the block-then-grant completes on the REQUESTER node across the network — not just in the master's local executive (#873). Node B (master): the GRANT frame it sends back now carries the granted mode honestly — a queued-reply carries NL (the requester stays genuinely pending), a grant carries the granted mode — plus the master's lock handle, so the requester can $DEQ the holder. On a cross-node $DEQ that flips a queued waiter to granted, B reads the executive's deferred-grant report and WIRES an unprompted deferred GRANT (SS$_NORMAL, the granted mode) to that requester (SCSD-I-DLMDEFER). Node A (requester, armed by OVMX_DLM_H5): drives block-then-grant over the wire — holds RESONE EX (#1), sends a second incompatible $ENQ (#2) that QUEUES on B, dispatches B's queued-reply into its OWN executive so #2's origin record is genuinely PENDING (GETLKI->NL, SCSD-I-DLMPEND), releases the holder (#1), and on B's deferred GRANT dispatches it into its executive — the origin record FLIPS NL->EX (GETLKI->EX, SCSD-I-DLMH5FLIP). The status flip is read from A's real executive, driven only by what the master sent over SCS. INV-6: no fabricated wire reply or grant; the BLKAST wire stays deferred (the holder releases on its own). scsd builds clean; vmsscs unit tests green. Co-Authored-By: Claude Opus 4.8 (1M context) --- src/vmsscs/scsd.c | 290 ++++++++++++++++++++++++++++++++++++++++++++-- 1 file changed, 278 insertions(+), 12 deletions(-) diff --git a/src/vmsscs/scsd.c b/src/vmsscs/scsd.c index d1232ab99..3ad2eb1b2 100644 --- a/src/vmsscs/scsd.c +++ b/src/vmsscs/scsd.c @@ -988,6 +988,16 @@ struct peer_state { int dlm_grant_recv; /* node-A: the peer's GRANT response arrived */ uint32_t dlm_grant_status; /* node-A: the VMS status the peer's executive returned */ unsigned long dlm_srv_responses; /* node-B: GRANT frames we emitted back to this peer */ + /* vms-6ca (DLM rung H5): node-A's block-then-grant-over-the-wire sequence. + * A holds RESONE EX (req_lkid 1), then a SECOND incompatible $ENQ (req_lkid 2) + * QUEUES on B; A releases #1; B WIRES the deferred GRANT and A's origin record + * for #2 flips NL->EX. */ + int dlm_h5; /* node-A: H5 sequence armed (OVMX_DLM_H5) */ + uint32_t dlm_master_lkid1; /* node-A: B's master handle for our holder (#1) */ + int dlm_enq2_sent; /* node-A: the second (contending) $ENQ sent */ + int dlm_deq1_sent; /* node-A: the holder's $DEQ sent */ + int dlm_pend_seen; /* node-A: the queued-reply for #2 landed (pending) */ + int dlm_flip_seen; /* node-A: #2's origin record flipped to granted */ /* vms-694 (§4(O.7)): OVMX's OWN model of "am I a connected VMS$VAXcluster * member of this peer", LATCHED the moment the VMS$VAXcluster SYSAP * connection (cdt_joiner, or cdt_member if it ever reaches OPEN) hits the @@ -7879,6 +7889,24 @@ struct scsd_dlm_held { uint32_t is_local_master; uint32_t n_granted; uint32_t held_for_csid; /* remote_holder_csid: whose CSID the grant is for */ + uint32_t master_lkid; /* the master's lock handle for the dispatched req + * (vms-6ca, H5): sent back so the requester can name + * the holder in a later $DEQ. */ +}; + +/* + * Deferred-GRANT report (DLM epic vms-7fa rung H5, vms-6ca). On a cross-node + * $DEQ that flipped a QUEUED cross-node waiter to granted, the executive names + * that waiter so node B can WIRE a deferred GRANT to the requester. `flipped` + * is 1 iff a waiter flipped; the rest name it. A READ of the executive's genuine + * decision, never fabricated (INV-6). + */ +struct scsd_dlm_defer { + int flipped; + uint32_t req_csid; /* requester to notify with the deferred GRANT */ + uint32_t master_lkid; /* the flipped waiter's master lock handle */ + uint32_t req_lkid; /* the flipped waiter's ORIGINAL requester handle */ + uint32_t mode; /* the mode it was granted at */ }; /* @@ -7894,10 +7922,16 @@ struct scsd_dlm_held { * on the SAME registered fd so the caller can print the held-lock proof. */ static uint32_t scsd_dlm_dispatch_to_executive(const struct scs_dlm_msg *m, - struct scsd_dlm_held *held) + struct scsd_dlm_held *held, + struct scsd_dlm_defer *defer, + uint32_t *out_origin_mode) { if (held) memset(held, 0, sizeof(*held)); + if (defer) + memset(defer, 0, sizeof(*defer)); + if (out_origin_mode) + *out_origin_mode = 0; #ifdef SCSD_UNIT_TEST /* Capture seam, parallel to send_frame_raw's: record the decoded request so * a test can assert node B decoded + dispatched the right fields. The real @@ -7956,6 +7990,11 @@ static uint32_t scsd_dlm_dispatch_to_executive(const struct scs_dlm_msg *m, return 2680u; /* honest failure, not a grant */ } + /* The master's lock handle for the dispatched request (vms-6ca, H5), so the + * requester can name the holder in a later cross-node $DEQ. */ + if (held != NULL) + held->master_lkid = args.master_lkid; + /* DLM rung 2 (vms-e8f1): if the master GRANTED this cross-node $ENQ, read its * OWN resource DB back on the SAME registered fd so the caller can prove the * grant is genuine -- the resource is mastered here and a lock is held FOR @@ -7976,6 +8015,32 @@ static uint32_t scsd_dlm_dispatch_to_executive(const struct scs_dlm_msg *m, } } + /* DLM rung H5 (vms-6ca): a cross-node $DEQ that flipped a queued waiter to + * granted names it in the fields a DEQ otherwise leaves 0, so the caller can + * WIRE the deferred GRANT to that requester. A READ of the executive's real + * report, never fabricated (INV-6). */ + if (defer != NULL && args.op == VMS_DLM_OP_DEQ && args.queued == 1u) { + defer->flipped = 1; + defer->req_csid = args.blocking_csid; + defer->master_lkid = args.blocking_master_lkid; + defer->req_lkid = args.req_lkid; + defer->mode = args.lkmode; + } + + /* DLM rung H5 (vms-6ca): the REQUESTER-SIDE GRANT RECEIVE completed our + * executive-resident ORIGIN record. Read the record BACK via GETLKI on the + * SAME registered fd so the caller can prove the status the master sent + * genuinely landed in executive state -- the NL->EX flip observed on the + * requester node, not a userspace flag (INV-6). */ + if (out_origin_mode != NULL && args.op == VMS_DLM_OP_GRANT && + args.status == 1u && args.req_lkid != 0) { + struct vms_getlki_args gl; + memset(&gl, 0, sizeof(gl)); + gl.lkid = args.req_lkid; + if (ioctl(fd, VMS_IOCTL_GETLKI, &gl) == 0 && gl.status == 1u) + *out_origin_mode = gl.granted_mode; + } + close(fd); return args.status; #endif @@ -8016,7 +8081,8 @@ static void scsd_dlm_srv_msg_input(struct scs_cdt *cdt, const void *msg, * present (the Docker harness -- Rule 9: Docker is not a runtime, there is * no /dev/vms). Either way it is HONEST -- never a fabricated grant. */ struct scsd_dlm_held held; - uint32_t status = scsd_dlm_dispatch_to_executive(&v.msg, &held); + struct scsd_dlm_defer defer; + uint32_t status = scsd_dlm_dispatch_to_executive(&v.msg, &held, &defer, NULL); rx_dlm_dispatched++; log_ts(stdout); printf(" SCSD-I-DLMRX, cross-node %s from CSID=%u resnam='%.*s'" @@ -8068,8 +8134,14 @@ static void scsd_dlm_srv_msg_input(struct scs_cdt *cdt, const void *msg, struct scs_dlm_msg g; memset(&g, 0, sizeof(g)); g.op = SCS_DLM_OP_GRANT; - g.mode = v.msg.mode; + /* The granted mode the requester must record (vms-6ca, H5): a GRANT (status + * SS$_NORMAL) carries the granted mode; a QUEUED-reply (status + * VMS_DLM_STS_QUEUED == 0) carries NL so the requester's origin record stays + * genuinely PENDING -- never a fabricated grant (INV-6). */ + g.mode = (status == 1u) ? v.msg.mode : (uint8_t)LCK$K_NLMODE; g.req_lkid = v.msg.req_lkid; + g.master_lkid = held.master_lkid; /* the master's handle (H5): lets + * the requester $DEQ the holder */ g.status = status; /* the HONEST executive status */ g.req_csid = v.msg.req_csid; g.master_csid = resolve_scssystemid(); /* this node mastered the request */ @@ -8091,15 +8163,132 @@ static void scsd_dlm_srv_msg_input(struct scs_cdt *cdt, const void *msg, : "honest response, request not granted"); fflush(stdout); } + + /* THE DEFERRED GRANT (vms-6ca, DLM epic vms-7fa rung H5). If THIS $DEQ + * released a holder and thereby GRANTED a previously-QUEUED cross-node + * request, the executive named that request in `defer`. WIRE a deferred + * SCS_DLM_OP_GRANT (status SS$_NORMAL, the granted mode) to the requester so + * ITS origin record flips from pending to granted -- the block-then-grant + * completion, carried over the live SCS wire, driven by a real release. This + * is the reply the master pushes UNPROMPTED (not in answer to a message the + * requester just sent). INV-6: it carries a grant the executive genuinely + * made (`defer` is a READ of the DEQ report), never a fabricated success. + * + * H5 proves the SAME-PEER case (the released holder and the queued waiter are + * the same peer, so the deferred GRANT rides this peer's VC); routing a + * deferred grant to a DIFFERENT node by CSID is a later multi-peer rung. */ + if (defer.flipped && defer.req_lkid != 0) { + struct scs_dlm_params dp; + memset(&dp, 0, sizeof(dp)); + memcpy(dp.dst_mac, ps_port_addr(ps), 6); + memcpy(dp.src_mac, rx->our_hw_mac, 6); + memcpy(dp.src_logical, rx->our_src_logical, 6); + memcpy(dp.peer_logical, ps_sys_addr(ps), 6); + dp.local_conid = PS_DLM_SERVER_CONID(ps); + dp.remote_conid = v.local_conid; /* the requester's DLM client handle */ + dp.recv_ack = ps->vc.seq.recv_seq; + dp.send_seq = scs_seq_advance(&ps->vc.seq); + dp.incarnation = ps->incarnation; + + struct scs_dlm_msg dg; + memset(&dg, 0, sizeof(dg)); + dg.op = SCS_DLM_OP_GRANT; + dg.mode = (uint8_t)defer.mode; /* the mode it was granted at (EX) */ + dg.req_lkid = defer.req_lkid; /* the queued request's own handle */ + dg.master_lkid = defer.master_lkid; + dg.status = 1u; /* SS$_NORMAL -- a REAL grant */ + dg.req_csid = defer.req_csid; + dg.master_csid = resolve_scssystemid(); + dg.namelen = v.msg.namelen; + memcpy(dg.resnam, v.msg.resnam, sizeof(dg.resnam)); + + uint8_t dframe[SCS_DLM_FRAME_LEN]; + if (scs_dlm_build_frame(&dp, &dg, dframe) == 0 && + send_frame_vc(rx->sock, rx->ifindex, ps, ps->pb, + "OVMX$DLM deferred GRANT (H5)", dframe, + sizeof(dframe)) > 0) { + ps->dlm_srv_responses++; + log_ts(stdout); + printf(" SCSD-I-DLMDEFER, WIRED deferred GRANT req_lkid=0x%08X mode=%s" + " to CSID=%u -- a queued cross-node request GRANTED by the real" + " release (block-then-grant over SCS)\n", + (unsigned)defer.req_lkid, scs_dlm_mode_name((uint8_t)defer.mode), + (unsigned)defer.req_csid); + fflush(stdout); + } + } +} + +/* + * scsd_dlm_client_send_op - node-A's generic DLM client SEND (vms-6ca, H5). + * Builds and sends one DLM message on our DLM client CDT toward the peer's DLM + * server handle (the same addressing scsd_dlm_send_enq resolves). Used for the + * H5 sequence's second $ENQ and the holder's $DEQ. Returns 1 on send. + */ +static int scsd_dlm_client_send_op(struct scsd_rx *rx, struct peer_state *ps, + uint8_t op, uint8_t mode, uint32_t req_lkid, + uint32_t master_lkid, const char *resname, + const char *label) +{ + if (rx == NULL || ps == NULL || ps->pb == NULL || ps->joiner_remote_conid == 0) + return 0; + + uint32_t dlm_server = (ps->joiner_remote_conid & 0xFFFFFFF0u) | + OVMX_CONID_CLS_DLMSRV; + + struct scs_dlm_params p; + memset(&p, 0, sizeof(p)); + memcpy(p.dst_mac, ps_port_addr(ps), 6); + memcpy(p.src_mac, rx->our_hw_mac, 6); + memcpy(p.src_logical, rx->our_src_logical, 6); + memcpy(p.peer_logical, ps_sys_addr(ps), 6); + p.local_conid = PS_DLM_CONID(ps); + p.remote_conid = dlm_server; + p.recv_ack = ps->vc.seq.recv_seq; + p.send_seq = scs_seq_advance(&ps->vc.seq); + p.incarnation = ps->incarnation; + + struct scs_dlm_msg m; + memset(&m, 0, sizeof(m)); + m.op = op; + m.mode = mode; + m.req_lkid = req_lkid; + m.master_lkid = master_lkid; + m.req_csid = resolve_scssystemid(); + if (resname != NULL) { + size_t rl = strlen(resname); + if (rl > SCS_DLM_RESNAM_MAX) + rl = SCS_DLM_RESNAM_MAX; + m.namelen = (uint8_t)rl; + memcpy(m.resnam, resname, rl); + } + + uint8_t frame[SCS_DLM_FRAME_LEN]; + if (scs_dlm_build_frame(&p, &m, frame) == 0 && + send_frame_vc(rx->sock, rx->ifindex, ps, ps->pb, label, frame, + sizeof(frame)) > 0) + return 1; + return 0; } /* - * scsd_dlm_cli_msg_input - node-A's DLM CLIENT input routine (vms-164d rung-1b). - * The GRANT the peer sent back in answer to our ENQ lands here (dest Con.ID = - * our PS_DLM_CONID). Decode it and complete the pending request by recording the - * peer's status. INV-6: we record whatever the peer's executive returned -- a - * GRANT carrying SS$_UNSUPPORTED/SS$_NOSUCHDEV is NOT a lock grant; the LIVE - * transport worked, the lock did not. + * scsd_dlm_cli_msg_input - node-A's DLM CLIENT input routine (vms-164d rung-1b; + * vms-6ca rung H5). The GRANT / queued-reply the peer sent back lands here (dest + * Con.ID = our PS_DLM_CONID). Decode it and complete the pending request. + * + * Rung 1b: record the status and print DLMDONE (the round-trip proof). + * + * Rung H5 (armed by OVMX_DLM_H5): drive the block-then-grant-over-the-wire + * sequence, keyed on the request handle (req_lkid) and the reply status the + * MASTER genuinely sent -- never a local decision (INV-6): + * - GRANT for #1 (status SS$_NORMAL): our holder is granted; remember B's + * master handle for it, then send the SECOND, incompatible $ENQ (#2). + * - queued-reply for #2 (status VMS_DLM_STS_QUEUED, mode NL): dispatch it into + * OUR executive (VMS_DLM_OP_GRANT receive) so #2's ORIGIN record is genuinely + * PENDING, read it back (GETLKI -> NL), then $DEQ the holder (#1). + * - deferred GRANT for #2 (status SS$_NORMAL, mode EX): dispatch it into our + * executive; the origin record FLIPS NL->EX -- the status flip observed on + * the REQUESTER node, read back (GETLKI -> EX) as the H5 proof. */ static void scsd_dlm_cli_msg_input(struct scs_cdt *cdt, const void *msg, size_t msglen, void *ctx) @@ -8121,13 +8310,84 @@ static void scsd_dlm_cli_msg_input(struct scs_cdt *cdt, const void *msg, log_ts(stdout); printf(" SCSD-I-DLMDONE, cross-node $ENQ round-trip COMPLETE: peer" " (master CSID=%u) answered GRANT status=0x%08X resnam='%.*s'" - " -- LIVE A->B->A transport proven; %s\n", + " req_lkid=0x%08X mode=%s -- LIVE A->B->A transport proven; %s\n", (unsigned)v.msg.master_csid, (unsigned)v.msg.status, - (int)v.msg.namelen, v.msg.resnam, + (int)v.msg.namelen, v.msg.resnam, (unsigned)v.msg.req_lkid, + scs_dlm_mode_name(v.msg.mode), v.msg.status == 1u ? "cross-node lock GRANTED by the master (SS$_NORMAL)" : "lock NOT granted (honest)"); fflush(stdout); + + if (!ps->dlm_h5) + return; /* rung 1b only: no H5 sequence armed */ + + struct scsd_rx *rx = scsd_rx_current.rx; + if (rx == NULL || !scsd_member_initiate_enabled()) + return; + + /* GRANT for our holder (#1): remember B's master handle, then contend (#2). */ + if (v.msg.req_lkid == 1u && v.msg.status == 1u) { + ps->dlm_master_lkid1 = v.msg.master_lkid; + if (!ps->dlm_enq2_sent) { + if (scsd_dlm_client_send_op(rx, ps, SCS_DLM_OP_ENQ, LCK$K_EXMODE, + 2u /*req_lkid*/, 0, (const char *)v.msg.resnam, + "OVMX$DLM ENQ #2 (H5 contend)")) { + ps->dlm_enq2_sent = 1; + log_ts(stdout); + printf(" SCSD-I-DLMENQ2, sent SECOND cross-node $ENQ (req_lkid=2," + " EX) -- incompatible with our held #1, expect it to QUEUE" + " on the master (H5)\n"); + fflush(stdout); + } + } + return; + } + + /* Replies for our contending request (#2). */ + if (v.msg.req_lkid == 2u) { + /* Dispatch the reply into OUR executive so #2's origin record reflects + * the master's genuine status; read the record back (GETLKI). */ + uint32_t origin_mode = 0xFFu; + uint32_t st = scsd_dlm_dispatch_to_executive(&v.msg, NULL, NULL, + &origin_mode); + + if (v.msg.status == 0u) { /* VMS_DLM_STS_QUEUED: pending, not granted */ + /* QUEUED-reply: the request is genuinely PENDING on the requester. */ + ps->dlm_pend_seen = 1; + log_ts(stdout); + printf(" SCSD-I-DLMPEND, cross-node $ENQ #2 QUEUED on the master;" + " requester origin record PENDING (recv rc=0x%08X granted_mode=%s)" + " -- genuine block, read from OUR executive\n", + (unsigned)st, scs_dlm_mode_name((uint8_t)origin_mode)); + fflush(stdout); + /* Release the holder (#1): the real $DEQ that unblocks #2. */ + if (!ps->dlm_deq1_sent && ps->dlm_master_lkid1 != 0) { + if (scsd_dlm_client_send_op(rx, ps, SCS_DLM_OP_DEQ, LCK$K_NLMODE, + 0, ps->dlm_master_lkid1, (const char *)v.msg.resnam, + "OVMX$DLM DEQ #1 (H5 release holder)")) { + ps->dlm_deq1_sent = 1; + log_ts(stdout); + printf(" SCSD-I-DLMDEQ1, released holder #1 (master_lkid=0x%08X)" + " -- expect the master to GRANT #2 and WIRE the deferred" + " GRANT back (H5)\n", (unsigned)ps->dlm_master_lkid1); + fflush(stdout); + } + } + } else if (v.msg.status == 1u) { + /* DEFERRED GRANT: the origin record flips NL->EX -- THE H5 PROOF. */ + ps->dlm_flip_seen = 1; + log_ts(stdout); + printf(" SCSD-I-DLMH5FLIP, requester origin FLIPPED NL->%s for #2" + " (recv rc=0x%08X GETLKI granted_mode=%s) -- a QUEUED cross-node" + " request, GRANTED by a real remote $DEQ, delivered over the LIVE" + " SCS wire; block-then-grant proven on the REQUESTER node\n", + scs_dlm_mode_name((uint8_t)origin_mode), (unsigned)st, + scs_dlm_mode_name((uint8_t)origin_mode)); + fflush(stdout); + } + return; + } } /* @@ -8171,6 +8431,12 @@ static void scsd_dlm_send_enq(struct scsd_rx *rx, struct peer_state *ps) scs_cdt_set_handlers(cli, scsd_dlm_cli_msg_input, NULL, NULL, ps); } + /* vms-6ca (H5): arm the block-then-grant-over-the-wire sequence if requested. + * The cli input routine drives the rest (second $ENQ, $DEQ, deferred-grant + * flip) off the GRANT replies. A no-op without OVMX_DLM_H5 (rung 1b only). */ + if (getenv("OVMX_DLM_H5") != NULL) + ps->dlm_h5 = 1; + uint32_t dlm_server = (ps->joiner_remote_conid & 0xFFFFFFF0u) | OVMX_CONID_CLS_DLMSRV; @@ -13471,7 +13737,7 @@ int main(int argc, char **argv) m.namelen = (uint8_t)rn; } - uint32_t status = scsd_dlm_dispatch_to_executive(&m, NULL); + uint32_t status = scsd_dlm_dispatch_to_executive(&m, NULL, NULL, NULL); printf("SCSD-I-DLMSELFTEST, executive DLM dispatch status=%u" " (0x%08X)\n", (unsigned)status, (unsigned)status); fflush(stdout); From 8cee1a0c5ad8daa81739b6fd775793ac73a4e737 Mon Sep 17 00:00:00 2001 From: alice Date: Fri, 28 Aug 2026 22:34:48 +0000 Subject: [PATCH 3/7] =?UTF-8?q?vms-6ca:=20DLM=20H5=20two-node=20QEMU=20har?= =?UTF-8?q?ness=20+=20CI=20job=20=E2=80=94=20block-then-grant=20over=20the?= =?UTF-8?q?=20wire?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Extends the H4 two-node real-/dev/vms harness to H5. Both nodes boot vms.ko and complete the VMS$VAXcluster join; node A is armed OVMX_DLM_ENQ=RESONE OVMX_DLM_H5=1 and drives the block-then-grant sequence over the live SCS wire: holds RESONE EX, a second incompatible $ENQ QUEUES on node B (B WIREs a queued-reply -> A's requester origin record PENDING, SCSD-I-DLMPEND), A releases the holder, node B WIREs the deferred GRANT off that real $DEQ (SCSD-I-DLMDEFER), and A's origin record FLIPS NL->EX (SCSD-I-DLMH5FLIP granted_mode=EX). H5 PASS asserts the status flip on the REQUESTER node A across the wire, plus B's deferred-grant wire — read from the nodes' own SCSD logs, never fabricated (INV-6). New Dockerfile.dlm-harness-h5 + init_dlm_h5.sh + run_dlm_harness_h5.sh; new dlm-harness-h5 CI job (path-filtered on the DLM sources + harness files); the existing dlm_harness filter already covers vms_lock.c + src/vmsscs/**. Shell + YAML syntax validated. Co-Authored-By: Claude Opus 4.8 (1M context) --- .github/workflows/ci.yml | 87 ++++++++++++ tests/qemu/Dockerfile.dlm-harness-h5 | 121 ++++++++++++++++ tests/qemu/init_dlm_h5.sh | 168 ++++++++++++++++++++++ tests/qemu/run_dlm_harness_h5.sh | 205 +++++++++++++++++++++++++++ 4 files changed, 581 insertions(+) create mode 100644 tests/qemu/Dockerfile.dlm-harness-h5 create mode 100755 tests/qemu/init_dlm_h5.sh create mode 100755 tests/qemu/run_dlm_harness_h5.sh diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 245a4f920..c39942ba4 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -399,6 +399,14 @@ jobs: - 'tests/qemu/Dockerfile.dlm-harness-h4' - 'tests/qemu/init_dlm_h4.sh' - 'tests/qemu/run_dlm_harness_h4.sh' + # H5 (rd vms-6ca, THE ASYNC-REPLY WIRE): after the join, node A holds + # RESONE then a second incompatible $ENQ QUEUES on node B; B WIREs a + # queued-reply (A's origin record PENDING) then, on A's $DEQ, WIREs + # the deferred GRANT -> A's origin record flips NL->EX. Proves the + # block-then-grant on the REQUESTER node across the live SCS wire. + - 'tests/qemu/Dockerfile.dlm-harness-h5' + - 'tests/qemu/init_dlm_h5.sh' + - 'tests/qemu/run_dlm_harness_h5.sh' - 'tests/cluster/two-ovmx/mk_sysgen_scratch.py' - 'tests/cluster/two-ovmx/verdict.sh' @@ -2604,6 +2612,85 @@ jobs: path: h4out/ if-no-files-found: ignore + # ----------------------------------------------------------------------- + # DLM Harness H5 (rd vms-6ca) -- THE ASYNC-REPLY SCS WIRE: block-then-grant on + # the REQUESTER node. tests/qemu/Dockerfile.dlm-harness-h5 builds the SAME + # real-executive ingredients as H4 (vms.ko -- now with the requester-side GRANT + # RECEIVE + the deferred-grant report -- SCSD.EXE static musl with the H5 wire + # sequence + sca_l2probe + the two OVMXA/OVMXB SYSGEN stores) and boots the SAME + # TWO QEMU nodes, each with a real /dev/vms, on one shared L2 via a QEMU `socket` + # (mcast) netdev. It runs the SAME join sequencer, then arms node A's + # block-then-grant sequence (OVMX_DLM_ENQ=RESONE OVMX_DLM_H5=1): node A holds + # RESONE, a second incompatible $ENQ QUEUES on node B (B WIREs a queued-reply -> + # A's origin record PENDING), A releases the holder, and node B WIREs the + # deferred GRANT off that real $DEQ -> A's origin record FLIPS NL->EX. H5 PASS + # iff node A prints SCSD-I-DLMH5FLIP (granted_mode=EX -- the flip read back from + # A's OWN executive via GETLKI) and node B prints SCSD-I-DLMDEFER. INV-6: the + # verdict READS A's flip + B's deferred-grant from the nodes' own logs; a + # fabricated wire reply or grant cannot pass. The BLKAST wire is deferred. + # ----------------------------------------------------------------------- + dlm-harness-h5: + name: DLM Harness H5 (block-then-grant proven on the REQUESTER over the SCS wire, QEMU) + needs: changes + if: ${{ github.event_name != 'pull_request' || needs.changes.outputs.dlm_harness == 'true' }} + runs-on: ubuntu-latest + # vms.ko + SCSD.EXE build + TWO QEMU boots, each running the join sequencer + + # the multi-step H5 sequence for ~90s. The inner H5_WALL_TIMEOUT (600s) is the + # real budget; this is the OUTER bound so a genuine hang surfaces. + timeout-minutes: 50 + steps: + - name: Checkout + uses: actions/checkout@v4 + + - name: Set up Docker Buildx + uses: docker/setup-buildx-action@v3 + + - name: Build DLM harness H5 image (gha layer cache) + uses: docker/build-push-action@v6 + with: + context: . + file: tests/qemu/Dockerfile.dlm-harness-h5 + push: false + load: true + tags: ovmx-dlm-h5:latest + cache-from: type=gha,scope=dlm-harness-h5 + cache-to: type=gha,mode=max,scope=dlm-harness-h5 + + - name: Boot TWO QEMU nodes, drive node A's block-then-grant over the SCS wire + run: | + set +e + # GitHub Linux runners expose /dev/kvm; pass it through so the two + # guests boot under KVM (much faster than TCG). Fall back cleanly if + # a runner lacks it -- the harness auto-detects and uses TCG. + KVM="" + [ -e /dev/kvm ] && KVM="--device /dev/kvm" + mkdir -p "$PWD/h5out" + RAW=$(docker run --rm $KVM -v "$PWD/h5out:/out" ovmx-dlm-h5:latest 2>&1) + RC=$? + set -e + # QEMU serial emits CRLF; strip \r so the line-anchored match holds. + OUTPUT=$(printf '%s\n' "$RAW" | tr -d '\r') + echo "$OUTPUT" + if [ "$RC" -ne 0 ]; then + echo "FAIL: DLM harness H5 exited nonzero" + exit 1 + fi + if printf '%s\n' "$OUTPUT" | grep -qE '^ DLM HARNESS H5 PASSED'; then + echo "PASS: node A's QUEUED cross-node request GRANTED on the REQUESTER over the live SCS wire (SCSD-I-DLMH5FLIP granted_mode=EX)." + else + echo "FAIL: H5 verdict 'DLM HARNESS H5 PASSED' not found." + printf '%s\n' "$OUTPUT" | grep -E 'verdict inputs|A_FLIP|granted_mode|A_PEND|B_DEFER|highest join rung|VAXCLMEMBER|DLMENQ|DLMPEND|DLMDEFER|DLMH5FLIP|H5-NODE|MISS:|FAIL:' || true + exit 1 + fi + + - name: Upload H5 pcap + node logs + if: always() + uses: actions/upload-artifact@v4 + with: + name: dlm-harness-h5-artifacts + path: h5out/ + if-no-files-found: ignore + # ----------------------------------------------------------------------- # Job 3b: Kernel Executive (vms-e4d, epic vms-6b8 PHASE 0 HARD BARRIER) # diff --git a/tests/qemu/Dockerfile.dlm-harness-h5 b/tests/qemu/Dockerfile.dlm-harness-h5 new file mode 100644 index 000000000..cb06d81f0 --- /dev/null +++ b/tests/qemu/Dockerfile.dlm-harness-h5 @@ -0,0 +1,121 @@ +# Dockerfile.dlm-harness-h5 - DLM cross-node harness, rung H5 (rd vms-6ca) -- +# THE ASYNC-REPLY SCS WIRE: block-then-grant proven on the REQUESTER node. +# +# The wire sibling of tests/qemu/Dockerfile.dlm-harness-h4. It builds the SAME +# real-executive ingredients -- vms.ko (the executive, now with the requester-side +# GRANT RECEIVE + the deferred-grant report in vms_lock_dlm_xnode_dispatch/_deq) +# + SCSD.EXE (static musl, now with the H5 wire sequence) + sca_l2probe -- bakes +# the SAME TWO per-node SYSGEN identity stores (OVMXA/OVMXB), assembles ONE +# symmetric initramfs, and boots TWO QEMU nodes wired on one shared L2 by a QEMU +# `socket` (mcast) netdev -- NO host bridge, NO privilege. It runs the SAME join +# sequencer as H2/H3/H4 (OVMX_MCAST_SOLICIT=1 OVMX_JOIN_SEQ=1) so both nodes reach +# VMS$VAXcluster membership, then the H5 init arms node A's block-then-grant +# sequence over the LIVE VC (OVMX_DLM_ENQ=RESONE OVMX_DLM_H5=1). +# +# WHAT H5 PROVES (the async-reply WIRE, on top of H4's single cross-node GRANT): +# node A holds RESONE EX, then a SECOND incompatible $ENQ QUEUES on node B (the +# master WIREs a queued-reply -> A's requester origin record is genuinely +# PENDING); A releases the holder; node B WIREs the deferred GRANT -> A's origin +# record FLIPS NL->EX. H5 PASS iff node A prints SCSD-I-DLMH5FLIP (the origin +# record read back GRANTED at EX via GETLKI, on the REQUESTER node, driven only by +# what the master sent over SCS) and node B prints SCSD-I-DLMDEFER (it WIRED the +# deferred GRANT off a real $DEQ). See tests/qemu/init_dlm_h5.sh + +# run_dlm_harness_h5.sh. The BLKAST wire is deferred honestly (the holder releases +# on its own); nothing here fabricates a wire reply or a grant (INV-6). +# +# Usage: +# docker build -f tests/qemu/Dockerfile.dlm-harness-h5 -t ovmx-dlm-h5 . +# docker run --rm --device /dev/kvm ovmx-dlm-h5 # --device optional (TCG fallback) +# Exit 0 = H5 PASS (block-then-grant proven on the requester over the wire), 1 = FAIL. + +FROM ubuntu:24.04 +ENV DEBIAN_FRONTEND=noninteractive + +# Same inputs as tests/qemu/Dockerfile.dlm-harness-h4, plus python3 to bake the +# two SYSGEN stores at build time (the busybox initramfs has no python). +RUN apt-get update && \ + ARCH=$(dpkg --print-architecture) && \ + if [ "$ARCH" = "arm64" ]; then QEMU_PKG=qemu-system-arm; else QEMU_PKG=qemu-system-x86; fi && \ + apt-get install -y --no-install-recommends \ + make gcc libc6-dev kmod \ + linux-headers-generic \ + linux-image-virtual \ + busybox-static \ + cpio \ + python3 \ + cmake musl-tools linux-libc-dev \ + "$QEMU_PKG" \ + && rm -rf /var/lib/apt/lists/* + +RUN KVER=$(ls /lib/modules/ | sort -V | tail -1) && echo "$KVER" > /tmp/kver && echo "Kernel version: $KVER" + +# vms.ko: the real executive (same inputs as tests/qemu/Dockerfile.dlm-harness-h4). +COPY src/kernel-core/ /src/kernel-core/ +COPY src/kernel/ /src/kernel/ +COPY src/vmsfs/include/ /src/vmsfs/include/ +COPY src/vmsfs/ods2/ /src/vmsfs/ods2/ +RUN KVER=$(cat /tmp/kver) && \ + cd /src/kernel && \ + make KDIR=/lib/modules/${KVER}/build 2>&1 && \ + ls -la /src/kernel/vms.ko && \ + modinfo /src/kernel/vms.ko + +# musl UAPI header symlinks (same trick as tests/qemu/Dockerfile.dlm-harness-h4). +RUN MUSL_INC=$(echo /usr/include/*-linux-musl) && \ + ARCH_INC=$(echo /usr/include/*-linux-gnu) && \ + ln -s /usr/include/linux "$MUSL_INC/linux" && \ + ln -s /usr/include/asm-generic "$MUSL_INC/asm-generic" && \ + ln -s "$ARCH_INC/asm" "$MUSL_INC/asm" + +# SCSD.EXE, static musl (OVMX_STATIC=ON) -- the real SCS datalink daemon. +COPY . /src/repo/ +RUN cd /src/repo && \ + cmake -B build-static \ + -DCMAKE_C_COMPILER=musl-gcc \ + -DCMAKE_BUILD_TYPE=Release \ + -DOVMX_STATIC=ON \ + && cmake --build build-static --target scsd_exe --parallel $(nproc) 2>&1 && \ + { [ -f build-static/bin/SCSD.EXE ] || { echo "FATAL: SCSD.EXE was not built"; exit 1; }; } && \ + ls -la build-static/bin/SCSD.EXE && \ + { readelf -d build-static/bin/SCSD.EXE 2>/dev/null | grep -q NEEDED && \ + { echo "FATAL: staged SCSD.EXE is not static"; exit 1; } || true; } + +# sca_l2probe: the tiny static raw-L2 pcap capturer. +RUN musl-gcc -static -O2 -Wall -Wextra /src/repo/tests/qemu/sca_l2probe.c -o /src/sca_l2probe && \ + { readelf -d /src/sca_l2probe 2>/dev/null | grep -q NEEDED && \ + { echo "FATAL: sca_l2probe is not static"; exit 1; } || true; } + +# The two per-node SYSGEN identity stores (distinct SCSNODE + SCSSYSTEMID), baked +# from the SAME generator the two-ovmx harness uses. ALLOCLASS=1, RECNXINTERVAL=20. +RUN mkdir -p /src/sysgen && \ + python3 /src/repo/tests/cluster/two-ovmx/mk_sysgen_scratch.py /src/sysgen/sysgen-A.dat OVMXA 1601 1 20 && \ + python3 /src/repo/tests/cluster/two-ovmx/mk_sysgen_scratch.py /src/sysgen/sysgen-B.dat OVMXB 1602 1 20 + +RUN KVER=$(cat /tmp/kver) && \ + if [ ! -e /boot/vmlinuz ]; then ln -s vmlinuz-${KVER} /boot/vmlinuz; fi && \ + ls -la /boot/vmlinuz + +# Assemble ONE symmetric initramfs: busybox, vms.ko, SCSD.EXE, sca_l2probe, both +# identity stores, and the H5 init (which selects its identity by ovmx.node=, +# turns the join sequencer ON, and arms node A's block-then-grant sequence). +COPY tests/qemu/init_dlm_h5.sh /src/init_dlm_h5.sh +RUN mkdir -p /initramfs/bin /initramfs/lib/modules /initramfs/etc/ovmx \ + /initramfs/dev /initramfs/proc /initramfs/sys /initramfs/tmp && \ + cp /bin/busybox /initramfs/bin/busybox && \ + ln -s busybox /initramfs/bin/sh && \ + cp /src/kernel/vms.ko /initramfs/lib/modules/ && \ + cp /src/repo/build-static/bin/SCSD.EXE /initramfs/bin/SCSD.EXE && \ + chmod +x /initramfs/bin/SCSD.EXE && \ + cp /src/sca_l2probe /initramfs/bin/sca_l2probe && \ + chmod +x /initramfs/bin/sca_l2probe && \ + cp /src/sysgen/sysgen-A.dat /initramfs/etc/ovmx/sysgen-A.dat && \ + cp /src/sysgen/sysgen-B.dat /initramfs/etc/ovmx/sysgen-B.dat && \ + cp /src/init_dlm_h5.sh /initramfs/init && \ + chmod +x /initramfs/init && \ + cd /initramfs && find . | cpio -o -H newc 2>/dev/null | gzip > /initramfs.cpio.gz && \ + echo "Initramfs: $(ls -lh /initramfs.cpio.gz | awk '{print $5}')" + +COPY tests/qemu/run_dlm_harness_h5.sh /run_dlm_harness_h5.sh +RUN chmod +x /run_dlm_harness_h5.sh + +CMD ["/run_dlm_harness_h5.sh"] diff --git a/tests/qemu/init_dlm_h5.sh b/tests/qemu/init_dlm_h5.sh new file mode 100755 index 000000000..9305d2284 --- /dev/null +++ b/tests/qemu/init_dlm_h5.sh @@ -0,0 +1,168 @@ +#!/bin/busybox sh +# init_dlm_h5.sh - PID 1 inside a DLM harness H5 OVMX node (rd vms-6ca). +# +# H5 (DLM epic vms-7fa, harness rung 5 -- THE ASYNC-REPLY SCS WIRE) takes H4's two +# joined real-/dev/vms QEMU nodes and, AFTER both reach SCSD-I-VAXCLMEMBER, drives +# node A's block-then-grant-over-the-wire sequence (OVMX_DLM_ENQ=RESONE +# OVMX_DLM_H5=1) for a resource mastered on node B: +# +# 1. node A holds RESONE EX (#1), granted by B's executive (the H4 grant). +# 2. node A sends a SECOND, incompatible $ENQ (#2); B QUEUES it on its real +# waiting queue and WIREs a queued-reply -> A dispatches it into ITS OWN +# executive, so #2's requester ORIGIN record is genuinely PENDING (GETLKI->NL, +# SCSD-I-DLMPEND). +# 3. node A releases the holder (#1) with a real cross-node $DEQ (SCSD-I-DLMDEQ1). +# 4. B releases #1, GRANTS #2, and WIREs the deferred GRANT to A (SCSD-I-DLMDEFER +# on B) -> A dispatches it into its executive and #2's origin record FLIPS +# NL->EX (GETLKI->EX, SCSD-I-DLMH5FLIP on A). +# +# WHAT H5 PROVES (the WIRE, on top of H4's single cross-node grant): the status +# flip is observed on the REQUESTER node A, across the LIVE SCS wire, driven ONLY +# by what the master genuinely sent -- a real queued-reply and a real deferred +# GRANT off a real remote $DEQ. The BLKAST wire is deferred honestly (the holder +# releases on its own). +# +# INV-6 / Rule 9: nothing here fakes a pass. If vms.ko does not yield /dev/vms the +# node FAILS loudly; SCSD prints verbatim what its own executive returned via +# GETLKI on the origin record; the host verdict reads A's SCSD-I-DLMH5FLIP and B's +# SCSD-I-DLMDEFER from the nodes' own SCSD logs -- never fabricates them. The flip +# happens in A's EXECUTIVE on a real /dev/vms. + +/bin/busybox --install -s /bin + +mount -t proc none /proc +mount -t sysfs none /sys +mount -t devtmpfs none /dev + +NODE=$(sed -n 's/.*ovmx.node=\([A-Za-z0-9]*\).*/\1/p' /proc/cmdline) +[ -z "$NODE" ] && NODE=X +DURATION=$(sed -n 's/.*ovmx.duration=\([0-9]*\).*/\1/p' /proc/cmdline) +[ -z "$DURATION" ] && DURATION=90 + +echo "" +echo "=== OVMX DLM Harness H5: block-then-grant over the SCS wire (node=$NODE) ===" +echo "Kernel: $(uname -r) ($(uname -m))" + +# --- 1. the real executive --------------------------------------------------- +echo "--- Loading vms.ko ---" +insmod /lib/modules/vms.ko +if [ -c /dev/vms ]; then + echo " PASS: vms.ko loaded, /dev/vms present (char device)" +else + echo " FAIL: /dev/vms is NOT a char device -- no real executive present" + dmesg | tail -20 + echo "H5-NODE-$NODE: FAIL (no /dev/vms)" > /dev/ttyS1 + poweroff -f +fi + +# --- 2. the shared L2 NIC ---------------------------------------------------- +ip link set eth0 up 2>/dev/null || ifconfig eth0 up 2>/dev/null +sleep 2 +MAC=$(cat /sys/class/net/eth0/address 2>/dev/null) +echo "eth0 mac=$MAC" + +# --- 3. this node's cluster identity (distinct SCSNODE/SCSSYSTEMID) ----------- +STORE=/etc/ovmx/sysgen-$NODE.dat +if [ ! -f "$STORE" ]; then + echo " FAIL: identity store $STORE missing" + echo "H5-NODE-$NODE: FAIL (no SYSGEN store)" > /dev/ttyS1 + poweroff -f +fi +export OVMX_SYSGEN_PATH="$STORE" + +# --- 3b. the join sequencer (SAME flags H2/H3/H4 complete the join with) ------ +export OVMX_MCAST_SOLICIT=1 +export OVMX_JOIN_SEQ=1 + +# --- 3c. THE H5 DELTA: arm node A's block-then-grant-over-the-wire sequence ---- +# Only node A drives the sequence; RESONE is mastered on node B. OVMX_DLM_ENQ arms +# the first $ENQ (identical to H4); OVMX_DLM_H5 arms the requester-side sequence +# (second $ENQ that queues, the $DEQ, and the origin-record flip). Node B runs a +# pure DLM server that GRANTS/QUEUES in its real executive and WIREs both the +# queued-reply and the deferred GRANT -- it needs no extra env. +if [ "$NODE" = "A" ]; then + export OVMX_DLM_ENQ=RESONE + export OVMX_DLM_H5=1 + echo "join sequencer ON; node A ARMED: OVMX_DLM_ENQ=RESONE OVMX_DLM_H5=1 (block-then-grant over the wire)" +else + echo "join sequencer ON; node $NODE is the DLM SERVER for RESONE (QUEUES + WIREs the deferred GRANT)" +fi + +if [ ! -x /bin/SCSD.EXE ]; then + echo " FAIL: /bin/SCSD.EXE missing" + echo "H5-NODE-$NODE: FAIL (no SCSD.EXE)" > /dev/ttyS1 + poweroff -f +fi + +# --- 4. passive pcap capture of the 0x6007 wire (artifact) ------------------- +if [ -x /bin/sca_l2probe ]; then + sca_l2probe recv eth0 $((DURATION + 3)) /tmp/$NODE.pcap > /tmp/pcap.log 2>&1 & + PCAP_PID=$! + sleep 1 +fi + +# --- 5. the real SCS datalink daemon, driving the join + the H5 sequence ------ +echo "--- SCSD.EXE --connect --iface eth0 --duration $DURATION (node $NODE) ---" +SCSD.EXE --connect --iface eth0 --duration "$DURATION" > /tmp/scsd-$NODE.log 2>&1 +scsd_rc=$? +echo "(SCSD.EXE exit code: $scsd_rc)" + +[ -n "${PCAP_PID:-}" ] && wait "$PCAP_PID" 2>/dev/null + +# --- 6. emit the machine-checkable node log on ttyS1 ------------------------- +# Join ladder markers (so a join stall is visible), PLUS the H5 wire markers the +# host verdict reads: +# node A: SCSD-I-DLMENQ (A sent the first $ENQ #1) +# SCSD-I-DLMENQ2 (A sent the second, contending $ENQ #2) +# SCSD-I-DLMPEND (A's origin record for #2 is PENDING -- genuine block) +# SCSD-I-DLMDEQ1 (A released the holder #1) +# SCSD-I-DLMH5FLIP (A's origin record for #2 FLIPPED NL->EX -- THE PROOF) +# node B: SCSD-I-DLMRX (B received + dispatched the cross-node requests) +# SCSD-I-DLMDEFER (B WIRED the deferred GRANT off a real $DEQ) +# These lines are lifted VERBATIM from THIS node's SCSD stdout -- never synthesised. +{ + echo "===H5-NODE-$NODE-BEGIN===" + echo "node=$NODE mac=$MAC store=$STORE scsd_rc=$scsd_rc" + if [ "$NODE" = "A" ]; then + echo "role=requester dlm_enq=RESONE dlm_h5=1" + else + echo "role=dlm_server res=RESONE" + fi + echo "--- join ladder markers (chronological NEW->MEMBER) ---" + for KEY in SCSD-I-HELLOSENT SCSD-I-DIRHELLO SCSD-I-STARTTX SCSD-I-STARTDONE \ + SCSD-I-VCOPEN SCSD-I-OWNDIRBOUND SCSD-I-MSCPBOUND \ + SCSD-I-CONNRESP SCSD-I-VAXCLMEMBER SCSD-I-CMCONFIG; do + if grep -qa "$KEY" /tmp/scsd-$NODE.log; then + echo " REACHED $KEY" + else + echo " ------- $KEY" + fi + done + echo "--- VAXCLMEMBER (membership complete on this node) ---" + grep -a 'SCSD-I-VAXCLMEMBER' /tmp/scsd-$NODE.log + echo "--- DLM H5 block-then-grant-over-the-wire markers (verbatim from SCSD) ---" + grep -a 'SCSD-I-DLMENQ' /tmp/scsd-$NODE.log + grep -a 'SCSD-I-DLMENQ2' /tmp/scsd-$NODE.log + grep -a 'SCSD-I-DLMRX' /tmp/scsd-$NODE.log + grep -a 'SCSD-I-DLMPEND' /tmp/scsd-$NODE.log + grep -a 'SCSD-I-DLMDEQ1' /tmp/scsd-$NODE.log + grep -a 'SCSD-I-DLMDEFER' /tmp/scsd-$NODE.log + grep -a 'SCSD-I-DLMH5FLIP' /tmp/scsd-$NODE.log + echo "--- IDENT ---" + grep -a 'SCSD-I-IDENT' /tmp/scsd-$NODE.log + echo "--- pcap capture summary ---" + grep -a 'SCA-L2PROBE-DONE' /tmp/pcap.log 2>/dev/null + echo "===H5-NODE-$NODE-END===" +} > /dev/ttyS1 2>&1 + +# --- 7. pcap artifact, base64 on a dedicated UART ---------------------------- +if [ -f /tmp/$NODE.pcap ]; then + { + echo "===PCAP-$NODE-B64-BEGIN===" + base64 /tmp/$NODE.pcap + echo "===PCAP-$NODE-B64-END===" + } > /dev/ttyS2 2>&1 +fi + +sync +poweroff -f diff --git a/tests/qemu/run_dlm_harness_h5.sh b/tests/qemu/run_dlm_harness_h5.sh new file mode 100755 index 000000000..fc0cd74ed --- /dev/null +++ b/tests/qemu/run_dlm_harness_h5.sh @@ -0,0 +1,205 @@ +#!/bin/bash +# run_dlm_harness_h5.sh - host side (runs INSIDE the container built from +# tests/qemu/Dockerfile.dlm-harness-h5): boot TWO OVMX QEMU nodes, EACH with a +# real /dev/vms, wired on ONE shared L2 by a QEMU `socket` (mcast) netdev -- NO +# host bridge, NO privilege -- complete the full VMS$VAXcluster join, then drive +# node A's BLOCK-THEN-GRANT-over-the-wire sequence (OVMX_DLM_ENQ=RESONE +# OVMX_DLM_H5=1) and verdict that the QUEUED cross-node request GRANTED on the +# REQUESTER node A, across the live SCS wire, driven by a real remote $DEQ +# (rd vms-6ca, DLM epic vms-7fa rung H5). +# +# H5 PASS iff, on top of the H2 join precondition (both nodes VAXCLMEMBER): +# 1. node A held RESONE (SCSD-I-DLMENQ) and CONTENDED (SCSD-I-DLMENQ2). +# 2. ⭐ node A's requester origin record for #2 was genuinely PENDING after B's +# queued-reply (SCSD-I-DLMPEND) -- a real block read from A's OWN executive, +# not a userspace flag. +# 3. node A released the holder with a real cross-node $DEQ (SCSD-I-DLMDEQ1). +# 4. ⭐ node B WIRED the deferred GRANT off that real $DEQ (SCSD-I-DLMDEFER). +# 5. ⭐⭐ node A's origin record for #2 FLIPPED NL->EX (SCSD-I-DLMH5FLIP with +# granted_mode=EX) -- the block-then-grant status flip observed on the +# REQUESTER node, across the wire, driven ONLY by what the master sent. THIS +# is the milestone. +# +# INV-6 / Rule 9: the verdict READS A's flip + B's deferred-grant from the nodes' +# own SCSD logs; it never fabricates them. The flip happens in node A's EXECUTIVE +# on a real /dev/vms (GETLKI on the origin record); the wire delivery is real SCS. +# The BLKAST wire is deferred honestly (the holder releases on its own). + +set -uo pipefail + +DURATION="${H5_DURATION:-90}" +NETDEV="${H5_NETDEV:-mcast}" # mcast (default) | sockpair +WALL="${H5_WALL_TIMEOUT:-600}" +OUT="${OUT_DIR:-/out}" +mkdir -p "$OUT" + +KERNEL=/boot/vmlinuz +INITRD=/initramfs.cpio.gz +ARCH=$(uname -m) +if [ "$ARCH" = "aarch64" ] || [ "$ARCH" = "arm64" ]; then + QEMU=qemu-system-aarch64 + MACHINE="-machine virt -cpu cortex-a57" + CONSOLE="console=ttyAMA0" +else + QEMU=qemu-system-x86_64 + if [ -w /dev/kvm ]; then MACHINE="-accel kvm -cpu host"; else MACHINE="-accel tcg"; fi + CONSOLE="console=ttyS0" +fi + +MAC_A=52:54:00:00:00:0a +MAC_B=52:54:00:00:00:0b +GROUP=230.0.0.8 +PORT=16008 + +echo "=== OVMX DLM Harness H5 Runner (vms-6ca) ===" +echo "arch=$ARCH qemu=$QEMU accel=${MACHINE#-accel } netdev=$NETDEV duration=${DURATION}s wall=${WALL}s" +echo "join sequencer: OVMX_MCAST_SOLICIT=1 OVMX_JOIN_SEQ=1; node A armed OVMX_DLM_ENQ=RESONE OVMX_DLM_H5=1" +echo "milestone: node A's QUEUED cross-node request GRANTS on the REQUESTER (SCSD-I-DLMH5FLIP granted_mode=EX)" +echo " after node B WIREs the deferred GRANT off a real \$DEQ (SCSD-I-DLMDEFER)" +echo "" + +netdev_arg() { + case "$NETDEV" in + mcast) echo "socket,id=net0,mcast=${GROUP}:${PORT},localaddr=127.0.0.1" ;; + sockpair) if [ "$1" = "A" ]; then echo "socket,id=net0,listen=127.0.0.1:${PORT}"; + else echo "socket,id=net0,connect=127.0.0.1:${PORT}"; fi ;; + *) echo "UNKNOWN" ;; + esac +} + +LAUNCH_PID=0 +launch_node() { + local role="$1" mac="$2" node="$3" + local nd; nd=$(netdev_arg "$role") + if [ "$nd" = "UNKNOWN" ]; then + echo "FATAL: unknown H5_NETDEV='$NETDEV' (want mcast|sockpair)"; exit 2 + fi + # ttyS0=console(file), ttyS1=node verdict log(file), ttyS2=pcap-b64(file). + $QEMU $MACHINE \ + -kernel "$KERNEL" -initrd "$INITRD" \ + -append "$CONSOLE net.ifnames=0 biosdevname=0 panic=-1 loglevel=4 ovmx.node=${node} ovmx.duration=${DURATION}" \ + -m 512M -smp 1 -nographic -no-reboot -nodefaults \ + -netdev "$nd" \ + -device "virtio-net-pci,netdev=net0,mac=${mac},romfile=" \ + -serial "file:$OUT/node${node}.console.log" \ + -serial "file:$OUT/node${node}.ttyS1.log" \ + -serial "file:$OUT/node${node}.pcap.b64" \ + >/dev/null 2>&1 & + LAUNCH_PID=$! +} + +echo "--- booting node A (SCSNODE OVMXA, mac=$MAC_A, requester / block-then-grant driver) ---" +launch_node A "$MAC_A" A; PA=$LAUNCH_PID +sleep 2 +echo "--- booting node B (SCSNODE OVMXB, mac=$MAC_B, DLM server / master of RESONE) ---" +launch_node B "$MAC_B" B; PB=$LAUNCH_PID + +( sleep "$WALL"; kill -9 "$PA" "$PB" 2>/dev/null ) & +GUARD=$! +wait "$PA" 2>/dev/null +wait "$PB" 2>/dev/null +kill "$GUARD" 2>/dev/null + +echo "" +echo "=== node A log (ttyS1) ==="; cat "$OUT/nodeA.ttyS1.log" 2>/dev/null || echo "(none)" +echo "" +echo "=== node B log (ttyS1) ==="; cat "$OUT/nodeB.ttyS1.log" 2>/dev/null || echo "(none)" + +# Reconstruct the pcap artifacts. +for N in A B; do + B64="$OUT/node${N}.pcap.b64" + if [ -s "$B64" ]; then + sed -n '/===PCAP-'"$N"'-B64-BEGIN===/,/===PCAP-'"$N"'-B64-END===/p' "$B64" \ + | grep -v '===PCAP-' | tr -d '\r' | base64 -d > "$OUT/node${N}.pcap" 2>/dev/null || true + [ -s "$OUT/node${N}.pcap" ] && \ + echo "reconstructed pcap: $OUT/node${N}.pcap ($(wc -c < "$OUT/node${N}.pcap") bytes)" + fi +done + +# --- verdict ----------------------------------------------------------------- +LA="$OUT/nodeA.ttyS1.log"; LB="$OUT/nodeB.ttyS1.log" + +# Precondition: the H2 join must have completed on both nodes. +A_VAXCLMEMBER=0; B_VAXCLMEMBER=0 +grep -qa 'REACHED SCSD-I-VAXCLMEMBER\|SCSD-I-VAXCLMEMBER,' "$LA" 2>/dev/null && A_VAXCLMEMBER=1 +grep -qa 'REACHED SCSD-I-VAXCLMEMBER\|SCSD-I-VAXCLMEMBER,' "$LB" 2>/dev/null && B_VAXCLMEMBER=1 + +# 1. node A held (#1) and contended (#2). +A_ENQ=0; grep -qa 'SCSD-I-DLMENQ,' "$LA" 2>/dev/null && A_ENQ=1 +A_ENQ2=0; grep -qa 'SCSD-I-DLMENQ2' "$LA" 2>/dev/null && A_ENQ2=1 +# 2. node A saw its origin record PENDING (genuine block on the requester). +A_PEND=0; grep -qa 'SCSD-I-DLMPEND' "$LA" 2>/dev/null && A_PEND=1 +# 3. node A released the holder. +A_DEQ1=0; grep -qa 'SCSD-I-DLMDEQ1' "$LA" 2>/dev/null && A_DEQ1=1 +# 4. node B WIRED the deferred grant. +B_DEFER=0; grep -qa 'SCSD-I-DLMDEFER' "$LB" 2>/dev/null && B_DEFER=1 +# 5. ⭐ node A's origin record FLIPPED NL->EX. THE MILESTONE. +A_FLIP=0; grep -qa 'SCSD-I-DLMH5FLIP' "$LA" 2>/dev/null && A_FLIP=1 +A_FLIP_MODE=$(grep -a 'SCSD-I-DLMH5FLIP' "$LA" 2>/dev/null \ + | sed -n 's/.*granted_mode=\([A-Za-z0-9]*\).*/\1/p' | tail -1) + +# Highest join rung each node climbed (for a stall diagnosis in CI output). +ladder="SCSD-I-HELLOSENT SCSD-I-DIRHELLO SCSD-I-STARTTX SCSD-I-STARTDONE SCSD-I-VCOPEN SCSD-I-OWNDIRBOUND SCSD-I-MSCPBOUND SCSD-I-CONNRESP SCSD-I-VAXCLMEMBER SCSD-I-CMCONFIG" +highest() { + local log="$1" top="" + for k in $ladder; do + grep -qa "REACHED $k" "$log" 2>/dev/null && top="$k" + done + echo "$top" +} + +echo "" +echo "highest join rung A : $(highest "$LA")" +echo "highest join rung B : $(highest "$LB")" +echo "" +echo "verdict inputs:" +echo " A_VAXCLMEMBER=$A_VAXCLMEMBER B_VAXCLMEMBER=$B_VAXCLMEMBER (H2 join precondition)" +echo " A_ENQ=$A_ENQ A_ENQ2=$A_ENQ2 (node A held RESONE, then contended with a second \$ENQ)" +echo " A_PEND=$A_PEND (node A's origin record for #2 was PENDING -- genuine block on the requester)" +echo " A_DEQ1=$A_DEQ1 (node A released the holder with a real cross-node \$DEQ)" +echo " B_DEFER=$B_DEFER (node B WIRED the deferred GRANT off the real \$DEQ)" +echo " A_FLIP=$A_FLIP granted_mode=${A_FLIP_MODE:-} (node A's origin record FLIPPED NL->EX -- want EX)" +echo "" + +FAIL=0 +[ "$A_VAXCLMEMBER" = 1 ] && [ "$B_VAXCLMEMBER" = 1 ] || { echo " MISS: H2 join precondition not met on both nodes"; FAIL=1; } +[ "$A_ENQ" = 1 ] || { echo " MISS: node A did not send the first \$ENQ (no SCSD-I-DLMENQ)"; FAIL=1; } +[ "$A_ENQ2" = 1 ] || { echo " MISS: node A did not contend with a second \$ENQ (no SCSD-I-DLMENQ2)"; FAIL=1; } +[ "$A_PEND" = 1 ] || { echo " MISS: node A's origin record was never PENDING (no SCSD-I-DLMPEND) -- the queued-reply wire did not land"; FAIL=1; } +[ "$A_DEQ1" = 1 ] || { echo " MISS: node A did not release the holder (no SCSD-I-DLMDEQ1)"; FAIL=1; } +[ "$B_DEFER" = 1 ] || { echo " MISS: node B did not WIRE the deferred GRANT (no SCSD-I-DLMDEFER) -- the deferred-grant wire did not fire"; FAIL=1; } + +# THE MILESTONE ASSERTION: node A's origin record flipped NL->EX across the wire. +if [ "$A_FLIP" != 1 ]; then + echo " FAIL: node A's origin record did NOT flip (no SCSD-I-DLMH5FLIP) --" + echo " the QUEUED cross-node request never GRANTED on the requester. The" + echo " requester-side GRANT RECEIVE (vms_lock_dlm_xnode_dispatch) or the" + echo " deferred-grant wire did not complete the round-trip." + FAIL=1 +elif [ "${A_FLIP_MODE:-}" != "EX" ]; then + echo " FAIL: node A's origin record flipped to '${A_FLIP_MODE:-}', expected EX (the granted mode)." + FAIL=1 +fi + +echo "" +echo "==========================================" +if [ "$FAIL" = 0 ]; then + echo " DLM HARNESS H5 PASSED: block-then-grant proven on the REQUESTER over the SCS wire" + echo " Node A held RESONE EX, then a SECOND incompatible \$ENQ QUEUED on node B; B WIRED" + echo " a queued-reply and node A's requester ORIGIN record went genuinely PENDING" + echo " (SCSD-I-DLMPEND, read from A's OWN executive). Node A released the holder with a" + echo " real cross-node \$DEQ; node B released it, GRANTED the queued request, and WIRED" + echo " the deferred GRANT (SCSD-I-DLMDEFER) off that real \$DEQ. Node A dispatched it" + echo " into its executive and the origin record FLIPPED NL->EX (SCSD-I-DLMH5FLIP," + echo " granted_mode=EX) -- the status flip observed on the REQUESTER node, across the" + echo " live SCS wire, driven ONLY by what the master sent. The BLKAST wire is deferred" + echo " honestly. INV-6: no fabricated wire reply or grant." + echo "==========================================" + exit 0 +else + echo " DLM HARNESS H5 FAILED" + echo "==========================================" + echo "--- node A console tail ---"; tail -n 40 "$OUT/nodeA.console.log" 2>/dev/null || true + echo "--- node B console tail ---"; tail -n 40 "$OUT/nodeB.console.log" 2>/dev/null || true + exit 1 +fi From 18ff9142a99ba262b6f52f8b9fb8e5785ceac0bc Mon Sep 17 00:00:00 2001 From: alice Date: Fri, 28 Aug 2026 22:35:57 +0000 Subject: [PATCH 4/7] =?UTF-8?q?vms-6ca:=20DLM=20H5=20=E2=80=94=20record=20?= =?UTF-8?q?the=20async-reply=20wire=20+=20requester-side=20completion=20in?= =?UTF-8?q?=20the=20compat=20ledger?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit docs/compat/facilities/cluster-dlm.yaml (single-ledger source, INV-LEDGER): new cluster-dlm$async-reply-wire item for the H5 rung — the master WIREs the queued-reply + deferred GRANT over SCS, the requester-side GRANT RECEIVE completes an executive-resident origin record, and the NL->EX flip is observed on the requester. Notes the BLKAST wire deferred honestly and multi-peer deferred-grant routing as a later rung. Summary updated; H5 evidence points at the two-node harness. YAML validated. Co-Authored-By: Claude Opus 4.8 (1M context) --- docs/compat/facilities/cluster-dlm.yaml | 32 ++++++++++++++++++++++++- 1 file changed, 31 insertions(+), 1 deletion(-) diff --git a/docs/compat/facilities/cluster-dlm.yaml b/docs/compat/facilities/cluster-dlm.yaml index a859d2444..f1f9c4150 100644 --- a/docs/compat/facilities/cluster-dlm.yaml +++ b/docs/compat/facilities/cluster-dlm.yaml @@ -17,7 +17,14 @@ summary: > waiting queue (VMS_DLM_STS_QUEUED, not a grant, not a NOQUEUE decline), the master FIRES a blocking-AST decision naming the remote holder, and a real cross-node $DEQ releases and grants the blocked request (block-then-grant). - See kernel-executive.yaml for the single-node lock manager this builds on. + Rung H5 (vms-6ca) carries the async replies over the real SCS wire: the master + WIREs the queued-reply and — on a real $DEQ — the deferred GRANT, and the + REQUESTER-SIDE GRANT RECEIVE (was SS$_UNSUPPORTED) completes an + executive-resident origin record on the requesting node, so the block-then-grant + status flip (NL->EX) is observed on the REQUESTER across the wire, not just in + the master's local state. The BLKAST wire is deferred honestly (the holder + releases on its own). See kernel-executive.yaml for the single-node lock manager + this builds on. last_reviewed: 2026-08-28 items: - id: cluster-dlm$cross-node-lock @@ -38,6 +45,29 @@ items: real /dev/vms (tests/qemu/test_syssvc_dlm_xnode.c, x86_64 + Alpha LP64). No fake grants OR blocks — a queued request is a real lock on the master's queue that releases only on a real $DEQ (INV-6). + - id: cluster-dlm$async-reply-wire + kind: feature + status: partial + authenticity: real + vms: "Async DLM replies (queued-reply + deferred GRANT) delivered over SCS; requester-side completion" + evidence: src/vmsscs/scsd.c + notes: > + vms-6ca rung H5. The master's async replies ride the live SCS wire: node B + WIREs the queued-reply (granted mode NL — the requester stays genuinely + pending) and, on a real cross-node $DEQ that flips a queued waiter to + granted, WIREs an unprompted deferred GRANT (SS$_NORMAL, the granted mode) + to that requester (vms_lock_dlm_xnode_deq reports the flipped waiter). The + REQUESTER-SIDE GRANT RECEIVE (VMS_DLM_OP_GRANT, was SS$_UNSUPPORTED) + completes an executive-resident ORIGIN record on the requesting node, whose + granted mode is set ONLY from what the master sent over SCS; GETLKI reads + the NL->EX flip back on the REQUESTER. Proven on a two-node live A<->B QEMU + harness with real /dev/vms on both (tests/qemu/run_dlm_harness_h5.sh: + SCSD-I-DLMPEND, SCSD-I-DLMDEFER, SCSD-I-DLMH5FLIP granted_mode=EX). The + BLKAST wire (master -> holder) is deferred honestly on this rung — BLKAST as + a receive op still returns SS$_UNSUPPORTED; the block-then-grant is proven + without it (the holder releases on its own). Multi-peer routing of a deferred + grant by CSID (releaser and queued waiter on different nodes) is a later + rung; H5 proves the same-peer case. No fabricated wire reply or grant (INV-6). - id: cluster-dlm$remaster-lvb-deadlock kind: feature status: absent From 1d4d5a1391e695a5c039c78b1751ed0e8a5f5088 Mon Sep 17 00:00:00 2001 From: alice Date: Fri, 28 Aug 2026 22:37:41 +0000 Subject: [PATCH 5/7] vms-6ca: regenerate compat surface register for the H5 async-reply-wire item Derived docs/compatibility-surface.md re-rendered from docs/compat/*.yaml via tools/compat/render_compat.py (INV-LEDGER: yaml is source, md is generated). 407 items now catalogued. Co-Authored-By: Claude Opus 4.8 (1M context) --- docs/compatibility-surface.md | 16 +++++++++------- 1 file changed, 9 insertions(+), 7 deletions(-) diff --git a/docs/compatibility-surface.md b/docs/compatibility-surface.md index aee223618..bc75cc150 100644 --- a/docs/compatibility-surface.md +++ b/docs/compatibility-surface.md @@ -5,15 +5,15 @@ ## Inventory -**406 surfaces catalogued** across 9 domains, each with a per-surface status. +**407 surfaces catalogued** across 9 domains, each with a per-surface status. > This register is an **inventory, not a percentage.** The total VMS compatibility surface has **no known denominator** — it is not version-scoped and cannot be counted — so no "% compatible" is claimed or computable. The catalogue is **incomplete by construction** and grows as surfaces are identified. Below are absolute counts; V1 progress is tracked separately against the commitment set we define, and is never conflated with the whole surface. | Status | Count | | Authenticity | Count | |---|---|---|---|---| -| ✅ verified | 24 | | real | 229 | +| ✅ verified | 24 | | real | 230 | | 🟢 implemented | 206 | | n/a | 99 | -| 🟡 partial | 44 | | advisory | 34 | +| 🟡 partial | 45 | | advisory | 34 | | 🟠 stub | 19 | | facade-risk | 44 | | 🔵 designed | 2 | | | | | ⬜ absent | 111 | | | | @@ -24,7 +24,7 @@ Legend: ✅ verified · 🟢 implemented · 🟡 partial · 🟠 stub · 🔵 de Of the surfaces **committed to V1** (`scope_1_0: in` — a set we define, not a measure of the whole surface): -- **364 committed** — **230 met** (implemented/verified), 43 in progress (partial), 91 not started (absent/stub/designed). +- **365 committed** — **230 met** (implemented/verified), 44 in progress (partial), 91 not started (absent/stub/designed). - ⚠ **42 of the committed surfaces carry facade-risk** — they must reach honest behaviour, not just "done". - Not in the V1 commitment set: 8 out · 25 stretch · 9 undecided (incl. the language scope calls, `vms-082`). @@ -854,20 +854,22 @@ Full UAF account record storage and SHA-256 password authentication are real. Ac _SCS, NISCA/NISCS, connection manager/quorum, cluster-wide DLM, MSCP serving, cluster-wide logicals, shadowing._ -`✅✅🟢🟢🟢🟢🟢🟢🟢🟢🟡🟡🟡🟠🟠🟠⬜⬜⬜⬜⬜⬜⬜⬜` — 15 surfaces catalogued (6 met · 2 in progress · 7 not started) · V1: 14 committed, 6 met · ⚠ 1 facade-risk +`✅✅🟢🟢🟢🟢🟢🟢🟢🟢🟡🟡🟡🟡🟠🟠🟠⬜⬜⬜⬜⬜⬜⬜⬜` — 16 surfaces catalogued (6 met · 3 in progress · 7 not started) · V1: 15 committed, 6 met · ⚠ 1 facade-risk ### cluster-dlm — Cluster-wide Distributed Lock Manager scope: in · plan: vms-694 · ref: OpenVMS Cluster Systems manual; $ENQ/$DEQ/$GETLKI system services · reviewed 2026-08-28 -A cross-node $ENQ now GRANTS, BLOCKS, and GRANTS-on-release on the mastering node — the DLM's core contention behaviour — while resource remastering, LVB replication, and distributed deadlock detection remain absent, that absence authentic (SS$_UNSUPPORTED, never a fabricated answer). Rung 1 (vms-94c) built the DLM message TRANSPORT (a DLM SYSAP SCS connection + the ENQ/GRANT/DEQ/BLKAST message class). Rung 2 (vms-e8f1) made the RECEIVE handler grant a compatible cross-node $ENQ, held for the remote requester's CSID. Rung 3 (vms-904c) lifts the ENQ scope-fence: an incompatible request now QUEUES on the master's real waiting queue (VMS_DLM_STS_QUEUED, not a grant, not a NOQUEUE decline), the master FIRES a blocking-AST decision naming the remote holder, and a real cross-node $DEQ releases and grants the blocked request (block-then-grant). See kernel-executive.yaml for the single-node lock manager this builds on. +A cross-node $ENQ now GRANTS, BLOCKS, and GRANTS-on-release on the mastering node — the DLM's core contention behaviour — while resource remastering, LVB replication, and distributed deadlock detection remain absent, that absence authentic (SS$_UNSUPPORTED, never a fabricated answer). Rung 1 (vms-94c) built the DLM message TRANSPORT (a DLM SYSAP SCS connection + the ENQ/GRANT/DEQ/BLKAST message class). Rung 2 (vms-e8f1) made the RECEIVE handler grant a compatible cross-node $ENQ, held for the remote requester's CSID. Rung 3 (vms-904c) lifts the ENQ scope-fence: an incompatible request now QUEUES on the master's real waiting queue (VMS_DLM_STS_QUEUED, not a grant, not a NOQUEUE decline), the master FIRES a blocking-AST decision naming the remote holder, and a real cross-node $DEQ releases and grants the blocked request (block-then-grant). Rung H5 (vms-6ca) carries the async replies over the real SCS wire: the master WIREs the queued-reply and — on a real $DEQ — the deferred GRANT, and the REQUESTER-SIDE GRANT RECEIVE (was SS$_UNSUPPORTED) completes an executive-resident origin record on the requesting node, so the block-then-grant status flip (NL->EX) is observed on the REQUESTER across the wire, not just in the master's local state. The BLKAST wire is deferred honestly (the holder releases on its own). See kernel-executive.yaml for the single-node lock manager this builds on. -3 items · 0 met · 2 in progress · 1 not started +4 items · 0 met · 3 in progress · 1 not started | | Surface | Kind | VMS | Status | Auth | Scope | Evidence / notes | |---|---|---|---|---|---|---|---| | 🟡 | `cluster-dlm$cross-node-lock` | feature | Cross-node $ENQ grant / block-then-grant / blocking-AST on the mastering node | partial | real | in | `src/kernel-core/vms_lock.c` — vms-94c/vms-e8f1/vms-904c. vms_lock_dlm_xnode_dispatch runs a decoded cross-node $ENQ through the real lock manager on the mastering node, held FOR the remote requester's CSID: a compatible request GRANTS (SS$_NORMAL, GET_RESMASTER shows held_for=); an incompatible one QUEUES on the real waiting queue (VMS_DLM_STS_QUEUED, GETLKI shows granted NL / requested EX) and the master emits the BLKAST directive (blocking_csid) for the remote holder; a cross-node $DEQ (vms_lock_dlm_xnode_deq, authorized by CSID) releases and grants the blocked request (GETLKI flips NL->EX). Proven on a real /dev/vms (tests/qemu/test_syssvc_dlm_xnode.c, x86_64 + Alpha LP64). No fake grants OR blocks — a queued request is a real lock on the master's queue that releases only on a real $DEQ (INV-6). | +| 🟡 | `cluster-dlm$async-reply-wire` | feature | Async DLM replies (queued-reply + deferred GRANT) delivered over SCS; requester-side completion | partial | real | in | `src/vmsscs/scsd.c` — vms-6ca rung H5. The master's async replies ride the live SCS wire: node B WIREs the queued-reply (granted mode NL — the requester stays genuinely pending) and, on a real cross-node $DEQ that flips a queued waiter to granted, WIREs an unprompted deferred GRANT (SS$_NORMAL, the granted mode) to that requester (vms_lock_dlm_xnode_deq reports the flipped waiter). The REQUESTER-SIDE GRANT RECEIVE (VMS_DLM_OP_GRANT, was SS$_UNSUPPORTED) completes an executive-resident ORIGIN record on the requesting node, whose granted mode is set ONLY from what the master sent over SCS; GETLKI reads the NL->EX flip back on the REQUESTER. Proven on a two-node live A<->B QEMU harness with real /dev/vms on both (tests/qemu/run_dlm_harness_h5.sh: SCSD-I-DLMPEND, SCSD-I-DLMDEFER, SCSD-I-DLMH5FLIP granted_mode=EX). The BLKAST wire (master -> holder) is deferred honestly on this rung — BLKAST as a receive op still returns SS$_UNSUPPORTED; the block-then-grant is proven without it (the holder releases on its own). Multi-peer routing of a deferred grant by CSID (releaser and queued waiter on different nodes) is a later rung; H5 proves the same-peer case. No fabricated wire reply or grant (INV-6). + | | ⬜ | `cluster-dlm$remaster-lvb-deadlock` | feature | Resource-directory consistency, dynamic remastering, LVB replication, distributed deadlock detection | absent | real | in | `src/kernel-core/vms_lock.c` — Still honestly out of scope, above the contention rung: resource-directory consistency (vms-1bba), remastering on membership change (vms-6ee), LVB replication (vms-d81), and distributed deadlock detection (vms-ec75). The cross-node ENQ carries no VALBLK and the queue path skips the single-node (proc-keyed) deadlock detector, both by design — SS$_UNSUPPORTED / omission, never a fabricated answer. A non-local directory/master on the SEND side still fails SS$_UNSUPPORTED (dlm_resolve_master). Gap is authentic (real). | | 🟡 | `cluster-dlm$scs-transport` | feature | DLM SYSAP SCS connection + lock-request message class (ENQ/GRANT/DEQ/BLKAST) between cluster nodes | partial | real | in | `src/vmsscs/scs_dlm.c` — vms-94c rung 1. A DLM message rides as a p.4-13 SCS application message (MTYPE 10) with its body at SCA content offset 58 (the scs_mscp.c nesting). The message TRANSPORT is real; the message HANDLER does not yet grant (rung 2). From cbf5da4d7f6832960c3fe1e403b64f59790e95a5 Mon Sep 17 00:00:00 2001 From: alice Date: Fri, 28 Aug 2026 22:43:30 +0000 Subject: [PATCH 6/7] =?UTF-8?q?ci:=20quote=20the=20job=20name=20with=20a?= =?UTF-8?q?=20mid-value=20colon=20=E2=80=94=20repairs=20ci.yml=20parse=20(?= =?UTF-8?q?dark=20since=20#850)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The job name added today by #850 (vms-dec) contained an unquoted mid-value colon ('... before Username: (vms-dec)'), which is invalid YAML. GitHub Actions could not parse ci.yml, so EVERY run of it — on main and every branch — completed with ZERO jobs (no Build & Test, no harnesses, nothing); the only checks still firing were the separate Ledger-drift and CodeQL workflows. Quoting the value restores the whole workflow. Pre-existing project-wide breakage, surfaced while wiring the H5 CI job (which could not run until ci.yml parsed). Co-Authored-By: Claude Opus 4.8 (1M context) --- .github/workflows/ci.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index c39942ba4..4d9c0f8b8 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -1738,7 +1738,7 @@ jobs: # run of echoed blank lines precedes "Username:" while login still works # (tests/qemu/test_console_boot_no_newline_spam.sh). console-newline-spam-e2e: - name: Console boot — no echoed-RETURN newline spam before Username: (vms-dec) + name: "Console boot — no echoed-RETURN newline spam before Username: (vms-dec)" needs: changes if: ${{ github.event_name != 'pull_request' || needs.changes.outputs.toolchain == 'true' || needs.changes.outputs.kernel == 'true' }} runs-on: ubuntu-latest From 6124d2275c6d20a6e96d3572b980eb2017feb6ed Mon Sep 17 00:00:00 2001 From: alice Date: Fri, 28 Aug 2026 22:53:14 +0000 Subject: [PATCH 7/7] vms-6ca: register the H5 client sender in the send-site census + repair H0's stale selftest op MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two fixes surfaced by re-enabling ci.yml: 1. scs_send_sites census: the new scsd_dlm_client_send_op() sender (node A's contending $ENQ + holder $DEQ) is added to the CHOKED half of the SEND SITE TABLE in scsd.c; the scsd_dlm_srv_msg_input entry now notes the deferred GRANT it also wires. test_scsd_send_sites: 109 checks, 0 failures. 2. DLM Harness H0 selftest: it drove a DEQ expecting SS$_UNSUPPORTED (2296), but rung 3 (#873, vms-904c) made DEQ real — a DEQ of an unknown handle now returns SS$_IVLOCKID (8484), so H0 had been red since #873 (hidden by the dark CI). The side-effect-free "executive reached, honest 2296" probe now uses BLKAST, whose receive op still honestly returns SS$_UNSUPPORTED (its wire is deferred, vms-6ca) and mutates no lock state. No run/init/ci grep change needed — still rc=2296 PASS. Co-Authored-By: Claude Opus 4.8 (1M context) --- src/vmsscs/scsd.c | 57 ++++++++++++++++++++++++++++------------------- 1 file changed, 34 insertions(+), 23 deletions(-) diff --git a/src/vmsscs/scsd.c b/src/vmsscs/scsd.c index 3ad2eb1b2..76066c45b 100644 --- a/src/vmsscs/scsd.c +++ b/src/vmsscs/scsd.c @@ -3618,9 +3618,23 @@ static ssize_t send_frame_raw(int sock, int ifindex, const uint8_t mac[6], * sequenced SCS traffic on a connection that * cannot exist before its circuit does. * scsd_dlm_srv_msg_input() node-B's GRANT response leg: the honest - * executive status (SS$_UNSUPPORTED/SS$_NOSUCHDEV, - * never a fabricated grant) sent back on the SAME - * OPEN VC, same argument as the MSCP responder. + * executive status (a granted cross-node $ENQ, + * a queued-reply, or -- new in vms-6ca -- the + * DEFERRED GRANT the master WIREs off a real + * $DEQ when a queued cross-node request finally + * grants; never a fabricated grant) sent back on + * the SAME OPEN VC, same argument as the MSCP + * responder. + * + * CHOKED, and new in vms-6ca (DLM rung H5): + * scsd_dlm_client_send_op() node-A's generic DLM client SEND: the second, + * contending $ENQ and the holder's $DEQ that drive + * the block-then-grant-over-the-wire sequence. + * MTYPE-10 SYSAP messages on the OPEN VMS$VAXcluster + * VC to the peer's OVMX$DLM server handle -- ordinary + * sequenced SCS traffic on a connection that cannot + * exist before its circuit does, exactly like + * scsd_dlm_send_enq above. * * CHOKED, and new in vms-600: * scsd_mscp_srv_xfer() the live scs_mscp_srv_xfer_fn: the SCA @@ -13708,34 +13722,31 @@ int main(int argc, char **argv) * with SS$_NOSUCHDEV (2680). It fabricates NOTHING (INV-6): it * prints verbatim whatever scsd_dlm_dispatch_to_executive() got. * - * It builds a synthetic-but-well-formed DEQ request (the same shape - * scs_dlm_parse hands scsd_dlm_srv_msg_input on a real received - * frame) so the executive's VMS_IOCTL_DLM_XNODE handler is reached - * exactly as an inbound cross-node request would reach it -- opens no - * socket, needs no CAP_NET_RAW, touches no wire. The 2296-not-2680 - * flip is the machine-checkable proof. + * It builds a synthetic-but-well-formed BLKAST request (the same shape + * scs_dlm_parse hands the input routines on a real received frame) so + * the executive's VMS_IOCTL_DLM_XNODE handler is reached exactly as an + * inbound cross-node message would reach it -- opens no socket, needs + * no CAP_NET_RAW, touches no wire. The 2296-not-2680 flip is the + * machine-checkable proof. * - * DEQ (not ENQ) on purpose: DLM rung 2 (vms-e8f1) makes a compatible - * cross-node ENQ GRANT (SS$_NORMAL), which would master + hold a lock - * as a side effect of a self-test. DEQ (cross-node release) is a - * LATER rung that still returns SS$_UNSUPPORTED (2296) and touches no - * lock state -- the ideal side-effect-free "executive reached, honest - * status" probe H0 needs. */ + * BLKAST (not ENQ or DEQ) on purpose. The side-effect-free + * "executive reached, honest status = 2296" probe has to name an op + * the handler still declines with SS$_UNSUPPORTED and that touches no + * lock state. ENQ now GRANTS a compatible cross-node lock (rung 2, + * vms-e8f1) and DEQ now RELEASES/authorizes real lock state (rung 3, + * vms-904c: a DEQ of an unknown handle returns SS$_IVLOCKID, not + * SS$_UNSUPPORTED) -- both have moved off 2296. BLKAST as a RECEIVE op + * is the holder-side blocking-AST delivery whose WIRE is deferred + * (vms-6ca, rung H5): it still returns SS$_UNSUPPORTED and mutates + * nothing -- the ideal probe H0 needs. */ struct scs_dlm_msg m; memset(&m, 0, sizeof(m)); - m.op = SCS_DLM_OP_DEQ; + m.op = SCS_DLM_OP_BLKAST; m.mode = LCK$K_NLMODE; m.req_lkid = 0x00040011u; m.master_lkid = 0x00080002u; m.req_csid = 1025u; m.master_csid = 1026u; - { - static const char *res = "DLMXNODE1"; - size_t rn = strlen(res); - if (rn > SCS_DLM_RESNAM_MAX) rn = SCS_DLM_RESNAM_MAX; - memcpy(m.resnam, res, rn); - m.namelen = (uint8_t)rn; - } uint32_t status = scsd_dlm_dispatch_to_executive(&m, NULL, NULL, NULL); printf("SCSD-I-DLMSELFTEST, executive DLM dispatch status=%u"