From d7e0e044bdf3fec47efae5270cda91a63946c31d Mon Sep 17 00:00:00 2001
From: pasta
Date: Fri, 25 Sep 2026 08:54:49 -0500
Subject: [PATCH 01/11] fix(refs): read ref updates with a cursor-free keyset
scan and a completeness check
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Paging the refState index (refNameHash, $createdAt) with a startAfter cursor loses rows on testnet protocol 13: measured on the nightly repo, 197 of 229 updates came back. Drive's v0 lowering applies the cursor document's lower index levels to every sibling refNameHash branch, not only the cursor's own, so later refs lose rows and a page can come back short, which the pager reads as the end (the family of dashpay/platform#4396; this orderBy-only shape is still unfixed there). PR #10 worked around it by reading every update of every ref in reflog order on every command.
Both clients now page refState by key: refNameHash > last, 100 rows, no cursor. Every ref but the last on a full page is complete, a ref that fills a page is read on its own with an equality query, and each round advances. A prevOid with no parent in the result (or an out-of-order page) falls back to the reflog read and unions by $id. On the nightly repo that is 3 queries returning the same 229 updates as the full read. base_ref_tips / readRefUpdates read one ref by equality. Mocks in both suites reproduce the sibling-branch drop. data-contracts §2.3 and design-freeze-2 now describe this.
Co-Authored-By: Claude Opus 5.5 (1M context)
---
crates/forge-core/src/collab.rs | 50 +--
crates/forge-core/src/lib.rs | 3 +
crates/forge-core/src/refs.rs | 599 +++++++++++++++++++++++++++++++
crates/forge-core/src/repo.rs | 97 +----
crates/forge-core/src/rules.rs | 2 +-
docs/contracts/data-contracts.md | 15 +-
docs/design-freeze-2.md | 2 +-
forge-web/lib/repo/index.ts | 3 +-
forge-web/lib/repo/refs.test.ts | 255 ++++++++-----
forge-web/lib/repo/refs.ts | 288 ++++++++++-----
forge-web/lib/sdk/query.ts | 2 +-
11 files changed, 1013 insertions(+), 303 deletions(-)
create mode 100644 crates/forge-core/src/refs.rs
diff --git a/crates/forge-core/src/collab.rs b/crates/forge-core/src/collab.rs
index 73e6205f9..063ed4ef6 100644
--- a/crates/forge-core/src/collab.rs
+++ b/crates/forge-core/src/collab.rs
@@ -42,8 +42,6 @@ const DOC_EVENT: &str = "event";
const DOC_REVIEW: &str = "review";
const DOC_LABEL: &str = "label";
const DOC_RELEASE: &str = "release";
-const DOC_REF_UPDATE: &str = "refUpdate";
-const DOC_PROTECTED_REF_UPDATE: &str = "protectedRefUpdate";
// Registry-contract document types.
const DOC_STAR: &str = "star";
const DOC_FOLLOW: &str = "follow";
@@ -926,46 +924,30 @@ impl<'a> PullRequestService<'a> {
}
/// Collect every oid that was ever a tip of `base_ref_name` (the monotonic merge-
- /// reachability set) plus the newest such tip. Walks the full `refUpdate` +
- /// `protectedRefUpdate` history for the ref (paginated), taking every non-null `newOid`.
+ /// reachability set) plus the newest such tip. Walks the ref's full `refUpdate` +
+ /// `protectedRefUpdate` history — [`crate::refs::read_ref_history`], an equality read on
+ /// one `refNameHash`, so its cost is this ref's pushes, not the repo's — taking every
+ /// non-null `newOid`.
async fn base_ref_tips(
&self,
contract: &LoadedContract,
base_ref_name: &str,
) -> Result<(std::collections::BTreeSet, Option)> {
let ref_name_hash = sha256(base_ref_name.as_bytes());
+ let updates = crate::refs::read_ref_history(self.client, contract, ref_name_hash).await?;
let mut tips: std::collections::BTreeSet = std::collections::BTreeSet::new();
let mut newest: Option<(u64, String, String)> = None; // (created_at, id, oid)
- for doc_type in [DOC_REF_UPDATE, DOC_PROTECTED_REF_UPDATE] {
- let docs = self
- .client
- .query_all_documents(
- contract,
- doc_type,
- &[QueryFilter::eq(
- "refNameHash",
- FieldValue::bytes32(ref_name_hash),
- )],
- &[QueryOrder::asc("$createdAt")],
- )
- .await?;
- for d in &docs {
- let Some(oid) = d.field_hex("newOid") else {
- continue;
- };
- if oid.is_empty() || oid.bytes().all(|b| b == b'0') {
- continue; // null oid = ref deletion, never a reachable tip
- }
- tips.insert(oid.clone());
- let created_at = d.created_at.unwrap_or(0);
- let candidate = (created_at, d.id.clone(), oid);
- let better = match &newest {
- None => true,
- Some(n) => (n.0, &n.1) < (candidate.0, &candidate.1),
- };
- if better {
- newest = Some(candidate);
- }
+ for u in updates {
+ if u.new_oid.is_empty() || u.new_oid.bytes().all(|b| b == b'0') {
+ continue; // null oid = ref deletion, never a reachable tip
+ }
+ tips.insert(u.new_oid.clone());
+ let better = match &newest {
+ None => true,
+ Some(n) => (n.0, &n.1) < (u.created_at, &u.id),
+ };
+ if better {
+ newest = Some((u.created_at, u.id, u.new_oid));
}
}
Ok((tips, newest.map(|(_, _, oid)| oid)))
diff --git a/crates/forge-core/src/lib.rs b/crates/forge-core/src/lib.rs
index 0311631eb..bae2c945e 100644
--- a/crates/forge-core/src/lib.rs
+++ b/crates/forge-core/src/lib.rs
@@ -8,6 +8,8 @@
//! `WriteEngine` document create/delete lifecycle + idempotent-retry journal types.
//! - [`repo`] — `RepoService`: the repo-lifecycle API (`create_repo` / `resolve_repo` /
//! ref + pack-manifest + chunk read/write) `git-remote-dash` calls.
+//! - [`refs`] — complete ref-update reads (keyset scan + completeness fallback) shared by
+//! ref listing and PR base-tip resolution.
//! - [`tokens`] — `TokenService`: the collaborator ACL (grant/suspend/revoke = token
//! mint/freeze/destroy; balances = the on-chain collaborator list).
//! - [`collab`] — issue / PR / review / release / label services + the registry social
@@ -32,6 +34,7 @@ pub mod keystore;
pub mod network;
pub mod pack;
pub mod platform;
+pub mod refs;
pub mod repo;
pub mod rules;
pub mod storage;
diff --git a/crates/forge-core/src/refs.rs b/crates/forge-core/src/refs.rs
new file mode 100644
index 000000000..ab7e54e5c
--- /dev/null
+++ b/crates/forge-core/src/refs.rs
@@ -0,0 +1,599 @@
+//! Ref-update reads: every ref's complete `refUpdate` + `protectedRefUpdate` history.
+//!
+//! Shared by [`crate::repo::RepoService::read_refs`] (list every ref) and the PR base-tip
+//! reader in [`crate::collab`] (one ref). forge-web's `lib/repo/refs.ts` implements the same
+//! reads the same way; the two must agree, because the fold over their output
+//! ([`crate::rules::resolve_ref`]) is only as parity-safe as its input.
+//!
+//! ## Why a keyset scan, and never a cursor over the `refState` index
+//!
+//! The `refState` index is `(refNameHash, $createdAt)`. Paging it with a `startAfter`
+//! document cursor LOSES rows on testnet (protocol 13): measured on the nightly repo, 32 of
+//! 229 updates never came back. Drive's v0 path-query lowering applies the cursor
+//! document's lower-level bounds (its `$createdAt`, then its `$id`) to every *sibling*
+//! `refNameHash` branch instead of only the cursor's own branch, so a page silently omits
+//! rows from later refs and can come back short — and a short page is the only end-of-data
+//! signal a pager has, so everything after it vanishes too. This is the multi-branch cursor
+//! family dashpay/platform#4396 fixed for `in` queries at protocol 14; the `orderBy`-only
+//! shape used here is one it lists as still unfixed.
+//!
+//! A range *where-clause* has no cursor document to leak, so this module never sends one on
+//! a multi-branch query. It pages by key instead:
+//!
+//! 1. `refNameHash > last` ordered `(refNameHash, $createdAt)`, `limit 100`. Every ref on
+//! a full page except the last one is complete; the next page starts after the last
+//! complete ref, re-reading the one that may have been cut off.
+//! 2. A ref that fills a whole page by itself is read on its own with an equality query
+//! (`refNameHash == h`), which is single-branch — the cursor bug needs sibling branches —
+//! and the scan resumes after it.
+//!
+//! Each round strictly advances `last`, so the scan terminates without leaning on a page
+//! cap, and a one-ref read costs pages of that ref only, not of the whole repo.
+//!
+//! ## Completeness check
+//!
+//! Every non-null `prevOid` a pusher records is the tip it saw, which is some earlier
+//! update's `newOid` for the same ref. An update whose `prevOid` matches nothing is
+//! therefore evidence that a row is missing, and triggers the fallback: every update of
+//! the type read in `$createdAt` (`reflog`) order, the only other complete read. Its rows are
+//! unioned with the scan's (both are proof-verified documents; a union can only add real
+//! rows). A dangling `prevOid` can also be written on purpose, which costs the fallback's
+//! extra reads but never changes the answer.
+
+use std::collections::{BTreeMap, BTreeSet};
+
+use crate::error::{Error, Result};
+use crate::platform::{
+ FetchedDocument, FieldValue, LoadedContract, PlatformClient, QueryFilter, QueryOrder,
+};
+use crate::rules::RefUpdate;
+
+/// The plain ref-update document type.
+pub(crate) const DOC_REF_UPDATE: &str = "refUpdate";
+/// The MAINTAIN-gated ref-update document type.
+pub(crate) const DOC_PROTECTED_REF_UPDATE: &str = "protectedRefUpdate";
+
+/// Both ref-update types, each with the `protected` flag its updates carry into the fold.
+const REF_UPDATE_TYPES: [(&str, bool); 2] =
+ [(DOC_REF_UPDATE, false), (DOC_PROTECTED_REF_UPDATE, true)];
+
+/// Rows per keyset page — Drive's per-query maximum.
+pub(crate) const KEYSET_PAGE: u32 = 100;
+
+/// A backstop against a node that never lets the scan finish. Every round advances past at
+/// least one ref, so this is a bound on distinct refs per type (with every ref on its own
+/// round), not on updates.
+const MAX_KEYSET_ROUNDS: usize = 100_000;
+
+/// Every ref's history, keyed by the raw `refNameHash`.
+pub type RefHistories = BTreeMap<[u8; 32], Vec>;
+
+/// The three reads the scan is built from. A trait so the scan (the part that decides
+/// whether a read is complete) runs against a mock in tests; [`PlatformRefSource`] is the
+/// live implementation.
+pub(crate) trait RefDocSource {
+ /// One page of `doc_type` ordered `(refNameHash, $createdAt)`, restricted to
+ /// `refNameHash > after` when `after` is set, at most [`KEYSET_PAGE`] rows. No cursor.
+ async fn keyset_page(
+ &self,
+ doc_type: &str,
+ after: Option<[u8; 32]>,
+ ) -> Result>;
+
+ /// Every `doc_type` update of one ref (an equality read, paged to exhaustion).
+ async fn ref_history(&self, doc_type: &str, hash: [u8; 32]) -> Result>;
+
+ /// Every `doc_type` update in `$createdAt` order — the fallback.
+ async fn full_scan(&self, doc_type: &str) -> Result>;
+}
+
+/// [`RefDocSource`] over a live Platform connection.
+pub(crate) struct PlatformRefSource<'a> {
+ pub(crate) client: &'a PlatformClient,
+ pub(crate) contract: &'a LoadedContract,
+}
+
+impl RefDocSource for PlatformRefSource<'_> {
+ async fn keyset_page(
+ &self,
+ doc_type: &str,
+ after: Option<[u8; 32]>,
+ ) -> Result> {
+ let filters: Vec = after
+ .map(|h| QueryFilter::gt("refNameHash", FieldValue::bytes32(h)))
+ .into_iter()
+ .collect();
+ self.client
+ .query_documents(
+ self.contract,
+ doc_type,
+ &filters,
+ &[
+ QueryOrder::asc("refNameHash"),
+ QueryOrder::asc("$createdAt"),
+ ],
+ KEYSET_PAGE,
+ None,
+ )
+ .await
+ }
+
+ async fn ref_history(&self, doc_type: &str, hash: [u8; 32]) -> Result> {
+ self.client
+ .query_all_documents(
+ self.contract,
+ doc_type,
+ &[QueryFilter::eq("refNameHash", FieldValue::bytes32(hash))],
+ &[QueryOrder::asc("$createdAt")],
+ )
+ .await
+ }
+
+ async fn full_scan(&self, doc_type: &str) -> Result> {
+ self.client
+ .query_all_documents(
+ self.contract,
+ doc_type,
+ &[],
+ &[QueryOrder::asc("$createdAt")],
+ )
+ .await
+ }
+}
+
+/// Read every ref's complete history from a live repo contract. See the module docs.
+pub async fn read_all_ref_updates(
+ client: &PlatformClient,
+ contract: &LoadedContract,
+) -> Result {
+ read_all_with(&PlatformRefSource { client, contract }).await
+}
+
+/// Read one ref's complete history (both types) from a live repo contract.
+pub async fn read_ref_history(
+ client: &PlatformClient,
+ contract: &LoadedContract,
+ ref_name_hash: [u8; 32],
+) -> Result> {
+ ref_history_with(&PlatformRefSource { client, contract }, ref_name_hash).await
+}
+
+pub(crate) async fn ref_history_with(
+ src: &impl RefDocSource,
+ hash: [u8; 32],
+) -> Result> {
+ let hash_hex = hex::encode(hash);
+ let mut out = Vec::new();
+ for (doc_type, protected) in REF_UPDATE_TYPES {
+ for d in src.ref_history(doc_type, hash).await? {
+ out.push(ref_update_from_doc(&d, &hash_hex, protected));
+ }
+ }
+ Ok(out)
+}
+
+/// The scan behind [`read_all_ref_updates`], over any [`RefDocSource`].
+pub(crate) async fn read_all_with(src: &impl RefDocSource) -> Result {
+ let mut docs: Vec<(Vec, bool, &str)> = Vec::new();
+ let mut consistent = true;
+ for (doc_type, protected) in REF_UPDATE_TYPES {
+ let (rows, ordered) = keyset_scan(src, doc_type).await?;
+ consistent &= ordered;
+ docs.push((rows, protected, doc_type));
+ }
+ let mut by_hash = group(&docs)?;
+ let dangling = by_hash.values().filter(|u| has_missing_parent(u)).count();
+ if consistent && dangling == 0 {
+ return Ok(by_hash);
+ }
+
+ tracing::warn!(
+ refs_with_missing_parent = dangling,
+ out_of_order_page = !consistent,
+ "ref-update keyset scan looks incomplete; re-reading every update in reflog order"
+ );
+ for (rows, _, doc_type) in &mut docs {
+ let seen: BTreeSet = rows.iter().map(|d| d.id.clone()).collect();
+ let extra: Vec = src
+ .full_scan(doc_type)
+ .await?
+ .into_iter()
+ .filter(|d| !seen.contains(&d.id))
+ .collect();
+ rows.extend(extra);
+ }
+ by_hash = group(&docs)?;
+ Ok(by_hash)
+}
+
+/// Page one type by key (see the module docs). Returns the rows and whether every page was
+/// ordered and within its bound — a page that is not says the node did not honor the query.
+async fn keyset_scan(
+ src: &impl RefDocSource,
+ doc_type: &str,
+) -> Result<(Vec, bool)> {
+ let mut out: Vec = Vec::new();
+ let mut ordered = true;
+ let mut after: Option<[u8; 32]> = None;
+ for _ in 0..MAX_KEYSET_ROUNDS {
+ let page = src.keyset_page(doc_type, after).await?;
+ let hashes = page
+ .iter()
+ .map(|d| ref_hash(d, doc_type))
+ .collect::>>()?;
+ ordered &= hashes.windows(2).all(|w| w[0] <= w[1])
+ && after.is_none_or(|a| hashes.iter().all(|h| *h > a));
+
+ if page.len() < KEYSET_PAGE as usize {
+ out.extend(page);
+ return Ok((out, ordered));
+ }
+ let last = *hashes.last().expect("a full page is non-empty");
+ let cut = hashes.iter().position(|h| *h == last).unwrap_or(0);
+ if cut == 0 {
+ // One ref filled the page: read it on its own, then move past it.
+ out.extend(src.ref_history(doc_type, last).await?);
+ after = Some(last);
+ } else {
+ // Every ref before the last is whole; the last may be cut off, so it is
+ // re-read from its start on the next page.
+ after = Some(hashes[cut - 1]);
+ out.extend(page.into_iter().take(cut));
+ }
+ }
+ Err(Error::IncompleteRead {
+ document_type: doc_type.to_string(),
+ fetched: out.len(),
+ reason: format!("the ref keyset scan did not finish in {MAX_KEYSET_ROUNDS} rounds"),
+ })
+}
+
+/// Group rows per ref. Within a ref, plain updates come before protected ones and each
+/// source keeps its read order; the fold re-sorts by `(createdAt, id)` regardless.
+fn group(docs: &[(Vec, bool, &str)]) -> Result {
+ let mut by_hash = RefHistories::new();
+ for (rows, protected, doc_type) in docs {
+ for d in rows {
+ let hash = ref_hash(d, doc_type)?;
+ by_hash.entry(hash).or_default().push(ref_update_from_doc(
+ d,
+ &hex::encode(hash),
+ *protected,
+ ));
+ }
+ }
+ Ok(by_hash)
+}
+
+/// The document's `refNameHash`. A row without one cannot be attributed to a ref; the schema
+/// forbids it, and skipping it would make the answer wrong rather than partial.
+fn ref_hash(d: &FetchedDocument, doc_type: &str) -> Result<[u8; 32]> {
+ d.field_bytes("refNameHash")
+ .and_then(|b| <[u8; 32]>::try_from(b).ok())
+ .ok_or_else(|| Error::Platform(format!("{doc_type} {} has no 32-byte refNameHash", d.id)))
+}
+
+/// Whether some update's non-null `prevOid` is no update's `newOid` in the same ref — a
+/// parent that should have been read and was not. Parity: forge-web `hasMissingParent`.
+pub fn has_missing_parent(updates: &[RefUpdate]) -> bool {
+ let tips: BTreeSet<&str> = updates.iter().map(|u| u.new_oid.as_str()).collect();
+ updates.iter().any(|u| {
+ let null = u.prev_oid.is_empty() || u.prev_oid.bytes().all(|b| b == b'0');
+ !null && !tips.contains(u.prev_oid.as_str())
+ })
+}
+
+/// Flatten a ref-update document to the [`RefUpdate`] shape the fold consumes.
+pub(crate) fn ref_update_from_doc(
+ d: &FetchedDocument,
+ hash_hex: &str,
+ protected: bool,
+) -> RefUpdate {
+ RefUpdate {
+ id: d.id.clone(),
+ ref_name_hash: hash_hex.to_string(),
+ ref_name: d.field_str("refName").unwrap_or_default(),
+ prev_oid: d.field_hex("prevOid").unwrap_or_default(),
+ new_oid: d.field_hex("newOid").unwrap_or_default(),
+ force: d.field_bool("force"),
+ protected,
+ author: d.owner_id.clone(),
+ created_at: d.created_at.unwrap_or(0),
+ }
+}
+
+#[cfg(test)]
+// The mocks below answer synchronously; `async fn` keeps them shaped like the live source.
+#[allow(clippy::unused_async_trait_impl)]
+mod tests {
+ use super::{
+ has_missing_parent, read_all_with, ref_history_with, RefDocSource, DOC_REF_UPDATE,
+ KEYSET_PAGE,
+ };
+ use crate::error::Result;
+ use crate::platform::{FetchedDocument, FieldValue};
+ use std::cell::{Cell, RefCell};
+ use std::collections::{BTreeMap, BTreeSet};
+
+ /// A refUpdate row. `ref_seed` picks the ref (its hash is `[ref_seed; 32]`), `id` its
+ /// `$id`; oids are one byte repeated so a chain is easy to spell.
+ fn row(ref_seed: u8, id: u32, prev: u8, new: u8) -> FetchedDocument {
+ let mut fields = BTreeMap::new();
+ fields.insert("refNameHash".into(), FieldValue::Bytes32([ref_seed; 32]));
+ fields.insert(
+ "refName".into(),
+ FieldValue::Text(format!("refs/heads/r{ref_seed}")),
+ );
+ fields.insert("newOid".into(), FieldValue::Bytes(vec![new; 20]));
+ if prev != 0 {
+ fields.insert("prevOid".into(), FieldValue::Bytes(vec![prev; 20]));
+ }
+ FetchedDocument {
+ id: format!("id{id:06}"),
+ owner_id: "pusher".into(),
+ // The deployed repo-v1 type never recorded `$createdAt` (design-freeze-2 §3).
+ created_at: None,
+ fields,
+ }
+ }
+
+ /// A ref with `n` linear updates (1 → 2 → … → n), ids drawn from `ids`.
+ fn chain(ref_seed: u8, n: u8, ids: &mut u32) -> Vec {
+ (1..=n)
+ .map(|k| {
+ *ids += 1;
+ row(ref_seed, *ids, k - 1, k)
+ })
+ .collect()
+ }
+
+ fn hash_of(d: &FetchedDocument) -> [u8; 32] {
+ <[u8; 32]>::try_from(d.field_bytes("refNameHash").unwrap()).unwrap()
+ }
+
+ /// An in-memory Drive for one document type, in `refState` order
+ /// (`refNameHash`, then — `$createdAt` being absent — `$id`).
+ ///
+ /// `cursor_page` reproduces the protocol-13 multi-branch cursor lowering: the cursor
+ /// document's `$id` bound leaks into every later `refNameHash` branch, so rows there
+ /// with a smaller `$id` are skipped. `drop_on_keyset` removes one id from keyset pages,
+ /// standing in for a node that answers a correct query incompletely.
+ struct MockDrive {
+ rows: Vec,
+ drop_on_keyset: Option,
+ keyset_calls: Cell,
+ history_calls: RefCell>,
+ full_scans: Cell,
+ }
+
+ impl MockDrive {
+ fn new(mut rows: Vec) -> Self {
+ rows.sort_by(|a, b| hash_of(a).cmp(&hash_of(b)).then_with(|| a.id.cmp(&b.id)));
+ Self {
+ rows,
+ drop_on_keyset: None,
+ keyset_calls: Cell::new(0),
+ history_calls: RefCell::new(Vec::new()),
+ full_scans: Cell::new(0),
+ }
+ }
+
+ /// The `startAfter` page the old reader requested, with the v0 sibling-branch drop.
+ fn cursor_page(&self, after_id: Option<&str>) -> Vec {
+ let Some(after_id) = after_id else {
+ return self
+ .rows
+ .iter()
+ .take(KEYSET_PAGE as usize)
+ .cloned()
+ .collect();
+ };
+ let cursor = self.rows.iter().find(|d| d.id == after_id).unwrap();
+ let pos = self.rows.iter().position(|d| d.id == after_id).unwrap();
+ self.rows[pos + 1..]
+ .iter()
+ .filter(|d| hash_of(d) == hash_of(cursor) || d.id > cursor.id)
+ .take(KEYSET_PAGE as usize)
+ .cloned()
+ .collect()
+ }
+ }
+
+ impl RefDocSource for MockDrive {
+ async fn keyset_page(
+ &self,
+ _doc_type: &str,
+ after: Option<[u8; 32]>,
+ ) -> Result> {
+ self.keyset_calls.set(self.keyset_calls.get() + 1);
+ Ok(self
+ .rows
+ .iter()
+ .filter(|d| after.is_none_or(|a| hash_of(d) > a))
+ .filter(|d| self.drop_on_keyset.as_deref() != Some(d.id.as_str()))
+ .take(KEYSET_PAGE as usize)
+ .cloned()
+ .collect())
+ }
+
+ async fn ref_history(
+ &self,
+ _doc_type: &str,
+ hash: [u8; 32],
+ ) -> Result> {
+ self.history_calls.borrow_mut().push(hash);
+ Ok(self
+ .rows
+ .iter()
+ .filter(|d| hash_of(d) == hash)
+ .cloned()
+ .collect())
+ }
+
+ async fn full_scan(&self, _doc_type: &str) -> Result> {
+ self.full_scans.set(self.full_scans.get() + 1);
+ Ok(self.rows.clone())
+ }
+ }
+
+ /// Serves only the plain type; protected reads come back empty.
+ struct PlainOnly(MockDrive);
+
+ impl RefDocSource for PlainOnly {
+ async fn keyset_page(
+ &self,
+ t: &str,
+ after: Option<[u8; 32]>,
+ ) -> Result> {
+ if t == DOC_REF_UPDATE {
+ self.0.keyset_page(t, after).await
+ } else {
+ Ok(Vec::new())
+ }
+ }
+ async fn ref_history(&self, t: &str, hash: [u8; 32]) -> Result> {
+ if t == DOC_REF_UPDATE {
+ self.0.ref_history(t, hash).await
+ } else {
+ Ok(Vec::new())
+ }
+ }
+ async fn full_scan(&self, t: &str) -> Result> {
+ if t == DOC_REF_UPDATE {
+ self.0.full_scan(t).await
+ } else {
+ Ok(Vec::new())
+ }
+ }
+ }
+
+ /// 60 refs × 1..4 updates with ids interleaved across refs — the shape of the nightly
+ /// repo, where every run pushes several refs and ids are random with respect to hashes.
+ fn nightly_like() -> Vec {
+ let mut ids = 0u32;
+ let mut rows = Vec::new();
+ for r in 1..=60u8 {
+ rows.extend(chain(r, 1 + r % 4, &mut ids));
+ }
+ // Scramble ids against hash order: reverse them.
+ let n = rows.len();
+ for (i, d) in rows.iter_mut().enumerate() {
+ d.id = format!("id{:06}", n - i);
+ }
+ rows
+ }
+
+ #[test]
+ fn the_mock_reproduces_the_cursor_drop() {
+ // The old reader: refState order, paged by `startAfter`, stop at a short page.
+ let drive = MockDrive::new(nightly_like());
+ let total = drive.rows.len();
+ assert!(total > KEYSET_PAGE as usize, "needs more than one page");
+ let mut got = 0;
+ let mut after: Option = None;
+ loop {
+ let page = drive.cursor_page(after.as_deref());
+ got += page.len();
+ if page.len() < KEYSET_PAGE as usize {
+ break;
+ }
+ after = page.last().map(|d| d.id.clone());
+ }
+ assert!(
+ got < total,
+ "the cursor read should lose rows ({got} of {total})"
+ );
+ }
+
+ #[tokio::test]
+ async fn keyset_scan_reads_every_update_without_a_cursor() {
+ let drive = PlainOnly(MockDrive::new(nightly_like()));
+ let total = drive.0.rows.len();
+ let got = read_all_with(&drive).await.unwrap();
+ assert_eq!(got.len(), 60);
+ assert_eq!(got.values().map(Vec::len).sum::(), total);
+ assert_eq!(
+ drive.0.full_scans.get(),
+ 0,
+ "a consistent scan needs no fallback"
+ );
+ // ~2 pages for ~150 rows, not one round-trip per ref.
+ assert!(
+ drive.0.keyset_calls.get() <= 3,
+ "{} pages",
+ drive.0.keyset_calls.get()
+ );
+ }
+
+ #[tokio::test]
+ async fn a_ref_that_fills_a_page_is_read_on_its_own() {
+ let mut ids = 0u32;
+ let mut rows = chain(5, 3, &mut ids);
+ rows.extend(chain(7, 150, &mut ids)); // one hot ref, more than a page
+ rows.extend(chain(9, 2, &mut ids));
+ let drive = PlainOnly(MockDrive::new(rows));
+ let got = read_all_with(&drive).await.unwrap();
+ assert_eq!(got[&[7; 32]].len(), 150);
+ assert_eq!(got[&[5; 32]].len(), 3);
+ assert_eq!(got[&[9; 32]].len(), 2);
+ assert_eq!(*drive.0.history_calls.borrow(), vec![[7u8; 32]]);
+ assert_eq!(drive.0.full_scans.get(), 0);
+ }
+
+ #[tokio::test]
+ async fn a_missing_parent_falls_back_to_the_full_scan_and_unions() {
+ let mut rows = nightly_like();
+ // Drop the middle update of a 3-update ref from keyset pages only.
+ let victim = rows
+ .iter()
+ .find(|d| hash_of(d) == [2; 32] && d.field_hex("newOid") == Some("02".repeat(20)))
+ .unwrap()
+ .id
+ .clone();
+ rows.sort_by(|a, b| a.id.cmp(&b.id));
+ let mut drive = MockDrive::new(rows);
+ drive.drop_on_keyset = Some(victim);
+ let drive = PlainOnly(drive);
+ let got = read_all_with(&drive).await.unwrap();
+ assert_eq!(drive.0.full_scans.get(), 1);
+ assert_eq!(got[&[2; 32]].len(), 3, "the dropped update is recovered");
+ assert_eq!(
+ got.values().map(Vec::len).sum::(),
+ drive.0.rows.len(),
+ "and nothing is counted twice"
+ );
+ }
+
+ #[tokio::test]
+ async fn ref_history_reads_both_types() {
+ let mut ids = 0u32;
+ let drive = MockDrive::new(chain(3, 4, &mut ids));
+ // The mock serves the same rows for both types; each is tagged with its source.
+ let got = ref_history_with(&drive, [3; 32]).await.unwrap();
+ assert_eq!(got.len(), 8);
+ assert_eq!(got.iter().filter(|u| u.protected).count(), 4);
+ let ids: BTreeSet<&str> = got.iter().map(|u| u.id.as_str()).collect();
+ assert_eq!(ids.len(), 4);
+ }
+
+ #[test]
+ fn missing_parent_detection() {
+ let u = |prev: &str, new: &str| crate::rules::RefUpdate {
+ id: new.into(),
+ ref_name_hash: "h".into(),
+ ref_name: "refs/heads/x".into(),
+ prev_oid: prev.into(),
+ new_oid: new.into(),
+ force: false,
+ protected: false,
+ author: "a".into(),
+ created_at: 0,
+ };
+ assert!(!has_missing_parent(&[u("", "aa"), u("aa", "bb")]));
+ assert!(
+ !has_missing_parent(&[u("0000", "aa"), u("aa", "0000")]),
+ "create + delete"
+ );
+ assert!(has_missing_parent(&[u("", "aa"), u("cc", "dd")]));
+ }
+}
diff --git a/crates/forge-core/src/repo.rs b/crates/forge-core/src/repo.rs
index 70a7109d1..1628f1421 100644
--- a/crates/forge-core/src/repo.rs
+++ b/crates/forge-core/src/repo.rs
@@ -25,10 +25,10 @@ use crate::backends::{ByteRange, PackBackend, PackMeta, PlatformBackend, Uri};
use crate::error::{Error, Result};
use crate::keystore::BridgeIdentity;
use crate::platform::{
- self, FetchedDocument, FieldValue, JournalStore, LoadedContract, LoadedIdentity,
- PlatformClient, PushJournal, QueryFilter, QueryOrder, WriteEngine, WriteIntent,
+ self, FieldValue, JournalStore, LoadedContract, LoadedIdentity, PlatformClient, PushJournal,
+ QueryFilter, QueryOrder, WriteEngine, WriteIntent,
};
-use crate::rules::{self, ConfigDoc, RefState, RefUpdate};
+use crate::rules::{self, ConfigDoc, RefState};
use crate::storage::{PackReader, Replication, StorageTarget};
/// The repo-v1 contract template (2 tokens + 15 doc types), embedded at build time.
@@ -44,8 +44,7 @@ const REPO_V1_TEMPLATE: &str = include_str!(concat!(
// Document type names (repo contract).
const DOC_CONFIG: &str = "config";
-const DOC_REF_UPDATE: &str = "refUpdate";
-const DOC_PROTECTED_REF_UPDATE: &str = "protectedRefUpdate";
+use crate::refs::{DOC_PROTECTED_REF_UPDATE, DOC_REF_UPDATE};
const DOC_PACK_MANIFEST: &str = "packManifest";
const DOC_MANIFEST_PART: &str = "manifestPart";
const DOC_CHUNK: &str = "chunk";
@@ -712,17 +711,19 @@ impl<'a> RepoService<'a> {
/// Enumerate every ref and its resolved [`RefState`].
///
- /// Both ref-update types are read whole, paged in `$createdAt` order, and grouped by
- /// `refNameHash` locally; each ref's combined update history + the repo's `config`
- /// history is folded by [`crate::rules::resolve_ref`].
+ /// Every ref's history comes from [`crate::refs::read_all_ref_updates`] — a keyset scan
+ /// over the `refState` index, ⌈updates/100⌉ queries per type plus one per ref that fills
+ /// a page by itself, with a `prevOid` completeness check — and each ref's combined
+ /// update history + the repo's `config` history is folded by
+ /// [`crate::rules::resolve_ref`].
///
- /// This used to discover refs with the S0.8 skip-scan (one `limit 1` query per ref per
- /// type) and then read each ref's history separately: about four sequential round-trips
- /// per ref, *including deleted ones*, since a delete is just another update. Every
- /// `git` command on a `dash://` remote reads refs at least once, so on the nightly's
- /// test repo (≈80 refs ever pushed, 170 updates) that was ~300 queries and over a
- /// minute per command — the reason a partial clone, which reads refs twice, outlived
- /// its command timeout. Paging costs ⌈updates/100⌉ queries per type instead.
+ /// Two earlier readers failed in opposite ways. The S0.8 skip-scan (one `limit 1` query
+ /// per ref per type, then one history read per ref) cost about four sequential
+ /// round-trips per ref ever pushed, deleted ones included: over a minute per `git`
+ /// command on the nightly repo. Paging the `refState` index with a `startAfter` cursor
+ /// was fast but silently lost rows on protocol 13 (see the `refs` module docs). Paging
+ /// the whole `reflog` index was correct but read every update of every ref on every
+ /// command, so it is kept only as the fallback the completeness check falls to.
///
/// The ancestry predicate is reflexive-only here (M1 has no read-side commit graph):
/// fast-forward supersession via `prevOid` still resolves, but descend-detection is
@@ -732,7 +733,7 @@ impl<'a> RepoService<'a> {
let repo_contract = self.client.fetch_contract(&repo.repo_contract_id).await?;
let configs = self.fetch_config_history(&repo_contract).await?;
- let by_hash = self.read_all_ref_updates(&repo_contract).await?;
+ let by_hash = crate::refs::read_all_ref_updates(self.client, &repo_contract).await?;
let mut out = Vec::with_capacity(by_hash.len());
for (hash, updates) in &by_hash {
@@ -1976,70 +1977,6 @@ impl<'a> RepoService<'a> {
})
.collect())
}
-
- /// Read every ref's full update history, grouped by `refNameHash`: both ref-update types
- /// paged to exhaustion in `$createdAt` order. Within a ref, plain updates come before
- /// protected ones and each source is `$createdAt asc` — the order the per-ref fold has
- /// always consumed (the fold itself re-sorts by `(createdAt, id)`).
- ///
- /// Paged to exhaustion: `resolve_ref` folds the whole causal chain, so stopping at one
- /// page would pin a branch at a stale tip — `list` would advertise it, clones would get
- /// a stale HEAD, and `plan_pushes` would call correct fast-forwards non-fast-forward.
- ///
- /// A document without a 32-byte `refNameHash` cannot be attributed to a ref; the
- /// schema forbids it, and skipping it silently would make the answer wrong rather than
- /// partial, so it fails the read.
- async fn read_all_ref_updates(
- &self,
- repo_contract: &LoadedContract,
- ) -> Result>> {
- let mut by_hash: BTreeMap<[u8; 32], Vec> = BTreeMap::new();
- for (doc_type, protected) in [(DOC_REF_UPDATE, false), (DOC_PROTECTED_REF_UPDATE, true)] {
- let docs = self
- .client
- .query_all_documents(
- repo_contract,
- doc_type,
- &[],
- // The `reflog` index. NOT `refState` (`refNameHash, $createdAt`): on
- // testnet, `start_after` paging over that compound index dropped 5 of
- // 168 updates at the page boundary (one ref vanished from `ls-remote`),
- // while `$createdAt` paging returns all of them.
- &[QueryOrder::asc("$createdAt")],
- )
- .await?;
- for d in &docs {
- let hash = d
- .field_bytes("refNameHash")
- .and_then(|b| <[u8; 32]>::try_from(b).ok())
- .ok_or_else(|| {
- Error::Platform(format!("{doc_type} {} has no 32-byte refNameHash", d.id))
- })?;
- by_hash.entry(hash).or_default().push(ref_update_from_doc(
- d,
- &hex::encode(hash),
- protected,
- ));
- }
- }
- Ok(by_hash)
- }
-}
-
-/// Flatten a `refUpdate` / `protectedRefUpdate` document to the [`RefUpdate`] shape
-/// [`crate::rules::resolve_ref`] consumes.
-fn ref_update_from_doc(d: &FetchedDocument, hash_hex: &str, protected: bool) -> RefUpdate {
- RefUpdate {
- id: d.id.clone(),
- ref_name_hash: hash_hex.to_string(),
- ref_name: d.field_str("refName").unwrap_or_default(),
- prev_oid: d.field_hex("prevOid").unwrap_or_default(),
- new_oid: d.field_hex("newOid").unwrap_or_default(),
- force: d.field_bool("force"),
- protected,
- author: d.owner_id.clone(),
- created_at: d.created_at.unwrap_or(0),
- }
}
/// The initial `config` document properties (`defaultBranch`, empty protected patterns,
diff --git a/crates/forge-core/src/rules.rs b/crates/forge-core/src/rules.rs
index b6ed4edfb..b39f275ec 100644
--- a/crates/forge-core/src/rules.rs
+++ b/crates/forge-core/src/rules.rs
@@ -110,7 +110,7 @@ impl Ancestry {
// ===========================================================================
/// A single append-only `refUpdate` / `protectedRefUpdate` document, flattened to the
-/// fields resolution needs. Callers fetch these (via the §2.3 skip-scan + §3
+/// fields resolution needs. Callers fetch these (via the §2.3 keyset scan, `crate::refs`, + §3
/// completeness fallback) and hand the slice in.
#[derive(Debug, Clone, Serialize, Deserialize)]
#[serde(rename_all = "camelCase")]
diff --git a/docs/contracts/data-contracts.md b/docs/contracts/data-contracts.md
index 65a2f7341..6bbc662c1 100644
--- a/docs/contracts/data-contracts.md
+++ b/docs/contracts/data-contracts.md
@@ -103,7 +103,18 @@ Why the non-deletables: a deletable `refUpdate` lets whoever authored the curren
Precise "supersedes" rule (pinned in `FORGE_RULES_V1`, Stage 2): the resolver tracks *live heads*. A head is superseded by a strictly-newer update that (a) is a delete or `force`, (b) fast-forwards off it (`prevOid == that head's newOid`), or (c) descends from it. **Consequence to note:** fast-forwarding only *one* racing head leaves the other still live → the ref stays `Diverged` (the strict reading of "supersedes both", not "newest wins"). Total-order tiebreak everywhere is `($createdAt, then $id)` — including which of two same-`$createdAt` configs applies (greatest `$id`). A merge event whose `oid` is unreachable from the base (or with no base tip supplied) is treated as inert.
-Branch/tag enumeration (no distinct-values query on Platform): **skip-scan with `limit 1` hops** — query `refNameHash > ` orderBy `refNameHash` **limit 1** to seek directly to the next distinct name (a `limit 100` page here would return *rows*, and one hyperactive ref with thousands of updates would make enumeration O(total pushes), not O(refs)); then batch tip lookups (`in` on ≤ 100 hashes with the §3 completeness fallback). Cost: one cheap seek per distinct ref — bounded by real branch/tag counts. **Skip-scan verified in S0.8 ✅** — a flat O(log n) seek regardless of duplicate-hash count. **Load-bearing:** `byteArray` where-operands (`refNameHash`, `packHash`, any oid/hash tip query) must be passed as **base64 strings** (not Uint8Array, not base58 — base58 is for identifiers only); results also return base64.
+Branch/tag enumeration (no distinct-values query on Platform) is a **keyset scan over the `refState` index**, per ref-update type (forge-core `refs.rs`, forge-web `lib/repo/refs.ts`; they behave the same):
+
+1. Query `refNameHash > `, ordered `(refNameHash, $createdAt)`, `limit 100`, **with no `startAfter` cursor**. On a full page, every ref except the last one is complete. The next page starts after the last complete ref, so a ref the page cut off is read again from its start.
+2. If a single ref fills a whole page, read it on its own with `refNameHash == h` (paged to exhaustion) and resume after it.
+3. **Completeness check:** every non-null `prevOid` must be some `newOid` of the same ref. A dangling parent, or a page that comes back out of order, triggers the fallback: read every update in `$createdAt` (`reflog`) order and union the result by `$id`.
+
+Cost: ⌈updates/100⌉ queries per type, plus one per ref with more than 100 updates. On the nightly repo that is 3 queries for 229 updates across 129 refs, and it returns the same 229 updates as the full reflog read. Two earlier designs failed, one on cost and one on correctness:
+
+- **`limit 1` skip-scan (S0.8):** one sequential seek per distinct ref, then a history read per ref. That meant about four round-trips per ref ever pushed, deleted refs included, and more than a minute per `git` command at about 80 refs.
+- **`startAfter` cursor over `refState`:** it silently loses rows on protocol 13. On the nightly repo it returned 197 of 229 updates. Drive's v0 path-query lowering applies the cursor document's lower index levels to *every* sibling `refNameHash` branch, not just the cursor's own branch, so later refs lose rows and a page can come back short, which a pager reads as the end. This is the same family as dashpay/platform#4396, which fixed the `in`-clause shape at protocol 14 and lists this `orderBy`-only shape as still unfixed. **Never page a multi-branch index with a document cursor.** An equality prefix (one branch) or a key range with no cursor is safe.
+
+A single ref's history, for PR base tips, is an equality read (`refNameHash == h`), so its cost scales with that ref's pushes, not the repo's. **Load-bearing:** `byteArray` where-operands (`refNameHash`, `packHash`, any oid/hash tip query) must be passed as **base64 strings** (not Uint8Array, not base58 — base58 is for identifiers only); results also return base64.
**`packManifest`** — `packHash` hash32; `kind` integer (**0 = git pack, 1 = objectLocator, 2 = flatIndex** — browse-plane artifacts share the entire pack storage/transport machinery); `sizeBytes`, `objectCount`, `chunkCount` integers; `storage` 0 platform / 1 external; `uris` list ≤ 8 of ≤ 300; `tips` list ≤ 16 oid (for kind 2: the tip commit it indexes); `supersedes` list ≤ 32 hash32; `offsetIndexParts` integer. **The list fields (`uris`, `tips`, `supersedes`) serialize as JSON-string / packed byteArray, not native arrays (§0).** Indices: unique `(packHash)`; `($createdAt desc)`; `(kind, $createdAt desc)` — readers grab newest locator/flatIndex in one query.
@@ -152,7 +163,7 @@ Count trees cost per-index storage overhead; they are assigned **only** where a
| release count | plain list query — newest-wins supersedes make raw doc counts wrong, and release pages are small | O(releases) |
| chunk availability audit | countable `chunk(packHash, seq)` vs `manifest.chunkCount` | O(1) |
| **open vs closed issue/PR counts** | **not natively countable — by design, not omission**: state is a fold of `event` docs (mutation ownership forbids an authoritative indexed `state` field on the author-owned doc). Strategy: list pages fold events per page (one `in` query on ≤ 100 targetIds); results cached in IndexedDB keyed by newest-event cursor; tabs render "Issues (N)" instantly and "open/closed" splits hydrate. Optional v1.1: MAINTAIN-posted `stateSummary` cache doc, explicitly non-authoritative, always corrected by fold. | fold |
-| branch/tag counts | skip-scan enumeration (cached); no count tree (distinct-count ≠ doc-count) | O(refs) |
+| branch/tag counts | keyset-scan enumeration (§2.3); no count tree (distinct-count ≠ doc-count) | O(updates / 100) |
Not countable (no story pays for the overhead): `refUpdate`, `event`, `manifestPart`, `checkRun`, `label`, `release`, `webhook`, `profile`. Flag validity for the unique+countable combinations is a named S0.6 check.
diff --git a/docs/design-freeze-2.md b/docs/design-freeze-2.md
index caad83c40..ab4b98d2e 100644
--- a/docs/design-freeze-2.md
+++ b/docs/design-freeze-2.md
@@ -20,7 +20,7 @@ The thesis is proven, not asserted:
5. **Trusted-connect proof-verifies plain reads** — the explicit `*WithProof` facade is redundant *and* browser-buggy in evo-sdk 4.0.0; forge-web uses plain `query()`/`get()` (still verified in trusted mode).
6. **Repo-v1 economics: instantiation ~1.18 DASH** (2 tokens + 15 doc types + count-trees), registry ~0.68 DASH — materially above the early <0.02 estimate; count-tree indices are the main driver. tokenCost payments **recirculate to the repo owner**, not burned.
7. **Contract templates: solo-owner by default.** A MainGroup needs ≥2 members; the source template now uses `ContractOwner` admin (org multi-sig is an opt-in v2 variant). Positions must be top-level-contiguous. Deployed contracts predate these and work via name-addressing + runtime compat shims; only new repos inherit the fixes.
-8. **Query completeness is mandatory, not optional** — `in`-batch starvation and the 100-row default truncate authz/fold reads; every fold/authz path paginates to exhaustion + the per-key completeness fallback. Ref enumeration uses flat `limit-1` skip-scan.
+8. **Query completeness is mandatory, not optional** — `in`-batch starvation and the 100-row default truncate authz/fold reads; every fold/authz path paginates to exhaustion + the per-key completeness fallback. Ref enumeration is a cursor-free keyset scan over `refState` (`refNameHash > last`, 100-row pages) with a `prevOid` completeness check that falls back to the full reflog read. **Superseded:** it was the `limit-1` skip-scan (one round-trip per ref, too slow), then briefly a `startAfter`-paged `refState` read, which drops sibling-branch rows on protocol 13 (data-contracts §2.3).
9. **Shallow clone dropped; partial clone kept** (S0.9) — a fetch/push helper has no depth reply channel; `--depth` fails loudly, `--filter=blob:none` works via `.promisor`. **jj works unmodified** (gitoxide).
10. **objectLocator: 36-byte fixed-stride rows** (deltaChainSpan fixed 4-byte, not varint; +1-byte delta-depth hint); single-span read for blobs, per-base walk for deep-delta trees. Cold browse loads root-tree-via-locator, not the O(files) flatIndex.
11. **CI reality**: per-push CI runs the web app + the 70-vector TS parity suite; the Rust workspace's large Platform dependency tree does not fit a per-push job for unrelated changes, so `rust.yml` gates pull requests and pushes that touch Rust paths, plus nightly. **Superseded in part:** the Platform SDK is no longer a path dep on a sibling checkout — it is a git dependency pinned to an immutable tag, so nothing clones the monorepo and `rust.yml` is the Rust gate rather than a backstop.
diff --git a/forge-web/lib/repo/index.ts b/forge-web/lib/repo/index.ts
index 8026f8756..a74184df7 100644
--- a/forge-web/lib/repo/index.ts
+++ b/forge-web/lib/repo/index.ts
@@ -38,7 +38,8 @@ export {
} from './config'
export {
branchesOf,
- enumerateRefHashes,
+ hasMissingParent,
+ readAllRefUpdates,
readRefUpdates,
readRefs,
resolveRefByHash,
diff --git a/forge-web/lib/repo/refs.test.ts b/forge-web/lib/repo/refs.test.ts
index b3ab1b936..45e47e3b6 100644
--- a/forge-web/lib/repo/refs.test.ts
+++ b/forge-web/lib/repo/refs.test.ts
@@ -1,7 +1,10 @@
/**
- * readRefs paths: the one-page fast path (small repos resolve from two parallel queries,
- * grouped locally) and the skip-scan fallback (a full page means the update set may be
- * incomplete, so enumeration must go back through `> last limit 1` hops + per-ref reads).
+ * readRefs: the keyset scan over the `refState` index, against a mock Drive that serves
+ * `refNameHash > x` pages exactly and reproduces the protocol-13 cursor bug for `startAfter`
+ * (the cursor's `$id` bound leaks into every later `refNameHash` branch). The reader must
+ * never send a cursor on that multi-branch query, must read a page-filling ref on its own,
+ * and must fall back to the reflog read when a `prevOid` has no parent in what came back.
+ * Parity: forge-core `refs::tests`.
*/
import type { EvoSDK } from '@dashevo/evo-sdk'
@@ -9,9 +12,9 @@ import { sha256 } from '@noble/hashes/sha2.js'
import { bytesToHex } from '@noble/hashes/utils.js'
import { describe, expect, it } from 'vitest'
-import { bytesToBase64 } from '../sdk'
+import { base64ToHex, bytesToBase64 } from '../sdk'
import { DOC, type RepoRef } from './contract'
-import { readRefs } from './refs'
+import { hasMissingParent, readRefs } from './refs'
const REPO: RepoRef = { contractId: 'contract', ownerId: 'owner' }
@@ -20,129 +23,201 @@ function refHashBytes(seed: number): Uint8Array {
return sha256(new TextEncoder().encode(`refs/heads/ref-${seed}`))
}
-function hashB64(seed: number): string {
- return bytesToBase64(refHashBytes(seed))
-}
-
function refHashHex(seed: number): string {
return bytesToHex(refHashBytes(seed))
}
-function hashHex(seed: number): string {
- return Array.from({ length: 32 }, () => seed.toString(16).padStart(2, '0')).join('')
+function oidHex(seed: number): string {
+ return Array.from({ length: 20 }, () => seed.toString(16).padStart(2, '0')).join('')
}
-let nextId = 0
-function updateDoc(
- hashSeed: number,
- oidSeed: number,
- createdAt: number,
- prevOidSeed: number | null = null,
-): Record {
- nextId += 1
+type Doc = Record
+
+/** A refUpdate row; `$createdAt` is absent, as on the deployed repo-v1 type. */
+function updateDoc(id: string, refSeed: number, newSeed: number, prevSeed = 0): Doc {
return {
- $id: `doc-${nextId}`,
+ $id: id,
$ownerId: 'pusher',
- $createdAt: createdAt,
- refNameHash: hashB64(hashSeed),
- refName: `refs/heads/ref-${hashSeed}`,
- prevOid: prevOidSeed === null ? null : bytesToBase64(new Uint8Array(32).fill(prevOidSeed)),
- newOid: bytesToBase64(new Uint8Array(32).fill(oidSeed)),
+ refNameHash: bytesToBase64(refHashBytes(refSeed)),
+ refName: `refs/heads/ref-${refSeed}`,
+ prevOid: prevSeed === 0 ? null : bytesToBase64(new Uint8Array(20).fill(prevSeed)),
+ newOid: bytesToBase64(new Uint8Array(20).fill(newSeed)),
force: false,
}
}
+/** A ref with `n` linear updates 1 → 2 → … → n. */
+function chain(refSeed: number, n: number, nextId: () => string): Doc[] {
+ return Array.from({ length: n }, (_, k) => updateDoc(nextId(), refSeed, k + 1, k))
+}
+
interface QueryLike {
documentTypeName: string
where?: readonly (readonly [string, string, unknown])[]
orderBy?: readonly (readonly [string, string])[]
limit?: number
+ startAfter?: string
}
-/** Mock SDK whose `documents.query` is routed through `dispatch`; records every query. */
-function mockSdk(
- dispatch: (q: QueryLike) => Record[],
- seen: QueryLike[],
-): EvoSDK {
- return {
+const hexOf = (d: Doc): string => base64ToHex(d['refNameHash'] as string)
+
+/**
+ * In-memory Drive for `refUpdate` (protected is empty). Rows are kept in `refState` order
+ * (`refNameHash`, then `$id` — `$createdAt` being absent). `dropOnKeyset` removes one `$id`
+ * from keyset pages only, standing in for a node answering a correct query incompletely.
+ */
+function mockDrive(rows: Doc[], opts: { dropOnKeyset?: string } = {}) {
+ const sorted = [...rows].sort((a, b) =>
+ hexOf(a) < hexOf(b) ? -1 : hexOf(a) > hexOf(b) ? 1 : String(a['$id']) < String(b['$id']) ? -1 : 1,
+ )
+ const seen: QueryLike[] = []
+ const page = (docs: Doc[], q: QueryLike): Doc[] => {
+ let from = 0
+ if (q.startAfter !== undefined) {
+ const cursor = docs.find((d) => d['$id'] === q.startAfter) as Doc
+ from = docs.indexOf(cursor) + 1
+ // The protocol-13 lowering: the cursor's `$id` bound applies in sibling branches too.
+ if (q.orderBy?.[0]?.[0] === 'refNameHash') {
+ docs = docs.filter(
+ (d, i) => i < from || hexOf(d) === hexOf(cursor) || String(d['$id']) > String(cursor['$id']),
+ )
+ from = docs.indexOf(cursor) + 1
+ }
+ }
+ return docs.slice(from, from + (q.limit ?? 100))
+ }
+ const sdk = {
documents: {
query: (q: QueryLike): Promise
+ {state.unavailable.length > 0 ? : null}
{children(state.reader)}
)
diff --git a/forge-web/hooks/use-browse-reader.ts b/forge-web/hooks/use-browse-reader.ts
index 26b7977d4..a8d79e24c 100644
--- a/forge-web/hooks/use-browse-reader.ts
+++ b/forge-web/hooks/use-browse-reader.ts
@@ -14,7 +14,7 @@ import { useBrowse } from '@/hooks/use-browse'
import { AUTO_LOAD_MAX_BYTES, useFallbackBrowse } from '@/hooks/use-fallback-browse'
import type { BrowseReader } from '@/lib/browse'
import type { RepoRef } from '@/lib/repo'
-import { formatBytes, type FallbackProgress } from '@/lib/view'
+import { formatBytes, type FallbackProgress, type UnavailablePack } from '@/lib/view'
export type BrowseReaderState =
| { readonly kind: 'loading'; readonly label: string }
@@ -24,6 +24,8 @@ export type BrowseReaderState =
/** Served from an in-browser clone because the published index is missing or behind. */
readonly local: boolean
readonly behind: boolean
+ /** Live external packs the in-browser clone could not fetch (empty: nothing skipped). */
+ readonly unavailable: readonly UnavailablePack[]
}
| { readonly kind: 'error'; readonly title?: string; readonly message: string; readonly retry: () => void }
| { readonly kind: 'no-packs' }
@@ -50,13 +52,19 @@ export function useBrowseReader(repo: RepoRef | null): BrowseReaderState {
if (error) return { kind: 'error', message: error, retry: reload }
if (data === null || data.kind === 'no-packs') return { kind: 'no-packs' }
if (data.kind === 'ready') {
- return { kind: 'ready', reader: data.context.reader, local: false, behind: false }
+ return { kind: 'ready', reader: data.context.reader, local: false, behind: false, unavailable: [] }
}
// No usable published index — the in-browser fallback clone takes over.
const behind = data.reason === 'index-behind'
if (fallback.status === 'ready' && fallback.context !== null) {
- return { kind: 'ready', reader: fallback.context.reader, local: true, behind }
+ return {
+ kind: 'ready',
+ reader: fallback.context.reader,
+ local: true,
+ behind,
+ unavailable: fallback.context.unavailable ?? [],
+ }
}
if (fallback.status === 'error') {
return {
diff --git a/forge-web/lib/browse/reader.ts b/forge-web/lib/browse/reader.ts
index 764d4f58a..22f7888f0 100644
--- a/forge-web/lib/browse/reader.ts
+++ b/forge-web/lib/browse/reader.ts
@@ -50,6 +50,11 @@ export interface BrowseReaderOptions {
* read). This is how the UI learns what was actually checked rather than assuming it.
*/
readonly onObject?: (verdict: ObjectVerdict) => void
+ /**
+ * The error for an OID the locator does not index. Defaults to `object not in locator`; a
+ * reader built over an incomplete pack set supplies one that names what is missing.
+ */
+ readonly missingObject?: (oidHex: string) => Error
}
/** Per-reader object-memo budget — readers live for the session (cached browse context). */
@@ -117,7 +122,9 @@ export class BrowseReader {
if (cached !== undefined) return cached
const entry = this.locate(oidHex)
- if (entry === null) throw new Error(`object not in locator: ${oidHex}`)
+ if (entry === null) {
+ throw this.opts.missingObject?.(oidHex) ?? new Error(`object not in locator: ${oidHex}`)
+ }
const obj = singleReadAdvised(entry)
? await this.readSpan(entry)
diff --git a/forge-web/lib/constants.ts b/forge-web/lib/constants.ts
index faf6c617e..88cbb060c 100644
--- a/forge-web/lib/constants.ts
+++ b/forge-web/lib/constants.ts
@@ -18,6 +18,7 @@ import {
type DeploymentFile,
type ForgeIds,
} from './deployments'
+import storageDefaults from '../../forge-contracts/config/storage-defaults.json'
// ---------------------------------------------------------------------------
// Network
@@ -233,6 +234,15 @@ export const QUORUM_KEY_ENDPOINT: Readonly> = {
devnet: quorumEndpoint(NETWORKS.devnet),
}
+/**
+ * Public IPFS gateways an `ipfs://` pack URI is fetched through, in order, after
+ * whatever `http(s)` mirrors the manifest itself records — the list every client shares
+ * (`forge-contracts/config/storage-defaults.json`, which forge-core embeds). Byte sources
+ * only: every external pack is sha256-checked against its proof-read manifest, so a gateway
+ * that lies or is down costs a retry, never integrity.
+ */
+export const IPFS_GATEWAYS: readonly string[] = storageDefaults.ipfsGateways
+
/** The config of the network this build targets. */
export const ACTIVE_NETWORK: NetworkConfig = NETWORKS[DEFAULT_NETWORK]
diff --git a/forge-web/lib/view/browse-fallback.test.ts b/forge-web/lib/view/browse-fallback.test.ts
index bf6baa931..38efac59e 100644
--- a/forge-web/lib/view/browse-fallback.test.ts
+++ b/forge-web/lib/view/browse-fallback.test.ts
@@ -9,7 +9,7 @@ import type { EvoSDK } from '@dashevo/evo-sdk'
import { sha256 } from '@noble/hashes/sha2.js'
import { bytesToHex } from '@noble/hashes/utils.js'
import { zlibSync } from 'fflate'
-import { describe, expect, it } from 'vitest'
+import { afterEach, describe, expect, it, vi } from 'vitest'
import { gitOidHex } from '../browse'
import {
@@ -25,6 +25,9 @@ import { CHUNK_PAYLOAD_MAX } from '../constants'
import type { PackManifest, RepoRef } from '../repo'
import { base64ToHex, bytesToBase64 } from '../sdk'
import { cachedFallback, startFallback, type FallbackProgress } from './browse-fallback'
+import { externalFetchUrls } from './browse-source'
+import { contentChecks, resetContentChecks } from './content-checks'
+import { deriveTrust } from './trust'
/** Mock SDK serving each pack's bytes as `chunk` docs split at CHUNK_PAYLOAD_MAX. */
function mockSdk(packsByHash: Map): EvoSDK {
@@ -121,3 +124,127 @@ describe('startFallback', () => {
await expect(startFallback(sdk, repo, [manifest])).rejects.toThrow(/header claims/)
})
})
+
+/** A single-blob pack (no deltas) holding `text`. */
+function blobPack(text: string): { pack: Uint8Array; oid: string } {
+ const body = new TextEncoder().encode(text)
+ return { pack: packFrame(concat(objHeader(T_BLOB, body.length), zlibSync(body))), oid: gitOidHex('blob', body) }
+}
+
+/** Route `fetch` by URL: a handler returns bytes, or throws for a dead host. */
+function stubFetch(routes: Record Uint8Array>): string[] {
+ const calls: string[] = []
+ vi.stubGlobal('fetch', (url: string) => {
+ calls.push(url)
+ const route = routes[url]
+ if (route === undefined) return Promise.reject(new TypeError('fetch failed: connection refused'))
+ const bytes = route()
+ return Promise.resolve(new Response(new Blob([bytes as BlobPart]), { status: 200 }))
+ })
+ return calls
+}
+
+describe('startFallback with external-storage packs', () => {
+ afterEach(() => {
+ vi.unstubAllGlobals()
+ resetContentChecks()
+ })
+
+ it('skips an external pack no mirror serves, reports it, and still serves the rest', async () => {
+ const plat = blobPack('on-chain content\n')
+ const ext = blobPack('content only the dead mirror had\n')
+ const platform = manifestFor(plat.pack, 1, { createdAt: 1, documentId: 'a' })
+ const external = manifestFor(ext.pack, 1, {
+ storage: 1,
+ chunkCount: 0,
+ uris: ['http://127.0.0.1:9000/forge-byo/pack', 's3://forge-byo/pack'],
+ createdAt: 2,
+ documentId: 'b',
+ })
+ const repo: RepoRef = { contractId: 'fallback-partial', ownerId: 'owner' }
+ const calls = stubFetch({})
+ const ctx = await startFallback(mockSdk(new Map([[platform.packHash, plat.pack]])), repo, [platform, external])
+
+ expect(calls).toEqual(['http://127.0.0.1:9000/forge-byo/pack']) // s3:// is not browser-fetchable
+ expect(ctx.unavailable).toHaveLength(1)
+ expect(ctx.unavailable?.[0]?.packHash).toBe(external.packHash)
+ expect(ctx.unavailable?.[0]?.hosts).toEqual(['127.0.0.1:9000'])
+ expect(Array.from((await ctx.reader.readObject(plat.oid)).bytes)).toEqual(
+ Array.from(new TextEncoder().encode('on-chain content\n')),
+ )
+ // A needed object from the skipped pack gets a per-view error naming pack and storage.
+ await expect(ctx.reader.readObject(ext.oid)).rejects.toThrow(
+ new RegExp(`${external.packHash.slice(0, 12)}.*127\\.0\\.0\\.1:9000`),
+ )
+ // The trust ledger reports the gap: content is partial, never verified.
+ const checks = contentChecks(repo.contractId)
+ expect(checks.unavailablePacks).toEqual([external.packHash])
+ const trust = deriveTrust({
+ network: 'testnet',
+ connection: 'trusted',
+ tip: 'missing',
+ checks,
+ configuredBackend: 'platform',
+ })
+ expect(trust.content.state).toBe('partial')
+ expect(trust.content.detail).toMatch(/1 pack could not be fetched from its storage; some objects may be missing|could not be fetched/)
+ })
+
+ it('reports a pack named by several manifests once', async () => {
+ const plat = blobPack('kept\n')
+ const ext = blobPack('gone\n')
+ const platform = manifestFor(plat.pack, 1, { createdAt: 1, documentId: 'a' })
+ const external = (documentId: string): PackManifest =>
+ manifestFor(ext.pack, 1, { storage: 1, uris: ['http://127.0.0.1:9000/p'], createdAt: 2, documentId })
+ stubFetch({})
+ const ctx = await startFallback(
+ mockSdk(new Map([[platform.packHash, plat.pack]])),
+ { contractId: 'fallback-dup', ownerId: 'owner' },
+ [platform, external('b'), external('c')],
+ )
+ expect(ctx.unavailable).toHaveLength(1)
+ })
+
+ it('fetches an ipfs:// pack through a gateway and verifies it', async () => {
+ const ext = blobPack('pinned on ipfs\n')
+ const external = manifestFor(ext.pack, 1, { storage: 1, chunkCount: 0, uris: ['ipfs://bafkreitest'] })
+ const repo: RepoRef = { contractId: 'fallback-ipfs', ownerId: 'owner' }
+ const [first, second] = externalFetchUrls(external.uris)
+ expect(first).toBe('https://ipfs.io/ipfs/bafkreitest')
+ // The first gateway is down; the second serves the right bytes.
+ stubFetch({ [second as string]: () => ext.pack })
+ const ctx = await startFallback(mockSdk(new Map()), repo, [external])
+ expect(ctx.unavailable).toEqual([])
+ expect((await ctx.reader.readObject(ext.oid)).type).toBe('blob')
+ expect(contentChecks(repo.contractId).sources).toEqual(['dweb.link'])
+ })
+
+ it('treats a mirror serving the wrong bytes as unavailable, not as content', async () => {
+ const plat = blobPack('real\n')
+ const ext = blobPack('expected\n')
+ const platform = manifestFor(plat.pack, 1, { createdAt: 1, documentId: 'a' })
+ const external = manifestFor(ext.pack, 1, { storage: 1, uris: ['https://mirror.example/p'], createdAt: 2, documentId: 'b' })
+ const repo: RepoRef = { contractId: 'fallback-liar', ownerId: 'owner' }
+ stubFetch({ 'https://mirror.example/p': () => blobPack('forged!!!\n').pack })
+ const ctx = await startFallback(mockSdk(new Map([[platform.packHash, plat.pack]])), repo, [platform, external])
+ expect(ctx.unavailable?.[0]?.reason).toMatch(/sha256/)
+ await expect(ctx.reader.readObject(ext.oid)).rejects.toThrow(/could not be fetched/)
+ })
+
+ it('still fails loudly when an on-chain pack cannot be read', async () => {
+ const plat = blobPack('on-chain\n')
+ const platform = manifestFor(plat.pack, 1)
+ const repo: RepoRef = { contractId: 'fallback-platform-missing', ownerId: 'owner' }
+ // The chunk documents are absent: platform storage must not be skipped.
+ await expect(startFallback(mockSdk(new Map()), repo, [platform])).rejects.toThrow(/missing chunk/)
+ })
+
+ it('fails with the reasons when no live pack at all could be fetched', async () => {
+ const ext = blobPack('x\n')
+ const external = manifestFor(ext.pack, 1, { storage: 1, uris: ['http://127.0.0.1:9000/p'] })
+ stubFetch({})
+ await expect(
+ startFallback(mockSdk(new Map()), { contractId: 'fallback-none', ownerId: 'owner' }, [external]),
+ ).rejects.toThrow(/none of this repo's 1 live packs could be fetched/)
+ })
+})
diff --git a/forge-web/lib/view/browse-fallback.ts b/forge-web/lib/view/browse-fallback.ts
index cda7d034f..ae18433bf 100644
--- a/forge-web/lib/view/browse-fallback.ts
+++ b/forge-web/lib/view/browse-fallback.ts
@@ -3,7 +3,9 @@
* objectLocator.
*
* Downloads the repo's live kind-0 packs whole (each sha256-verified against its
- * consensus-proven `packManifest.packHash`, mirroring `git-remote-dash::fetch`), indexes
+ * consensus-proven `packManifest.packHash`, mirroring `git-remote-dash::fetch`: platform
+ * packs from `chunk` documents, external packs from any of their mirrors or IPFS gateways —
+ * an external pack none serves is skipped and reported, never silently), indexes
* them client-side (`lib/browse/indexer` — dynamically imported so pako stays out of the
* main bundles), and assembles the same {@link BrowseContext} the locator path produces,
* so every downstream view works unchanged.
@@ -21,7 +23,12 @@ import { bytesToHex } from '@noble/hashes/utils.js'
import { BrowseReader, ObjectLocator } from '../browse'
import type { PackManifest, RepoRef } from '../repo'
-import { loadArtifactBytesProgress, type BrowseContext } from './browse-source'
+import {
+ loadArtifactBytesProgress,
+ PackUnavailableError,
+ type BrowseContext,
+ type UnavailablePack,
+} from './browse-source'
import { noteContentCheck, objectObserver } from './content-checks'
import {
deleteStoredFallback,
@@ -162,6 +169,81 @@ function contextFromStored(repo: RepoRef, stored: StoredFallback): Promise `${p.packHash.slice(0, 12)}… (${p.hosts.length > 0 ? p.hosts.join(', ') : 'no fetchable mirror'})`)
+ .join('; ')
+ return new Error(
+ `object ${oidHex.slice(0, 12)}… is not in any pack this browser could load. ` +
+ `${unavailable.length === 1 ? 'One pack' : `${unavailable.length} packs`} could not be fetched from ` +
+ `${unavailable.length === 1 ? 'its' : 'their'} external storage and may hold it: ${where}. ` +
+ 'Cloning with dash:// reads the same packs; if their mirrors are down it will fail the same way.',
+ )
+}
+
+/** A downloaded live pack, or the record of why it could not be. */
+type PackOutcome =
+ | { readonly manifest: PackManifest; readonly bytes: Uint8Array }
+ | { readonly manifest: PackManifest; readonly unavailable: UnavailablePack }
+
+/**
+ * Download every live pack. Platform packs (storage 0) are the repo's own chain data: they
+ * download one after another and any failure fails the clone. External packs race their
+ * mirrors concurrently from the start, so dead mirrors cost one timeout in total, not one per
+ * pack; one no mirror serves authentically is skipped and reported, mirroring the dash://
+ * helper (a clone is not held hostage by one dead mirror, and git's connectivity check —
+ * here, the reader's missing-object error — still fails anything that truly needed it).
+ */
+async function downloadPacks(
+ sdk: EvoSDK,
+ repo: RepoRef,
+ livePacks: readonly PackManifest[],
+ report: (bytesFetched: number) => void,
+): Promise {
+ let fetched = 0
+ const external = new Map(
+ livePacks
+ .filter((m) => m.storage !== 0)
+ .map((m) => [
+ m,
+ loadArtifactBytesProgress(sdk, repo, m).then(
+ (bytes): PackOutcome => {
+ fetched += m.sizeBytes
+ report(fetched)
+ return { manifest: m, bytes }
+ },
+ (e: unknown): PackOutcome => {
+ if (!(e instanceof PackUnavailableError)) throw e
+ return {
+ manifest: m,
+ unavailable: { packHash: m.packHash, hosts: e.hosts, reason: e.message },
+ }
+ },
+ ),
+ ]),
+ )
+ const outcomes: PackOutcome[] = []
+ for (const manifest of livePacks) {
+ const ext = external.get(manifest)
+ if (ext !== undefined) {
+ outcomes.push(await ext)
+ continue
+ }
+ const base = fetched
+ const bytes = await loadArtifactBytesProgress(sdk, repo, manifest, (done) => report(base + done))
+ fetched += manifest.sizeBytes
+ outcomes.push({ manifest, bytes })
+ }
+ return outcomes
+}
+
async function runFallback(
sdk: EvoSDK,
repo: RepoRef,
@@ -179,24 +261,38 @@ async function runFallback(
...p,
})
- const packs: Uint8Array[] = []
- let fetchedBefore = 0
- for (const manifest of livePacks) {
- const bytes = await loadArtifactBytesProgress(sdk, repo, manifest, (done) =>
- report({ phase: 'download', bytesFetched: fetchedBefore + done }),
+ const outcomes = await downloadPacks(sdk, repo, livePacks, (bytesFetched) =>
+ report({ phase: 'download', bytesFetched }),
+ )
+ const got = outcomes.filter((o): o is Extract => 'bytes' in o)
+ // Keyed by packHash: a repo can carry several manifests for one pack (a re-push, a
+ // re-announced mirror), and a pack is missing once however many documents name it.
+ const unavailable = [
+ ...new Map(
+ outcomes.flatMap((o) =>
+ 'unavailable' in o ? [[o.unavailable.packHash.toLowerCase(), o.unavailable] as const] : [],
+ ),
+ ).values(),
+ ]
+ for (const u of unavailable) noteContentCheck(repo.contractId, { unavailablePack: u.packHash })
+ if (got.length === 0) {
+ throw new Error(
+ `none of this repo's ${livePacks.length} live packs could be fetched from their storage: ` +
+ unavailable.map((u) => u.reason).join('; '),
)
- fetchedBefore += manifest.sizeBytes
- packs.push(bytes)
}
+
+ const packs = got.map((o) => o.bytes)
+ const manifests = got.map((o) => o.manifest)
try {
- validatePacks(packs, livePacks)
+ validatePacks(packs, manifests)
} catch (e) {
// A downloaded pack that does not match its proof-read manifest is a content-check
// failure the trust panel must report, not only an error on this page.
noteContentCheck(repo.contractId, { packsFailed: 1 })
throw e
}
- noteContentCheck(repo.contractId, { packsVerified: livePacks.length })
+ noteContentCheck(repo.contractId, { packsVerified: packs.length })
const { indexPacks, serializeLocator, memoryPackSource } = await import('../browse/indexer')
const objects = await indexPacks(packs, (objectsIndexed, objectsTotal) =>
@@ -204,8 +300,17 @@ async function runFallback(
)
const locatorBytes = serializeLocator(objects)
const locator = ObjectLocator.parse(locatorBytes)
+ // The synthesized locator's packRef space is exactly the packs that downloaded.
const packSource = memoryPackSource(packs)
- const reader = new BrowseReader(locator, packSource, { onObject: objectObserver(repo.contractId) })
- await storeFallback(repo.contractId, livePacks, { locator: locatorBytes, packs })
- return { locator, packs: packSource, reader }
+ const reader = new BrowseReader(locator, packSource, {
+ onObject: objectObserver(repo.contractId),
+ missingObject:
+ unavailable.length > 0 ? (oid) => missingObjectError(oid, unavailable) : undefined,
+ })
+ // Only a complete clone is persisted. A skipped pack's mirror may come back, and a reload
+ // is the natural moment to try it again; a persisted partial clone would never retry.
+ if (unavailable.length === 0) {
+ await storeFallback(repo.contractId, livePacks, { locator: locatorBytes, packs })
+ }
+ return { locator, packs: packSource, reader, unavailable }
}
diff --git a/forge-web/lib/view/browse-source.ts b/forge-web/lib/view/browse-source.ts
index 73876471b..f8a1fc255 100644
--- a/forge-web/lib/view/browse-source.ts
+++ b/forge-web/lib/view/browse-source.ts
@@ -15,7 +15,10 @@
import type { EvoSDK } from '@dashevo/evo-sdk'
-import { CHUNK_PAYLOAD_MAX, PACK_KIND } from '../constants'
+import { sha256 } from '@noble/hashes/sha2.js'
+import { bytesToHex } from '@noble/hashes/utils.js'
+
+import { CHUNK_PAYLOAD_MAX, IPFS_GATEWAYS, PACK_KIND } from '../constants'
import {
BrowseReader,
FlatIndex,
@@ -236,30 +239,161 @@ async function fetchPlatformRange(
return out
}
+// ---------------------------------------------------------------------------
+// External (storage 1) artifacts
+// ---------------------------------------------------------------------------
+
/**
- * Fetch a contiguous range of an external artifact via HTTP Range. `onServed` is told which
- * URI answered, so the trust panel can name the host bytes actually came from.
+ * Budget for one request to one external mirror. The mirror may be a dead host (a manifest
+ * written by a push to someone's local MinIO) or a public gateway searching the IPFS network
+ * for a CID nobody pins; neither may hold a browse hostage.
*/
-async function fetchExternalRange(
+const EXTERNAL_FETCH_TIMEOUT_MS = 15_000
+
+/**
+ * An external artifact that no mirror served authentically. Carries what a view needs to
+ * say *which* pack is missing and *where* it was looked for.
+ */
+export class PackUnavailableError extends Error {
+ constructor(
+ readonly packHash: string,
+ /** Hosts actually tried (empty: the manifest records no browser-fetchable mirror). */
+ readonly hosts: readonly string[],
+ /** Whether some mirror answered with bytes that failed the sha256 check. */
+ readonly corrupt: boolean,
+ reason: string,
+ ) {
+ super(
+ `pack ${packHash.slice(0, 12)}… could not be fetched from its storage (${
+ hosts.length > 0 ? hosts.join(', ') : 'no browser-fetchable mirror recorded'
+ }): ${reason}`,
+ )
+ this.name = 'PackUnavailableError'
+ }
+}
+
+/**
+ * The HTTP(S) URLs an external artifact can be fetched from, in order: the manifest's own
+ * `http(s)` mirrors, then each `ipfs://` through every gateway in `gateways`. Schemes a
+ * browser cannot fetch (`s3://`, `platform://`) are skipped — an `s3://` locator always
+ * travels with the bucket's public `https` URL, which is listed separately.
+ */
+export function externalFetchUrls(
uris: readonly string[],
+ gateways: readonly string[] = IPFS_GATEWAYS,
+): string[] {
+ const out: string[] = []
+ for (const uri of uris) {
+ if (/^https?:\/\//i.test(uri)) out.push(uri)
+ const ipfs = /^ipfs:\/\/(.+)$/i.exec(uri)
+ if (ipfs !== null) {
+ for (const gw of gateways) out.push(`${gw.replace(/\/+$/, '')}/ipfs/${ipfs[1]}`)
+ }
+ }
+ return [...new Set(out)]
+}
+
+/**
+ * GET `url` and read the whole body under one hard deadline (a stalled body counts, not only
+ * a stalled connect). A 206 is accepted alongside 2xx.
+ */
+async function fetchBody(url: string, init: RequestInit = {}): Promise {
+ const controller = new AbortController()
+ const timer = setTimeout(() => controller.abort(), EXTERNAL_FETCH_TIMEOUT_MS)
+ try {
+ const resp = await fetch(url, { ...init, signal: controller.signal })
+ if (!resp.ok && resp.status !== 206) throw new Error(`HTTP ${resp.status}`)
+ return new Uint8Array(await resp.arrayBuffer())
+ } catch (e) {
+ throw controller.signal.aborted ? new Error(`no answer in ${EXTERNAL_FETCH_TIMEOUT_MS / 1000}s`) : e
+ } finally {
+ clearTimeout(timer)
+ }
+}
+
+function hostsOf(urls: readonly string[]): string[] {
+ return [...new Set(urls.map(externalSourceName))]
+}
+
+/**
+ * Fetch a contiguous range of an external artifact via HTTP Range, trying each fetchable
+ * mirror in turn. `onServed` is told which URL answered, so the trust panel can name the host
+ * bytes actually came from. A range cannot be hashed on its own: the reader re-hashes every
+ * object it reconstructs from it.
+ */
+async function fetchExternalRange(
+ manifest: PackManifest,
start: number,
end: number,
onServed?: (uri: string) => void,
): Promise {
- let lastErr: unknown
- for (const uri of uris) {
+ const urls = externalFetchUrls(manifest.uris)
+ let lastErr: unknown = 'no browser-fetchable mirror'
+ for (const url of urls) {
try {
- const resp = await fetch(uri, { headers: { Range: `bytes=${start}-${end - 1}` } })
- if (!resp.ok && resp.status !== 206) throw new Error(`HTTP ${resp.status}`)
- const buf = new Uint8Array(await resp.arrayBuffer())
- onServed?.(uri)
+ const buf = await fetchBody(url, { headers: { Range: `bytes=${start}-${end - 1}` } })
+ onServed?.(url)
// Some hosts ignore Range and return the whole body — slice defensively.
return buf.length > end - start ? buf.subarray(start, end) : buf
} catch (e) {
lastErr = e
}
}
- throw new Error(`no external URI served the range: ${String(lastErr)}`)
+ throw new PackUnavailableError(manifest.packHash, hostsOf(urls), false, errorText(lastErr))
+}
+
+function errorText(e: unknown): string {
+ return e instanceof Error ? e.message : String(e)
+}
+
+/**
+ * Fetch a whole external artifact, racing every fetchable mirror; the first body whose size
+ * and sha256 match the proof-read manifest wins. Mirrors are availability, never authority:
+ * one answering with other bytes is treated as down (and flagged `corrupt`).
+ */
+async function fetchExternalWhole(
+ manifest: PackManifest,
+ onServed?: (uri: string) => void,
+): Promise {
+ const urls = externalFetchUrls(manifest.uris)
+ const want = manifest.packHash.toLowerCase()
+ if (urls.length === 0) {
+ throw new PackUnavailableError(manifest.packHash, [], false, 'nothing to try')
+ }
+ let corrupt = false
+ const reasons: string[] = []
+ const attempt = async (url: string): Promise<{ url: string; bytes: Uint8Array }> => {
+ try {
+ const bytes = await fetchBody(url)
+ if (bytes.length !== manifest.sizeBytes || bytesToHex(sha256(bytes)) !== want) {
+ corrupt = true
+ throw new Error('served bytes that do not match the manifest sha256')
+ }
+ return { url, bytes }
+ } catch (e) {
+ reasons.push(`${externalSourceName(url)}: ${errorText(e)}`)
+ throw e
+ }
+ }
+ try {
+ const { url, bytes } = await firstFulfilled(urls.map(attempt))
+ onServed?.(url)
+ return bytes
+ } catch {
+ throw new PackUnavailableError(manifest.packHash, hostsOf(urls), corrupt, reasons.join('; '))
+ }
+}
+
+/** The first promise to fulfil; rejects only once every one has rejected. */
+function firstFulfilled(promises: readonly Promise[]): Promise {
+ return new Promise((resolve, reject) => {
+ let pending = promises.length
+ for (const p of promises) {
+ p.then(resolve, () => {
+ if (--pending === 0) reject(new Error('every attempt failed'))
+ })
+ }
+ })
}
/** Record in the repo's content-check ledger where an artifact's bytes came from. */
@@ -277,7 +411,7 @@ export function artifactRangeFetch(
): RangeFetch {
return async (start: number, end: number) => {
if (manifest.storage !== 0) {
- return fetchExternalRange(manifest.uris, start, end, (uri) => noteSource(repo, uri))
+ return fetchExternalRange(manifest, start, end, (uri) => noteSource(repo, uri))
}
const bytes = await fetchPlatformRange(sdk, repo.contractId, manifest.packHash, start, end)
noteSource(repo)
@@ -311,8 +445,9 @@ const DOWNLOAD_WINDOW = CHUNK_QUERY_MAX * CHUNK_PAYLOAD_MAX
/**
* Load a whole artifact with download progress — the fallback-clone path for full git
* packs, which can exceed the single-query chunk window. Platform storage downloads in
- * `DOWNLOAD_WINDOW` strides; external storage fetches the body whole (progress reported
- * only at completion).
+ * `DOWNLOAD_WINDOW` strides; external storage races its mirrors for the whole body, which
+ * must match the manifest's size and sha256 (progress reported only at completion), and
+ * throws {@link PackUnavailableError} when none does.
*/
export async function loadArtifactBytesProgress(
sdk: EvoSDK,
@@ -324,8 +459,7 @@ export async function loadArtifactBytesProgress(
if (total <= 0) return new Uint8Array(0)
onProgress?.(0, total)
if (manifest.storage !== 0) {
- const bytes = await fetchExternalRange(manifest.uris, 0, total, (uri) => noteSource(repo, uri))
- if (bytes.length !== total) throw new Error('external artifact length mismatch')
+ const bytes = await fetchExternalWhole(manifest, (uri) => noteSource(repo, uri))
onProgress?.(total, total)
return bytes
}
@@ -424,11 +558,24 @@ export function buildPackSource(
}
}
+/** A live pack the browser could not obtain from its external storage. */
+export interface UnavailablePack {
+ readonly packHash: string
+ /** Hosts tried (empty: the manifest records no browser-fetchable mirror). */
+ readonly hosts: readonly string[]
+ readonly reason: string
+}
+
/** The assembled browse context for a repo, or a reason it is unavailable. */
export interface BrowseContext {
readonly locator: ObjectLocator
readonly packs: PackSource
readonly reader: BrowseReader
+ /**
+ * External packs the in-browser clone skipped because no mirror served them. Objects only
+ * those packs hold are absent from this context; empty/absent means nothing was skipped.
+ */
+ readonly unavailable?: readonly UnavailablePack[]
}
/**
diff --git a/forge-web/lib/view/content-checks.test.ts b/forge-web/lib/view/content-checks.test.ts
index bc0b7c945..cbe53533c 100644
--- a/forge-web/lib/view/content-checks.test.ts
+++ b/forge-web/lib/view/content-checks.test.ts
@@ -58,6 +58,14 @@ describe('content-check ledger', () => {
unsubscribe()
})
+ it('records each unavailable pack once, case-insensitively', () => {
+ noteContentCheck('repo-a', { unavailablePack: 'AB'.repeat(32) })
+ const snap = contentChecks('repo-a')
+ noteContentCheck('repo-a', { unavailablePack: 'ab'.repeat(32) })
+ expect(contentChecks('repo-a')).toBe(snap)
+ expect(snap.unavailablePacks).toEqual(['ab'.repeat(32)])
+ })
+
it('names an external source by its host', () => {
expect(externalSourceName('https://ipfs.io/ipfs/bafy')).toBe('ipfs.io')
expect(externalSourceName('https://bucket.s3.amazonaws.com/p.pack')).toBe('bucket.s3.amazonaws.com')
diff --git a/forge-web/lib/view/content-checks.ts b/forge-web/lib/view/content-checks.ts
index 43fb8367f..d6bca5181 100644
--- a/forge-web/lib/view/content-checks.ts
+++ b/forge-web/lib/view/content-checks.ts
@@ -29,6 +29,12 @@ export interface ContentChecks {
readonly packsFailed: number
/** Where bytes actually came from: `platform`, `browser cache`, or an external host. */
readonly sources: readonly string[]
+ /**
+ * Live external packs (hash hex) the in-browser clone skipped because no mirror served
+ * them. What is shown was checked, but it is not the whole repo: objects only those packs
+ * hold are missing.
+ */
+ readonly unavailablePacks: readonly string[]
}
export const NO_CONTENT_CHECKS: ContentChecks = {
@@ -38,12 +44,19 @@ export const NO_CONTENT_CHECKS: ContentChecks = {
packsVerified: 0,
packsFailed: 0,
sources: [],
+ unavailablePacks: [],
}
-type Counter = Exclude
+type Counter = Exclude
-/** A change to one repo's ledger: counter increments and/or a byte source seen. */
-export type ContentCheckDelta = Partial> & { readonly source?: string }
+/**
+ * A change to one repo's ledger: counter increments, a byte source seen, and/or a live pack
+ * that could not be fetched (hash hex).
+ */
+export type ContentCheckDelta = Partial> & {
+ readonly source?: string
+ readonly unavailablePack?: string
+}
const ledger = new Map()
const listeners = new Set<() => void>()
@@ -60,11 +73,14 @@ export function noteContentCheck(contractId: string, delta: ContentCheckDelta):
'packsFailed',
]
const bumped = counters.some((k) => (delta[k] ?? 0) > 0)
- if (!newSource && !bumped) return
+ const missing = delta.unavailablePack?.toLowerCase()
+ const newMissing = missing !== undefined && !prev.unavailablePacks.includes(missing)
+ if (!newSource && !bumped && !newMissing) return
const next: { -readonly [K in keyof ContentChecks]: ContentChecks[K] } = { ...prev }
for (const k of counters) next[k] = prev[k] + Math.max(0, delta[k] ?? 0)
if (newSource && delta.source !== undefined) next.sources = [...prev.sources, delta.source]
+ if (newMissing) next.unavailablePacks = [...prev.unavailablePacks, missing]
ledger.set(contractId, next)
for (const l of listeners) l()
}
diff --git a/forge-web/lib/view/index.ts b/forge-web/lib/view/index.ts
index b15774781..a64e11d49 100644
--- a/forge-web/lib/view/index.ts
+++ b/forge-web/lib/view/index.ts
@@ -67,8 +67,10 @@ export {
loadFlatIndex,
orderGitPacks,
peekBrowseState,
+ PackUnavailableError,
type BrowseContext,
type BrowseState,
+ type UnavailablePack,
} from './browse-source'
export {
cachedFallback,
diff --git a/forge-web/lib/view/trust.test.ts b/forge-web/lib/view/trust.test.ts
index d6401b65f..74e59c032 100644
--- a/forge-web/lib/view/trust.test.ts
+++ b/forge-web/lib/view/trust.test.ts
@@ -123,6 +123,25 @@ describe('deriveTrust — content hashes', () => {
expect(pack.content.detail).toMatch(/1 pack did not match its manifest/)
})
+ it('is partial, never verified, when a live pack could not be fetched', () => {
+ const read = deriveTrust(
+ inputs({ checks: checks({ objectsVerified: 5, packsVerified: 3, unavailablePacks: ['ab'.repeat(32)] }) }),
+ )
+ expect(read.content.state).toBe('partial')
+ expect(read.content.summary).toBe('1 pack missing')
+ expect(read.content.detail).toMatch(/1 pack could not be fetched from its storage, so some objects may be missing/)
+ expect(read.overall).toBe('partial')
+
+ // Known before any object is read, so not `pending` either.
+ const none = deriveTrust(inputs({ checks: checks({ unavailablePacks: ['ab'.repeat(32), 'cd'.repeat(32)] }) }))
+ expect(none.content.state).toBe('partial')
+ expect(none.content.summary).toBe('2 packs missing')
+
+ // A hash failure still wins.
+ const bad = deriveTrust(inputs({ checks: checks({ objectsFailed: 1, unavailablePacks: ['ab'.repeat(32)] }) }))
+ expect(bad.content.state).toBe('failed')
+ })
+
it('names every source bytes actually came from', () => {
const r = deriveTrust(inputs({ checks: checks({ objectsVerified: 1, sources: ['platform', 'ipfs.io'] }) }))
expect(r.source.summary).toBe('2 sources')
diff --git a/forge-web/lib/view/trust.ts b/forge-web/lib/view/trust.ts
index 7f7d2ddf6..6228d3fc4 100644
--- a/forge-web/lib/view/trust.ts
+++ b/forge-web/lib/view/trust.ts
@@ -170,6 +170,22 @@ function deriveRefs(input: TrustInputs): TrustLink {
}
function deriveContent(checks: ContentChecks): TrustLink {
+ const link = deriveReadContent(checks)
+ const missing = checks.unavailablePacks.length
+ if (missing === 0 || link.state === 'failed') return link
+ // Everything shown passed its check, but the answer is incomplete: objects only the
+ // skipped packs hold cannot be shown at all. That is at best `partial`, never `verified`
+ // — and never `pending` either, since the skip is known before any object is read.
+ const note = `${plural(missing, 'pack')} could not be fetched from ${missing === 1 ? 'its' : 'their'} storage, so some objects may be missing.`
+ return {
+ state: link.state === 'unverified' ? 'unverified' : 'partial',
+ summary: `${missing} ${missing === 1 ? 'pack' : 'packs'} missing`,
+ detail: `${note} ${link.detail}`,
+ }
+}
+
+/** The content link from what was read, before accounting for packs that were skipped. */
+function deriveReadContent(checks: ContentChecks): TrustLink {
const failed = checks.objectsFailed + checks.packsFailed
if (failed > 0) {
const parts: string[] = []
From 57cfb0d928955dc2c9715a95cee653380563c345 Mon Sep 17 00:00:00 2001
From: pasta
Date: Fri, 25 Sep 2026 08:55:02 -0500
Subject: [PATCH 03/11] fix(web): give white-text status badges AA fills and
test every one
The Merged badge put white on the brand blue (3.54:1) and the Open badges white on verify green (3.3:1). They now use dash-700 (7.27:1) and verify-700 (5.02:1). The contrast test reads every solid bg-* behind text-white in app/ and components/, including computed badge backgrounds, and fails below 4.5:1.
Co-Authored-By: Claude Opus 5.5 (1M context)
---
docs/design/style-guide.md | 9 ++-
forge-web/components/repo/issue-content.tsx | 2 +-
forge-web/components/repo/pull-content.tsx | 4 +-
forge-web/lib/design/contrast.test.ts | 77 ++++++++++++++++++++-
forge-web/tailwind.config.js | 6 +-
5 files changed, 90 insertions(+), 8 deletions(-)
diff --git a/docs/design/style-guide.md b/docs/design/style-guide.md
index 1f9d0e87f..1ada714f0 100644
--- a/docs/design/style-guide.md
+++ b/docs/design/style-guide.md
@@ -19,17 +19,22 @@ colors: {
50:'#fafaf9',100:'#f5f5f4',200:'#e7e5e4',300:'#d6d3d1',400:'#a8a29e',
500:'#78716c',600:'#57534e',700:'#44403c',750:'#3a3835',800:'#292524',850:'#211e1c',900:'#1c1917',950:'#0f0d0c'
},
- verify: '#16a34a', // proof/hash verified
+ verify: { // proof/hash verified
+ DEFAULT:'#16a34a', // icons, borders, tints
+ 700:'#15803d' // solid fill behind white text
+ },
caution:'#d97706', // degraded availability
danger: '#dc2626', // force-push, delete, failed verification
dash: { // Dash brand blue — reserved for identity/credits/network UI only
DEFAULT:'#008de4', // fills, tints, icons
400:'#4aaef0', // TEXT on dark surfaces
- 600:'#006bb0' // TEXT on light surfaces
+ 600:'#006bb0', // TEXT on light surfaces
+ 700:'#005a94' // solid fill behind white text
}
}
```
- **Dash-blue text is `text-dash-600 dark:text-dash-400`**, never plain `text-dash`: the brand value is under WCAG AA's 4.5:1 as text on every surface in both themes (4.28:1 on `anvil-800`). `forge-web/lib/design/contrast.test.ts` pins the ratios and fails on a raw `text-dash` that is not an icon.
+- **White text sits on `-700` fills**: `bg-dash-700`, `bg-verify-700`, `bg-forge-700`. The base `dash` (3.54:1) and `verify` (3.3:1) values fail AA behind white; the same test fails any `text-white` class string whose background is one of them.
- **Dark mode is the primary theme** (class-based, `next-themes`); light mode fully supported. Backgrounds: `anvil-950/900/850` layered surfaces (dark), `anvil-50/white` (light).
- Semantic colors are *meaningful*, never decorative: green = cryptographically verified, amber = availability risk, red = destructive/unverified, dash-blue = platform identity & credits. Don't repurpose.
diff --git a/forge-web/components/repo/issue-content.tsx b/forge-web/components/repo/issue-content.tsx
index 756ade7e7..e6109f8a1 100644
--- a/forge-web/components/repo/issue-content.tsx
+++ b/forge-web/components/repo/issue-content.tsx
@@ -107,7 +107,7 @@ export function IssueContent({
{issue.title || '(untitled)'} #{issue.number}
-
+
{open ? : }
{open ? 'Open' : 'Closed'}
diff --git a/forge-web/components/repo/pull-content.tsx b/forge-web/components/repo/pull-content.tsx
index 8cf2e3efc..15fb41e8b 100644
--- a/forge-web/components/repo/pull-content.tsx
+++ b/forge-web/components/repo/pull-content.tsx
@@ -81,10 +81,10 @@ export function PullContent({ home, addr, number }: { home: RepoHome; addr: Repo
const base = pull.baseRefName || 'the base branch'
const status = merged
- ? { label: 'Merged', icon: , bg: 'bg-dash' }
+ ? { label: 'Merged', icon: , bg: 'bg-dash-700' }
: !open
? { label: 'Closed', icon: , bg: 'bg-danger' }
- : { label: pull.state.draft ? 'Draft' : 'Open', icon: , bg: pull.state.draft ? 'bg-anvil-500' : 'bg-verify' }
+ : { label: pull.state.draft ? 'Draft' : 'Open', icon: , bg: pull.state.draft ? 'bg-anvil-500' : 'bg-verify-700' }
const postComment = async (): Promise => {
if (!identity || !signer) {
diff --git a/forge-web/lib/design/contrast.test.ts b/forge-web/lib/design/contrast.test.ts
index 0c0240589..8fc9b3a7b 100644
--- a/forge-web/lib/design/contrast.test.ts
+++ b/forge-web/lib/design/contrast.test.ts
@@ -14,11 +14,23 @@ import tailwindConfig from '@/tailwind.config.js'
// The config's JSDoc type makes every theme key optional and possibly a function; this
// config is a plain object literal, so read it as one.
-const { anvil, dash } = (
+const colors = (
tailwindConfig as unknown as {
- theme: { extend: { colors: Record<'anvil' | 'dash', Record> } }
+ theme: { extend: { colors: Record> } }
}
).theme.extend.colors
+const anvil = colors['anvil'] as Record
+const dash = colors['dash'] as Record
+
+/** A Tailwind color class suffix (`dash-700`, `verify`, `anvil-500`) to its hex, if known. */
+function tokenHex(token: string): string | undefined {
+ if (token === 'white') return '#ffffff'
+ const m = /^([a-z]+)(?:-(\d+))?$/.exec(token)
+ if (m === null) return undefined
+ const entry = colors[m[1] as string]
+ if (typeof entry === 'string') return m[2] === undefined ? entry : undefined
+ return entry?.[m[2] ?? 'DEFAULT']
+}
type Rgb = [number, number, number]
@@ -121,3 +133,64 @@ describe('no component renders text in the raw brand blue', () => {
expect(flagged('{n}')).toBe(false)
})
})
+
+describe('white text on solid fills meets WCAG AA', () => {
+ const root = resolve(__dirname, '../..')
+
+ function sources(dir: string): string[] {
+ return readdirSync(dir, { withFileTypes: true }).flatMap((e) => {
+ const p = join(dir, e.name)
+ if (e.isDirectory()) return sources(p)
+ return /\.tsx$/.test(e.name) ? [p] : []
+ })
+ }
+
+ // A solid `bg-` class (not a `/10` tint, not a `hover:` / `dark:` variant).
+ const SOLID_BG = /(? {
+ const isBgValue = computed && /\bbg:\s*/.test(l)
+ if (!l.includes('text-white') && !isBgValue) return
+ for (const m of l.matchAll(SOLID_BG)) out.push({ token: m[1] as string, line: i + 1 })
+ })
+ return out
+ }
+
+ it('checks the Merged / Open / Closed / Draft badges and every other white-text fill', () => {
+ const failures: string[] = []
+ let checked = 0
+ for (const file of ['app', 'components'].flatMap((d) => sources(join(root, d)))) {
+ for (const { token, line } of whiteTextBackgrounds(readFileSync(file, 'utf8'))) {
+ const hex = tokenHex(token)
+ if (hex === undefined) continue
+ checked += 1
+ const ratio = contrast(rgb('#ffffff'), rgb(hex))
+ if (ratio < AA_TEXT) {
+ failures.push(`${file.slice(root.length + 1)}:${line} bg-${token} ${ratio.toFixed(2)}:1`)
+ }
+ }
+ }
+ expect(checked).toBeGreaterThanOrEqual(6)
+ expect(failures).toEqual([])
+ })
+
+ it('catches the regressions it exists for', () => {
+ const tokens = (src: string): string[] => whiteTextBackgrounds(src).map((b) => b.token)
+ // The Merged badge before the fix: brand blue behind white is 3.54:1.
+ const merged = "const s = { bg: 'bg-dash' }\n"
+ expect(tokens(merged)).toEqual(['dash'])
+ expect(contrast(rgb('#ffffff'), rgb(tokenHex('dash') as string))).toBeLessThan(AA_TEXT)
+ expect(contrast(rgb('#ffffff'), rgb(tokenHex('verify') as string))).toBeLessThan(AA_TEXT)
+ expect(tokens('')).toEqual(['forge-700'])
+ expect(tokens('')).toEqual([])
+ })
+})
diff --git a/forge-web/tailwind.config.js b/forge-web/tailwind.config.js
index fb1c88e1b..ec8669404 100644
--- a/forge-web/tailwind.config.js
+++ b/forge-web/tailwind.config.js
@@ -43,7 +43,10 @@ module.exports = {
950: '#0f0d0c',
},
// Semantic colors are MEANINGFUL, never decorative. Do not repurpose.
- verify: '#16a34a', // proof/hash verified
+ verify: {
+ DEFAULT: '#16a34a', // proof/hash verified — icons, borders, tints
+ 700: '#15803d', // solid fill behind white text (5.02:1; the base value is 3.3:1)
+ },
caution: '#d97706', // degraded availability
danger: '#dc2626', // force-push, delete, failed verification
// Dash brand blue — identity/credits/network UI only. The brand value is for fills,
@@ -54,6 +57,7 @@ module.exports = {
DEFAULT: '#008de4',
400: '#4aaef0', // text on dark surfaces (anvil-950…800)
600: '#006bb0', // text on light surfaces (white, anvil-50/100)
+ 700: '#005a94', // solid fill behind white text (7.27:1; the brand value is 3.54:1)
},
},
fontFamily: {
From 6738d672a56e15855fcda4cb3ba16521dcc9b707 Mon Sep 17 00:00:00 2001
From: pasta
Date: Fri, 25 Sep 2026 08:55:02 -0500
Subject: [PATCH 04/11] test(e2e): give the browser specs a read fixture no
other suite writes
The Playwright specs read m1-75299, the repo the CLI suite pushes to and that an ad-hoc storage run also used, so a pack stored on someone's local MinIO broke every browse spec. They now read m1-5124, written only by e2e/cli/seed-read-fixture.sh: one deterministic Platform-stored commit on main (README.md, src/, lib/), no browse index (DASH_FORGE_NO_BROWSE_INDEX) so the fallback clone is exercised, force-pushed only when main differs. The nightly seeds it in its own job before Playwright runs; the CLI suite's repo is configurable via E2E_REPO_NAME. e2e/README.md reserves each repo per suite, including storage-e2e-a/b for the BYO-storage script.
Co-Authored-By: Claude Opus 5.5 (1M context)
---
.github/workflows/testnet-nightly.yml | 55 +++++++++++++++++-
Makefile | 8 ++-
crates/git-remote-dash/src/helper.rs | 10 ++++
e2e/README.md | 33 ++++++++---
e2e/cli/config.sh | 19 ++++--
e2e/cli/run.sh | 2 +-
e2e/cli/seed-read-fixture.sh | 84 +++++++++++++++++++++++++++
forge-web/e2e/helpers.ts | 14 +++--
8 files changed, 202 insertions(+), 23 deletions(-)
create mode 100644 e2e/cli/seed-read-fixture.sh
diff --git a/.github/workflows/testnet-nightly.yml b/.github/workflows/testnet-nightly.yml
index 0412d8184..761fc8d84 100644
--- a/.github/workflows/testnet-nightly.yml
+++ b/.github/workflows/testnet-nightly.yml
@@ -8,15 +8,17 @@ name: Testnet Nightly
# frozen push, no-token push, third-party verify, depth+filter
#
# They split by what they need:
-# * READ-ONLY specs need only a network path to testnet DAPI and the seeded fixture repo,
-# so they run unconditionally and their failure fails this workflow.
+# * READ-ONLY specs need only a network path to testnet DAPI and the read fixture repo
+# (seeded by `seed-read-fixture` when the secrets exist; e2e/README.md reserves it), so
+# they run unconditionally and their failure fails this workflow.
# * The CLI suite and the write specs need FUNDED fixture identities, which cannot live in
# a fork's PR context. A separate `fixtures` probe job decides whether they run, so when
# the secrets are absent the suite job's conclusion is literally `skipped` rather than a
# green job with everything gated out — a distinction any badge or alert rule can see.
#
# Fixture identity secrets (e2e/cli/config.sh names the roles):
-# FORGE_TEST_IDENTITY_DEPLOYER - owns the reused m1 repo, grants tokens
+# FORGE_TEST_IDENTITY_DEPLOYER - owns the CLI suite's repo and the read fixture, grants
+# tokens (which repo each suite may write: e2e/README.md)
# FORGE_TEST_IDENTITY_COLLAB - holds an unfrozen WRITE token
# FORGE_TEST_IDENTITY_CONTRIB - holds no token (negative push case)
# FORGE_TEST_IDENTITY_FROZEN - holds a frozen WRITE token (negative push case)
@@ -35,6 +37,11 @@ concurrency:
jobs:
web-read-paths:
name: playwright (read-only, live testnet)
+ # Reads the dedicated read fixture (e2e/README.md), so it waits for the seed job — but
+ # still runs when that job is skipped (no secrets: the fixture is read as it stands) or
+ # failed (the specs then say what is wrong with it).
+ needs: seed-read-fixture
+ if: ${{ !cancelled() }}
runs-on: ubuntu-latest
timeout-minutes: 30
defaults:
@@ -108,6 +115,46 @@ jobs:
echo "present=true" >> "$GITHUB_OUTPUT"
fi
+ # Make the browser specs' read fixture exist and hold exactly the expected commit. It is
+ # its own repo, written only by this job, so nothing the CLI suite or an ad-hoc storage run
+ # stores can reach what the browser reads (e2e/README.md). Idempotent: a no-op once seeded.
+ seed-read-fixture:
+ name: seed the read fixture
+ needs: fixtures
+ if: needs.fixtures.outputs.present == 'true'
+ runs-on: ubuntu-latest
+ timeout-minutes: 45
+ env:
+ ID_DEPLOYER_JSON: ${{ secrets.FORGE_TEST_IDENTITY_DEPLOYER }}
+ steps:
+ - uses: actions/checkout@v5
+ - name: Install protoc
+ env:
+ PROTOC_VERSION: '28.3'
+ run: |
+ curl -sSLo /tmp/protoc.zip \
+ "https://github.com/protocolbuffers/protobuf/releases/download/v${PROTOC_VERSION}/protoc-${PROTOC_VERSION}-linux-x86_64.zip"
+ sudo unzip -q -o /tmp/protoc.zip -d /usr/local bin/protoc 'include/*'
+ sudo chmod +x /usr/local/bin/protoc
+ - uses: dtolnay/rust-toolchain@stable
+ - uses: Swatinem/rust-cache@v2
+ with:
+ shared-key: e2e-cli
+ - name: Materialize the DEPLOYER identity
+ run: |
+ dir="${HOME}/.config/dash-forge/test-identities"
+ mkdir -p "$dir" && chmod 700 "$dir"
+ printf '%s' "$ID_DEPLOYER_JSON" > "$dir/DEPLOYER.identity.json"
+ # harness_init checks for the whole role set; only DEPLOYER signs here.
+ for r in COLLAB CONTRIB; do cp "$dir/DEPLOYER.identity.json" "$dir/$r.identity.json"; done
+ chmod 600 "$dir"/*.identity.json
+ - name: Build CLI binaries
+ timeout-minutes: 30
+ run: cargo build --locked -p dg -p git-remote-dash
+ - name: Seed
+ timeout-minutes: 12
+ run: bash e2e/cli/seed-read-fixture.sh
+
cli-suite:
name: cli e2e (live testnet)
needs: fixtures
@@ -140,6 +187,8 @@ jobs:
- uses: dtolnay/rust-toolchain@stable
- uses: Swatinem/rust-cache@v2
+ with:
+ shared-key: e2e-cli
- name: Materialize fixture identities
run: |
diff --git a/Makefile b/Makefile
index 93e5b994a..e5ee50712 100644
--- a/Makefile
+++ b/Makefile
@@ -3,7 +3,7 @@ SHELL := /bin/bash
COMPOSE_FILE := infra/docker-compose.yml
-.PHONY: check check-rust check-web build build-rust build-web infra-up infra-down e2e devnet-identities devnet-identities-verify storage-it storage-e2e
+.PHONY: check check-rust check-web build build-rust build-web infra-up infra-down e2e e2e-fixture devnet-identities devnet-identities-verify storage-it storage-e2e
## check: run rust + web lint/test suites; tolerant of dirs that don't exist yet
check: check-rust check-web
@@ -54,13 +54,17 @@ infra-up:
infra-down:
docker compose -f $(COMPOSE_FILE) down -v
-## e2e: run the CLI end-to-end suite (LIVE testnet) against the reused m1 repo.
+## e2e: run the CLI end-to-end suite (LIVE testnet) against its reserved repo (e2e/README.md).
## Builds the binaries if needed, then drives real git push/clone through the
## dash:// helper. See e2e/cli/README-less run.sh header for env knobs
## (RUN_ID, E2E_TIMEOUT, E2E_NO_CLEANUP, subset args). Exits non-zero on any FAIL.
e2e: build-rust
@bash e2e/cli/run.sh
+## e2e-fixture: seed the browser specs' read fixture (idempotent; see e2e/README.md).
+e2e-fixture: build-rust
+ @bash e2e/cli/seed-read-fixture.sh
+
## devnet-identities: mint (or resume) the 9-role identity pool on a devnet,
## funded from the devnet's faucet wallet key, then verify every identity on
## Platform. The key is read from dash-network-configs at runtime (process
diff --git a/crates/git-remote-dash/src/helper.rs b/crates/git-remote-dash/src/helper.rs
index 8dbc39371..273bcdc14 100644
--- a/crates/git-remote-dash/src/helper.rs
+++ b/crates/git-remote-dash/src/helper.rs
@@ -923,6 +923,16 @@ async fn publish_browse_index(
replication: &Replication,
externals: &[ExternalTarget],
) {
+ // `DASH_FORGE_NO_BROWSE_INDEX=1` skips it on purpose: the nightly's read fixture
+ // (e2e/cli/seed-read-fixture.sh) must stay unindexed so the web app's fallback clone is
+ // what the browser specs exercise.
+ if matches!(
+ std::env::var("DASH_FORGE_NO_BROWSE_INDEX").as_deref(),
+ Ok("1" | "true")
+ ) {
+ tracing::info!("DASH_FORGE_NO_BROWSE_INDEX set; not publishing a browse-index fragment");
+ return;
+ }
let chain = replication.has_platform().then(|| {
PlatformChunkTarget::new(ctx.svc, ctx.repo, forge_core::storage::PLATFORM_PROFILE)
});
diff --git a/e2e/README.md b/e2e/README.md
index c60cd9f37..a252487e5 100644
--- a/e2e/README.md
+++ b/e2e/README.md
@@ -1,17 +1,32 @@
# End-to-end suites
-- `cli/`: the CLI suite, run against **live testnet** (`make e2e`). Configuration is in `cli/config.sh`.
+- `cli/run.sh`: the CLI suite, run against **live testnet** (`make e2e`). Its configuration is in `cli/config.sh`.
+- `cli/seed-read-fixture.sh`: seeds the read fixture that the browser specs read (`make e2e-fixture`). The nightly runs it before Playwright.
+- `../forge-web/e2e/`: the Playwright specs. The read specs only read the read fixture.
- `cli/storage-byo.sh`: bring-your-own storage (`make storage-e2e`). A real `git push` / `git clone` whose packs go to the local MinIO + kubo from `infra/docker-compose.yml`.
## Reserved fixture repos (testnet)
-Each fixture repo belongs to one suite. Don't push to a repo that another suite owns.
-Packs stored on a repo's local-only storage (MinIO/kubo on `127.0.0.1`) can't be read by anyone else, so a repo shared with other suites would make their clones depend on your laptop.
+Each fixture repo belongs to one suite. Don't push to a repo that another suite owns, and don't run ad-hoc experiments on any of them.
-| Repo (owner = DEPLOYER `8hJmcHWT…`) | Used by | Notes |
-|---|---|---|
-| `m1-75299` | `cli/scenarios/*` (`make e2e`) | The shared M1 repo; Platform-stored packs only. |
-| `storage-e2e-a` | `cli/storage-byo.sh` steps 1–4 | Created by the script on its first run (repo-v1, about 1.18 tDASH once). Each run uses a fresh `e2e//byo` branch and deletes it afterwards. |
-| `storage-e2e-b` | `cli/storage-byo.sh` steps 5–6 | Same, plus the step-5 "copy deleted" scenario. Step 6 restores that copy, so the repo stays clonable. |
+Why this matters: a pack is only as readable as the storage its manifest names. Packs stored on local-only storage (MinIO/kubo on `127.0.0.1`) can't be read by anyone else. On 2026-09-25 a BYO-storage run pushed such packs into the CLI suite's repo, and every browser spec that browsed that repo failed with "no external URI served the range". A repo shared between suites turns one suite's storage choices into another suite's failures.
-Override the names with `STORAGE_E2E_REPO` / `STORAGE_E2E_REPO_B`.
+All of these repos are owned by DEPLOYER (`8hJmcHWTsdvkHyCrk4UgjbyugDAmE7QfuCTQXpXAc7nB`).
+
+| Repo | Written by | Read by | Notes |
+|---|---|---|---|
+| `m1-5124` | `cli/seed-read-fixture.sh` only | `forge-web/e2e/*` (read-paths, fallback-browse, zero-backend, a11y) | The read fixture. `main` holds one deterministic commit (`README.md`, `src/`, `lib/`) with its pack stored on Platform and no browse index published, so the browser takes the in-browser fallback clone. The seeder is idempotent: it force-pushes `main` only when `main` is not at that commit. Override the name with `NIGHTLY_FIXTURE_REPO` (seeder) and `E2E_FIXTURE_NAME` (Playwright). |
+| `m1-75299` | `cli/scenarios/*` (`make e2e`) | the same scenarios | The CLI suite's repo. It holds Platform-stored packs only. Each run pushes fresh `e2e//…` refs and deletes them afterwards. Override with `E2E_REPO_NAME`. It still holds seven stale external packs (MinIO/IPFS on `127.0.0.1`) from the 2026-09-25 incident; clones skip them with a warning. |
+| `storage-e2e-a` | `cli/storage-byo.sh` steps 1–4 | the same script | Created by the script on its first run (repo-v1, about 1.18 tDASH once). Each run uses a fresh `e2e//byo` branch and deletes it afterwards. |
+| `storage-e2e-b` | `cli/storage-byo.sh` steps 5–6 | the same script | Same, plus the step-5 "copy deleted" scenario. Step 6 restores that copy, so the repo stays clonable. |
+
+Override the storage repo names with `STORAGE_E2E_REPO` / `STORAGE_E2E_REPO_B`. When you add a suite that writes, give it its own repo and add a row here. `dg repo create ` costs about 1.18 tDASH once.
+
+## The partial-clone rule both clients follow
+
+A live pack whose storage cannot be reached does not fail a whole clone or browse:
+
+- `git-remote-dash` skips an external pack whose mirrors are down, and git's connectivity check then fails the fetch only if a wanted object was in it.
+- The web app's in-browser clone tries every recorded URI and the shared IPFS gateway list (`forge-contracts/config/storage-defaults.json`). If none of them serves the pack's exact sha256, it skips that pack. It then shows "N packs could not be fetched from their storage; some objects may be missing", marks the trust panel's content link `partial`, and names the missing pack and its hosts in any view that needs an object from it.
+
+Packs stored on Platform are never skipped: if one can't be read, the clone fails loudly.
diff --git a/e2e/cli/config.sh b/e2e/cli/config.sh
index b998bf458..3159ac22a 100644
--- a/e2e/cli/config.sh
+++ b/e2e/cli/config.sh
@@ -12,11 +12,14 @@
: "${DASH_FORGE_NETWORK:=testnet}"
export DASH_FORGE_NETWORK
-# --- the reused test repo ----------------------------------------------------
-# DEPLOYER is the owner + token granter for this repo.
+# --- the CLI suite's repo ----------------------------------------------------
+# DEPLOYER is the owner + token granter. Reserved for `run.sh` (e2e/README.md): every
+# scenario pushes Platform-stored packs to fresh `e2e//…` refs and deletes them.
+# An ad-hoc run that stores packs anywhere else (local MinIO/kubo) must use its own repo —
+# set E2E_REPO_NAME — so the nightly's clones never depend on someone's laptop.
export E2E_OWNER_ID="8hJmcHWTsdvkHyCrk4UgjbyugDAmE7QfuCTQXpXAc7nB"
-export E2E_REPO_NAME="m1-75299"
-export E2E_REPO_CONTRACT="5rrwgjjVUqMghnessfiXPXubpiM2QLNNXH142Hv4PDyX"
+: "${E2E_REPO_NAME:=m1-75299}"
+export E2E_REPO_NAME
export E2E_REMOTE="dash://${E2E_OWNER_ID}/${E2E_REPO_NAME}"
# --- dedicated bring-your-own-storage repos (e2e/cli/storage-byo.sh) ------------
@@ -27,6 +30,14 @@ export E2E_REMOTE="dash://${E2E_OWNER_ID}/${E2E_REPO_NAME}"
: "${STORAGE_E2E_REPO_B:=storage-e2e-b}"
export STORAGE_E2E_REPO STORAGE_E2E_REPO_B
+# --- the nightly's read fixture ----------------------------------------------
+# The repo the browser (Playwright) specs read: DEPLOYER-owned, written ONLY by
+# e2e/cli/seed-read-fixture.sh, which pins `main` to one deterministic commit and publishes
+# no browse index (so the in-browser fallback clone is what gets exercised). forge-web's
+# e2e/helpers.ts names the same repo.
+: "${NIGHTLY_FIXTURE_REPO:=m1-5124}"
+export NIGHTLY_FIXTURE_REPO
+
# --- fixture identity files --------------------------------------------------
: "${E2E_IDENTITY_DIR:=${HOME}/.config/dash-forge/test-identities}"
export E2E_IDENTITY_DIR
diff --git a/e2e/cli/run.sh b/e2e/cli/run.sh
index bf567996c..5d8e97a47 100755
--- a/e2e/cli/run.sh
+++ b/e2e/cli/run.sh
@@ -1,7 +1,7 @@
#!/usr/bin/env bash
# run.sh — Dash Forge CLI end-to-end suite driver.
#
-# Runs every scenario against LIVE testnet, reusing the DEPLOYER-owned m1 repo
+# Runs every scenario against LIVE testnet, on the CLI suite's reserved DEPLOYER-owned repo
# (config.sh). Prints a PASS/FAIL/SKIP matrix and exits non-zero if ANY scenario
# FAILs. A scenario SKIPs only when a check flaked on every retry; one SKIP is reported
# but tolerated, more than E2E_MAX_SKIPS fails the run.
diff --git a/e2e/cli/seed-read-fixture.sh b/e2e/cli/seed-read-fixture.sh
new file mode 100644
index 000000000..2d9896df3
--- /dev/null
+++ b/e2e/cli/seed-read-fixture.sh
@@ -0,0 +1,84 @@
+#!/usr/bin/env bash
+# seed-read-fixture.sh — make the nightly's browser read fixture exist and hold exactly
+# the expected content. Idempotent: a run that finds `main` already at the fixture commit
+# writes nothing.
+#
+# The fixture (NIGHTLY_FIXTURE_REPO, config.sh; reserved in e2e/README.md) is what
+# forge-web's Playwright specs read. It is written only here, so no other suite's refs,
+# packs or storage choices can reach what the browser sees:
+# * `main` = one deterministic commit (fixed author, committer and dates) holding a
+# README, a `src/` and a `lib/` directory — the entries the tree-browse spec looks for;
+# * its pack is stored on Platform (chunk documents), never external storage;
+# * no browse index is published (DASH_FORGE_NO_BROWSE_INDEX), so the web app takes the
+# in-browser fallback clone, which fallback-browse.spec.ts exercises.
+#
+# The repo is created (repo-v1, ~1.18 tDASH, once) only if absent. Exit 0 = seeded or
+# already seeded; 1 = could not seed.
+SCENARIO_NAME="seed the nightly read fixture (${NIGHTLY_FIXTURE_REPO:-m1-5124})"
+source "$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)/lib.sh"
+harness_init
+
+FIX_REPO="${E2E_OWNER_ID}/${NIGHTLY_FIXTURE_REPO}"
+FIX_REMOTE="dash://${FIX_REPO}"
+SRC="${WORKROOT}/fixture-src"
+LOG="${WORKROOT}/fixture"
+
+step "fixture repo ${FIX_REPO}"
+if dg_read_retry "$ID_DEPLOYER" "$LOG-view.json" "$LOG-view.err" --json repo view "$FIX_REPO"; then
+ info "reusing ${FIX_REPO}"
+elif grep -qiE 'not found|no such|does not exist|unknown repo' "$LOG-view.err"; then
+ info "creating ${FIX_REPO} (one-time, ~1.18 tDASH)"
+ dg_as "$ID_DEPLOYER" --yes --json repo create "$NIGHTLY_FIXTURE_REPO" \
+ --description "Dash Forge nightly read fixture (reserved; see e2e/README.md)" \
+ >"$LOG-create.out" 2>"$LOG-create.err" || {
+ cat "$LOG-create.err" >&2; bad "could not create ${FIX_REPO}"; finish_scenario
+ }
+else
+ cat "$LOG-view.err" >&2; bad "could not read ${FIX_REPO}"; finish_scenario
+fi
+
+step "build the deterministic fixture commit"
+rm -rf "$SRC"; mkdir -p "$SRC/src" "$SRC/lib"
+git init -q -b main "$SRC"
+cat >"$SRC/README.md" <<'EOF'
+# Dash Forge nightly read fixture
+
+This repository is the fixture the Dash Forge web app's browser tests read. It is written
+only by `e2e/cli/seed-read-fixture.sh`; do not push to it.
+EOF
+printf 'fn main() {\n println!("hello from the nightly fixture");\n}\n' >"$SRC/src/main.rs"
+printf 'pub fn answer() -> u32 {\n 42\n}\n' >"$SRC/lib/answer.rs"
+export GIT_AUTHOR_NAME="Dash Forge Fixture" GIT_AUTHOR_EMAIL="fixture@dash-forge.test"
+export GIT_COMMITTER_NAME="$GIT_AUTHOR_NAME" GIT_COMMITTER_EMAIL="$GIT_AUTHOR_EMAIL"
+export GIT_AUTHOR_DATE="2026-09-25T00:00:00Z" GIT_COMMITTER_DATE="2026-09-25T00:00:00Z"
+git -C "$SRC" -c commit.gpgsign=false add -A
+git -C "$SRC" -c commit.gpgsign=false commit -q -m "nightly read fixture"
+WANT="$(git -C "$SRC" rev-parse HEAD)"
+info "fixture commit ${WANT}"
+
+remote_main() { # prints the remote main oid, or nothing
+ git_dash_retry "$ID_DEPLOYER" "$1" ls-remote "$FIX_REMOTE" refs/heads/main || return 1
+ awk '$2 == "refs/heads/main" { print $1 }' "$1.out"
+}
+
+step "is ${FIX_REMOTE} main already at the fixture commit?"
+if ! HAVE="$(remote_main "$LOG-ls1")"; then
+ cat "$LOG-ls1.err" >&2; bad "could not list the fixture's refs"; finish_scenario
+fi
+if [[ "$HAVE" == "$WANT" ]]; then
+ ok "already seeded (main = ${WANT}); nothing written"
+ finish_scenario
+fi
+info "main is ${HAVE:-absent}; force-pushing the fixture commit"
+
+step "push the fixture (Platform storage, no browse index)"
+if ! DASH_FORGE_NO_BROWSE_INDEX=1 git_dash_retry "$ID_DEPLOYER" "$LOG-push" \
+ -C "$SRC" push "$FIX_REMOTE" "+refs/heads/main:refs/heads/main"; then
+ cat "$LOG-push.err" >&2; bad "fixture push failed"; finish_scenario
+fi
+if HAVE="$(remote_main "$LOG-ls2")" && [[ "$HAVE" == "$WANT" ]]; then
+ ok "seeded: main = ${WANT}"
+else
+ bad "after the push, main is ${HAVE:-unreadable}, not ${WANT}"
+fi
+finish_scenario
diff --git a/forge-web/e2e/helpers.ts b/forge-web/e2e/helpers.ts
index 2801fa59e..8b39b7b74 100644
--- a/forge-web/e2e/helpers.ts
+++ b/forge-web/e2e/helpers.ts
@@ -2,11 +2,17 @@ import { type Page, type ConsoleMessage } from '@playwright/test'
import { mkdirSync } from 'node:fs'
import { join } from 'node:path'
-/** Real testnet fixture: the m1 repo seeded by the CLI e2e suite. */
+/**
+ * Real testnet fixture: the nightly READ fixture, written only by
+ * `e2e/cli/seed-read-fixture.sh` (reserved in e2e/README.md). `main` holds one deterministic
+ * commit (README.md, src/, lib/) stored on Platform with no browse index, so the fallback
+ * clone is what the browse specs exercise. Never point these specs at a repo another suite
+ * writes: the storage e2e once left packs on a laptop's MinIO in the shared CLI repo, and
+ * every browse spec failed on them. Override with E2E_FIXTURE_OWNER / E2E_FIXTURE_NAME.
+ */
export const M1 = {
- owner: '8hJmcHWTsdvkHyCrk4UgjbyugDAmE7QfuCTQXpXAc7nB',
- name: 'm1-75299',
- contract: '5rrwgjjVUqMghnessfiXPXubpiM2QLNNXH142Hv4PDyX',
+ owner: process.env['E2E_FIXTURE_OWNER'] ?? '8hJmcHWTsdvkHyCrk4UgjbyugDAmE7QfuCTQXpXAc7nB',
+ name: process.env['E2E_FIXTURE_NAME'] ?? 'm1-5124',
} as const
export function repoUrl(path = ''): string {
From 670dbb37420eccdfd724bf7e4cf45fbb446ddd33 Mon Sep 17 00:00:00 2001
From: pasta
Date: Fri, 25 Sep 2026 09:47:00 -0500
Subject: [PATCH 05/11] fix: address review on external-pack fetches, fixture
gating and the contrast scan
External pack downloads use an idle deadline re-armed per chunk instead of a whole-body one, so a large pack from a healthy mirror is not reported unavailable; losing mirrors are cancelled once one has served a verified copy; a body larger than the manifest records is refused early. The read-fixture seed job needs only the DEPLOYER secret, and Playwright honours the seeder's NIGHTLY_FIXTURE_REPO override. The contrast scan reads whole class expressions, so text-white and a fill on different lines are still paired.
Co-Authored-By: Claude Opus 5.5 (1M context)
---
.github/workflows/testnet-nightly.yml | 9 +++-
forge-web/e2e/helpers.ts | 4 +-
forge-web/lib/design/contrast.test.ts | 49 +++++++++++++++---
forge-web/lib/view/browse-fallback.test.ts | 3 +-
forge-web/lib/view/browse-source.ts | 58 ++++++++++++++++++++--
5 files changed, 107 insertions(+), 16 deletions(-)
diff --git a/.github/workflows/testnet-nightly.yml b/.github/workflows/testnet-nightly.yml
index 761fc8d84..d1c3813f5 100644
--- a/.github/workflows/testnet-nightly.yml
+++ b/.github/workflows/testnet-nightly.yml
@@ -83,6 +83,8 @@ jobs:
runs-on: ubuntu-latest
outputs:
present: ${{ steps.probe.outputs.present }}
+ # The read-fixture seeder signs with DEPLOYER alone.
+ deployer: ${{ steps.probe.outputs.deployer }}
steps:
- name: Probe
id: probe
@@ -92,6 +94,11 @@ jobs:
FORGE_TEST_IDENTITY_CONTRIB: ${{ secrets.FORGE_TEST_IDENTITY_CONTRIB }}
FORGE_TEST_IDENTITY_FROZEN: ${{ secrets.FORGE_TEST_IDENTITY_FROZEN }}
run: |
+ if [ -n "$FORGE_TEST_IDENTITY_DEPLOYER" ]; then
+ echo "deployer=true" >> "$GITHUB_OUTPUT"
+ else
+ echo "deployer=false" >> "$GITHUB_OUTPUT"
+ fi
missing=()
# Report the SECRET names an operator has to create, not internal aliases.
for name in FORGE_TEST_IDENTITY_DEPLOYER FORGE_TEST_IDENTITY_COLLAB \
@@ -121,7 +128,7 @@ jobs:
seed-read-fixture:
name: seed the read fixture
needs: fixtures
- if: needs.fixtures.outputs.present == 'true'
+ if: needs.fixtures.outputs.deployer == 'true'
runs-on: ubuntu-latest
timeout-minutes: 45
env:
diff --git a/forge-web/e2e/helpers.ts b/forge-web/e2e/helpers.ts
index 8b39b7b74..1342b925b 100644
--- a/forge-web/e2e/helpers.ts
+++ b/forge-web/e2e/helpers.ts
@@ -12,7 +12,9 @@ import { join } from 'node:path'
*/
export const M1 = {
owner: process.env['E2E_FIXTURE_OWNER'] ?? '8hJmcHWTsdvkHyCrk4UgjbyugDAmE7QfuCTQXpXAc7nB',
- name: process.env['E2E_FIXTURE_NAME'] ?? 'm1-5124',
+ // The seeder's override (NIGHTLY_FIXTURE_REPO) applies here too, so a renamed fixture is
+ // seeded and read as the same repo.
+ name: process.env['E2E_FIXTURE_NAME'] ?? process.env['NIGHTLY_FIXTURE_REPO'] ?? 'm1-5124',
} as const
export function repoUrl(path = ''): string {
diff --git a/forge-web/lib/design/contrast.test.ts b/forge-web/lib/design/contrast.test.ts
index 8fc9b3a7b..2ccfe0ffc 100644
--- a/forge-web/lib/design/contrast.test.ts
+++ b/forge-web/lib/design/contrast.test.ts
@@ -154,15 +154,45 @@ describe('white text on solid fills meets WCAG AA', () => {
* value the file assigns.
*/
function whiteTextBackgrounds(text: string): { token: string; line: number }[] {
- const lines = text.split('\n')
const out: { token: string; line: number }[] = []
- const computed = /text-white[^\n]*\$\{[\w.]*\bbg\}/.test(text)
- lines.forEach((l, i) => {
- const isBgValue = computed && /\bbg:\s*/.test(l)
- if (!l.includes('text-white') && !isBgValue) return
- for (const m of l.matchAll(SOLID_BG)) out.push({ token: m[1] as string, line: i + 1 })
- })
- return out
+ const lineAt = (offset: number): number => text.slice(0, offset).split('\n').length
+ const scan = (chunk: string, base: number): void => {
+ for (const m of chunk.matchAll(SOLID_BG)) {
+ out.push({ token: m[1] as string, line: lineAt(base + (m.index ?? 0)) })
+ }
+ }
+ // Whole class expressions, however many lines they span: a `className="…"` string, a
+ // `className={…}` expression (template, `cn(…)` call — braces balanced), or a
+ // `cva`-style quoted string. Each that contains `text-white` is scanned as a unit.
+ const CLASS_EXPR = /className=(?:"[^"]*"|'[^']*'|\{)/g
+ for (const m of text.matchAll(CLASS_EXPR)) {
+ const start = m.index ?? 0
+ let end = start + m[0].length
+ if (m[0].endsWith('{')) {
+ let depth = 1
+ while (end < text.length && depth > 0) {
+ const c = text[end]
+ if (c === '{') depth += 1
+ else if (c === '}') depth -= 1
+ end += 1
+ }
+ }
+ const expr = text.slice(start, end)
+ if (expr.includes('text-white')) scan(expr, start)
+ }
+ // Class strings defined away from the element (a variants table): any quoted string
+ // holding both `text-white` and a solid fill.
+ for (const m of text.matchAll(/'[^'\n]*'|"[^"\n]*"/g)) {
+ if (m[0].includes('text-white') && !text.slice(Math.max(0, (m.index ?? 0) - 10), m.index).includes('className=')) {
+ scan(m[0], m.index ?? 0)
+ }
+ }
+ // A badge whose background is computed (`${status.bg}`): every `bg:` value the file
+ // assigns can end up behind the white text.
+ if (/text-white[\s\S]{0,200}?\$\{[\w.]*\bbg\}/.test(text)) {
+ for (const m of text.matchAll(/\bbg:\s*[^\n]*/g)) scan(m[0], m.index ?? 0)
+ }
+ return [...new Map(out.map((b) => [`${b.line}:${b.token}`, b])).values()]
}
it('checks the Merged / Open / Closed / Draft badges and every other white-text fill', () => {
@@ -191,6 +221,9 @@ describe('white text on solid fills meets WCAG AA', () => {
expect(contrast(rgb('#ffffff'), rgb(tokenHex('dash') as string))).toBeLessThan(AA_TEXT)
expect(contrast(rgb('#ffffff'), rgb(tokenHex('verify') as string))).toBeLessThan(AA_TEXT)
expect(tokens('')).toEqual(['forge-700'])
+ // text-white and the fill on different lines of one class expression.
+ expect(tokens('')).toEqual(['dash'])
+ expect(tokens('')).toEqual(['verify'])
expect(tokens('')).toEqual([])
})
})
diff --git a/forge-web/lib/view/browse-fallback.test.ts b/forge-web/lib/view/browse-fallback.test.ts
index 38efac59e..a5da58a72 100644
--- a/forge-web/lib/view/browse-fallback.test.ts
+++ b/forge-web/lib/view/browse-fallback.test.ts
@@ -225,7 +225,8 @@ describe('startFallback with external-storage packs', () => {
const platform = manifestFor(plat.pack, 1, { createdAt: 1, documentId: 'a' })
const external = manifestFor(ext.pack, 1, { storage: 1, uris: ['https://mirror.example/p'], createdAt: 2, documentId: 'b' })
const repo: RepoRef = { contractId: 'fallback-liar', ownerId: 'owner' }
- stubFetch({ 'https://mirror.example/p': () => blobPack('forged!!!\n').pack })
+ // Same length as the real pack, different bytes: only the sha256 check can catch it.
+ stubFetch({ 'https://mirror.example/p': () => blobPack('forgery!\n').pack })
const ctx = await startFallback(mockSdk(new Map([[platform.packHash, plat.pack]])), repo, [platform, external])
expect(ctx.unavailable?.[0]?.reason).toMatch(/sha256/)
await expect(ctx.reader.readObject(ext.oid)).rejects.toThrow(/could not be fetched/)
diff --git a/forge-web/lib/view/browse-source.ts b/forge-web/lib/view/browse-source.ts
index f8a1fc255..05e9234a4 100644
--- a/forge-web/lib/view/browse-source.ts
+++ b/forge-web/lib/view/browse-source.ts
@@ -297,17 +297,61 @@ export function externalFetchUrls(
* GET `url` and read the whole body under one hard deadline (a stalled body counts, not only
* a stalled connect). A 206 is accepted alongside 2xx.
*/
-async function fetchBody(url: string, init: RequestInit = {}): Promise {
+async function fetchBody(
+ url: string,
+ init: RequestInit = {},
+ opts: { readonly cancel?: AbortSignal; readonly maxBytes?: number } = {},
+): Promise {
const controller = new AbortController()
- const timer = setTimeout(() => controller.abort(), EXTERNAL_FETCH_TIMEOUT_MS)
+ let timedOut = false
+ let timer: ReturnType | undefined
+ // An IDLE deadline, re-armed on every chunk: a slow mirror streaming a large pack is
+ // fine, a silent one is not. A whole-body deadline would make every pack larger than
+ // bandwidth × deadline permanently "unavailable" from healthy mirrors.
+ const arm = (): void => {
+ clearTimeout(timer)
+ timer = setTimeout(() => {
+ timedOut = true
+ controller.abort()
+ }, EXTERNAL_FETCH_TIMEOUT_MS)
+ }
+ const onCancel = (): void => controller.abort()
+ opts.cancel?.addEventListener('abort', onCancel)
+ if (opts.cancel?.aborted) controller.abort()
+ arm()
try {
const resp = await fetch(url, { ...init, signal: controller.signal })
if (!resp.ok && resp.status !== 206) throw new Error(`HTTP ${resp.status}`)
- return new Uint8Array(await resp.arrayBuffer())
+ if (resp.body === null) return new Uint8Array(await resp.arrayBuffer())
+ const reader = resp.body.getReader()
+ const parts: Uint8Array[] = []
+ let total = 0
+ for (;;) {
+ arm()
+ const { done, value } = await reader.read()
+ if (done) break
+ total += value.length
+ // A mirror streaming more than the manifest says cannot be serving this pack.
+ if (opts.maxBytes !== undefined && total > opts.maxBytes) {
+ controller.abort()
+ throw new Error('served more bytes than the manifest records')
+ }
+ parts.push(value)
+ }
+ const out = new Uint8Array(total)
+ let at = 0
+ for (const p of parts) {
+ out.set(p, at)
+ at += p.length
+ }
+ return out
} catch (e) {
- throw controller.signal.aborted ? new Error(`no answer in ${EXTERNAL_FETCH_TIMEOUT_MS / 1000}s`) : e
+ if (timedOut) throw new Error(`no data for ${EXTERNAL_FETCH_TIMEOUT_MS / 1000}s`)
+ if (opts.cancel?.aborted) throw new Error('another mirror served it first')
+ throw e
} finally {
clearTimeout(timer)
+ opts.cancel?.removeEventListener('abort', onCancel)
}
}
@@ -362,9 +406,12 @@ async function fetchExternalWhole(
}
let corrupt = false
const reasons: string[] = []
+ // Cancels the losing mirrors once one has served the pack: an ipfs:// URI fans out to
+ // one request per gateway, and each would otherwise download (and hold) the whole pack.
+ const winner = new AbortController()
const attempt = async (url: string): Promise<{ url: string; bytes: Uint8Array }> => {
try {
- const bytes = await fetchBody(url)
+ const bytes = await fetchBody(url, {}, { cancel: winner.signal, maxBytes: manifest.sizeBytes })
if (bytes.length !== manifest.sizeBytes || bytesToHex(sha256(bytes)) !== want) {
corrupt = true
throw new Error('served bytes that do not match the manifest sha256')
@@ -377,6 +424,7 @@ async function fetchExternalWhole(
}
try {
const { url, bytes } = await firstFulfilled(urls.map(attempt))
+ winner.abort()
onServed?.(url)
return bytes
} catch {
From 3dfce37a09772ca4036af5d6a292327cb519d693 Mon Sep 17 00:00:00 2001
From: pasta
Date: Fri, 25 Sep 2026 10:36:55 -0500
Subject: [PATCH 06/11] test(e2e): accept either issue state label on the
fixture's issues page
The read fixture has no issues, so its issues page shows both the 'Open'/'Closed' filter tabs and the 'No issues' empty state, and the or() locator matched two elements under Playwright's strict mode. Take the first match.
Co-Authored-By: Claude Opus 5.5 (1M context)
---
forge-web/e2e/read-paths.spec.ts | 4 ++--
1 file changed, 2 insertions(+), 2 deletions(-)
diff --git a/forge-web/e2e/read-paths.spec.ts b/forge-web/e2e/read-paths.spec.ts
index 8fc9759d2..75f96df7a 100644
--- a/forge-web/e2e/read-paths.spec.ts
+++ b/forge-web/e2e/read-paths.spec.ts
@@ -90,7 +90,7 @@ test.describe('logged-out read paths', () => {
const openClosed = page.getByText(/\bOpen\b|\bClosed\b/i).first()
const emptyState = page.getByText(/no (open )?issues/i).first()
- await expect(openClosed.or(emptyState).or(readErrorBanner(page))).toBeVisible({
+ await expect(openClosed.or(emptyState).or(readErrorBanner(page)).first()).toBeVisible({
timeout: 30_000,
})
await shot(page, '03-issues')
@@ -101,7 +101,7 @@ test.describe('logged-out read paths', () => {
'rejects with a wasm-bindgen error). Open/closed issue folding could not be asserted.',
)
}
- await expect(openClosed.or(emptyState)).toBeVisible()
+ await expect(openClosed.or(emptyState).first()).toBeVisible()
})
test('4. tree browse reaches an honest terminal state (best-effort)', async ({ page }) => {
From dd81c46b801b455b77dc5753670be746f476cbdd Mon Sep 17 00:00:00 2001
From: pasta
Date: Fri, 25 Sep 2026 10:56:08 -0500
Subject: [PATCH 07/11] fix(refs): abandon the keyset scan on a page that
ignores its bound
Review #1: a page out of refNameHash order, holding a row at or below its '> last' bound, or a round that would not advance 'last' now stops the scan at once, and the answer comes from the reflog read of both types alone, deduplicated by $id (it used to keep paging and union). Both clients; a mock node that ignores the bound is covered. Review #7: the module docs now say what the prevOid check cannot see (a missing newest update or a whole ref) and that a 100+ update ref's equality read still carries protocol 13's same-$createdAt boundary gap until forge-v2.
Co-Authored-By: Claude Opus 5.5 (1M context)
---
crates/forge-core/src/refs.rs | 151 ++++++++++++++++++++++----------
forge-web/lib/repo/refs.test.ts | 20 ++++-
forge-web/lib/repo/refs.ts | 89 ++++++++++++-------
3 files changed, 181 insertions(+), 79 deletions(-)
diff --git a/crates/forge-core/src/refs.rs b/crates/forge-core/src/refs.rs
index ab7e54e5c..f34d54fdb 100644
--- a/crates/forge-core/src/refs.rs
+++ b/crates/forge-core/src/refs.rs
@@ -27,18 +27,39 @@
//! (`refNameHash == h`), which is single-branch — the cursor bug needs sibling branches —
//! and the scan resumes after it.
//!
-//! Each round strictly advances `last`, so the scan terminates without leaning on a page
-//! cap, and a one-ref read costs pages of that ref only, not of the whole repo.
+//! Each round must strictly advance `last` — the scan checks it — so it terminates without
+//! leaning on a page cap, and a one-ref read costs pages of that ref only, not of the whole
+//! repo.
//!
-//! ## Completeness check
+//! ## When the scan is not trusted
//!
-//! Every non-null `prevOid` a pusher records is the tip it saw, which is some earlier
-//! update's `newOid` for the same ref. An update whose `prevOid` matches nothing is
-//! therefore evidence that a row is missing, and triggers the fallback: every update of
-//! the type read in `$createdAt` (`reflog`) order, the only other complete read. Its rows are
-//! unioned with the scan's (both are proof-verified documents; a union can only add real
-//! rows). A dangling `prevOid` can also be written on purpose, which costs the fallback's
-//! extra reads but never changes the answer.
+//! The scan is abandoned, and every update of both types is re-read in `$createdAt`
+//! (`reflog`) order instead — the result then comes from that read ALONE, deduplicated by
+//! `$id` — when either:
+//!
+//! * a page comes back out of `refNameHash` order, holds a row at or below its
+//! `refNameHash > last` bound, or would not advance `last` (the node did not honor the
+//! query, so nothing it returned is trusted); or
+//! * the completeness check fails. Every non-null `prevOid` a pusher records is the tip it
+//! saw, which is some earlier update's `newOid` for the same ref; an update whose
+//! `prevOid` matches nothing is evidence of a missing row. A dangling `prevOid` can also
+//! be written on purpose (by a WRITE holder), which costs the fallback's extra reads but
+//! never changes the answer.
+//!
+//! ## What the completeness check cannot see
+//!
+//! It finds a gap only in the MIDDLE of a chain. A missing newest update, or a ref missing
+//! entirely, leaves no dangling `prevOid`, so the ref would quietly resolve to an older tip
+//! (or not be listed). The scan itself does not drop rows that way — it is cursor-free,
+//! which is the point — so this is a limit of the safety net, not a known failure.
+//!
+//! One path does still use a cursor: a ref with more than a page of updates is read with
+//! `refNameHash == h` paged by `startAfter`. That is single-branch, so the sibling-branch
+//! drop cannot happen, but protocol 13 still skips rows sharing the page boundary's
+//! `$createdAt` (docs/BUILDING.md, "same-block ties"), and repo-v1 ref updates do not return
+//! `$createdAt` from a proved query, so the tie probe cannot repair it. That read is the one
+//! `base_ref_tips` always used; it goes away with forge-v2 on protocol 14, whose cursor is
+//! bounded by document id. The mock's `ref_history` is exact, so no test here covers it.
use std::collections::{BTreeMap, BTreeSet};
@@ -175,45 +196,55 @@ pub(crate) async fn ref_history_with(
/// The scan behind [`read_all_ref_updates`], over any [`RefDocSource`].
pub(crate) async fn read_all_with(src: &impl RefDocSource) -> Result {
let mut docs: Vec<(Vec, bool, &str)> = Vec::new();
- let mut consistent = true;
+ let mut misbehaved = false;
for (doc_type, protected) in REF_UPDATE_TYPES {
- let (rows, ordered) = keyset_scan(src, doc_type).await?;
- consistent &= ordered;
+ let Some(rows) = keyset_scan(src, doc_type).await? else {
+ misbehaved = true;
+ break;
+ };
docs.push((rows, protected, doc_type));
}
- let mut by_hash = group(&docs)?;
- let dangling = by_hash.values().filter(|u| has_missing_parent(u)).count();
- if consistent && dangling == 0 {
- return Ok(by_hash);
+ if misbehaved {
+ tracing::warn!(
+ "a ref-update keyset page came back out of order or out of range; re-reading \
+ every update in reflog order"
+ );
+ } else {
+ let by_hash = group(&docs)?;
+ let dangling = by_hash.values().filter(|u| has_missing_parent(u)).count();
+ if dangling == 0 {
+ return Ok(by_hash);
+ }
+ tracing::warn!(
+ refs_with_missing_parent = dangling,
+ "ref-update keyset scan is missing parents; re-reading every update in reflog order"
+ );
}
-
- tracing::warn!(
- refs_with_missing_parent = dangling,
- out_of_order_page = !consistent,
- "ref-update keyset scan looks incomplete; re-reading every update in reflog order"
- );
- for (rows, _, doc_type) in &mut docs {
- let seen: BTreeSet = rows.iter().map(|d| d.id.clone()).collect();
- let extra: Vec = src
- .full_scan(doc_type)
- .await?
- .into_iter()
- .filter(|d| !seen.contains(&d.id))
- .collect();
- rows.extend(extra);
+ // The fallback stands alone: a scan that misbehaved or lost rows is not trusted for any
+ // of them, and the reflog read is complete on its own.
+ let mut full = Vec::with_capacity(REF_UPDATE_TYPES.len());
+ for (doc_type, protected) in REF_UPDATE_TYPES {
+ full.push((dedupe(src.full_scan(doc_type).await?), protected, doc_type));
}
- by_hash = group(&docs)?;
- Ok(by_hash)
+ group(&full)
}
-/// Page one type by key (see the module docs). Returns the rows and whether every page was
-/// ordered and within its bound — a page that is not says the node did not honor the query.
+/// Drop repeated `$id`s, keeping the first occurrence.
+fn dedupe(rows: Vec) -> Vec {
+ let mut seen = BTreeSet::new();
+ rows.into_iter()
+ .filter(|d| seen.insert(d.id.clone()))
+ .collect()
+}
+
+/// Page one type by key (see the module docs). `None` when the node did not honor the
+/// query: a page out of order, a row at or below the `refNameHash > after` bound, or a round
+/// that would not move `after` forward. The caller then discards the scan entirely.
async fn keyset_scan(
src: &impl RefDocSource,
doc_type: &str,
-) -> Result<(Vec, bool)> {
+) -> Result