a cell the matrix actually defines?
+# The grid is deliberately not fully populated: a backend that cannot run
+# as root has no cell for a target that needs root, because such a cell
+# would measure privilege rather than the filesystem. Every consumer of
+# the matrix asks this rather than assuming backends x targets, so the
+# workflow, the README grid, the badges and the report page agree on
+# which cells exist.
+cell_enabled() {
+ local backend="$1" version="$2" target="$3"
+ if backend_no_root "$(backend_slug "$backend" "$version")" \
+ && target_needs_root "$target"; then
+ return 1
+ fi
+ return 0
+}
diff --git a/bin/ci/render-report.py b/bin/ci/render-report.py
index 405e74d..88dca65 100755
--- a/bin/ci/render-report.py
+++ b/bin/ci/render-report.py
@@ -28,7 +28,7 @@
from pathlib import Path
import known_issues
-from evals import backend_slug, load_matrix
+from evals import backend_slug, cell_enabled, load_matrix
HERE = Path(__file__).resolve().parent
@@ -167,8 +167,8 @@ def main() -> int:
# Matrix order, so the page reads like the README grid.
m = load_matrix()
- backends = [(backend_slug(b["backend"], b["version"]), b["label"]) for b in m["backends"]]
- targets = [(t["name"], t["label"]) for t in m["targets"]]
+ backends = [(backend_slug(b["backend"], b["version"]), b["label"], b) for b in m["backends"]]
+ targets = [(t["name"], t["label"], t) for t in m["targets"]]
_, issues = known_issues.load_valid()
def state(c: dict) -> str:
@@ -186,9 +186,15 @@ def state(c: dict) -> str:
f"{npass}/{total} passing" + (f", {nnew} unexpected" if nnew else ""))
rows = []
- for bslug, blabel in backends:
+ for bslug, blabel, bdef in backends:
tds = [f"| {html.escape(blabel)} | "]
- for tname, tlabel in targets:
+ for tname, tlabel, tdef in targets:
+ # A pair the matrix does not define (see cell_enabled) is an
+ # explicit gap, not an "unknown" badge: nothing was measured
+ # here and nothing ever will be.
+ if not cell_enabled(bdef, tdef):
+ tds.append('n/a | ')
+ continue
slug = f"{bslug}-{tname}"
c = cells.get(slug, {"conclusion": "unknown"})
st = state(c)
@@ -214,7 +220,7 @@ def state(c: dict) -> str:
tds.append(f'{body}{meta}{cell_notes(c, st, fixed)} | ')
rows.append("" + "".join(tds) + "
")
- head = "".join(f"{html.escape(l)} | " for _, l in targets)
+ head = "".join(f"{html.escape(l)} | " for _, l, _ in targets)
repo = m.get("repo-slug", "con/eval-under")
doc = f"""
diff --git a/bin/ci/run-under.sh b/bin/ci/run-under.sh
index ff1a2cb..e35b2f8 100755
--- a/bin/ci/run-under.sh
+++ b/bin/ci/run-under.sh
@@ -13,10 +13,11 @@
# usage:
# bin/ci/run-under.sh [target]
#
-# backend = beegfs | nfs | loop
+# backend = beegfs | nfs | loop | sshfs
# version = for beegfs: point release (e.g. 7.4.6, 8.1.0)
# for loop: filesystem type (e.g. vfat, ext4)
# for nfs: literal "n/a"
+# for sshfs: literal "n/a"
# target = git-annex (default) | git | stress-ng | pjdfstest
#
# env overrides:
@@ -46,6 +47,20 @@ target_known "$TARGET" || {
exit 1
}
+# A root-requiring suite under a backend that can only run as the
+# invoking user would measure nothing: every privileged syscall it exists
+# to test fails for want of privilege, not because of the filesystem. The
+# NFS backend has --no-root-squash for this; sshfs has no equivalent,
+# because a FUSE mount belongs to whoever mounted it. Refuse the pair up
+# front rather than produce a meaningless red result.
+if [ "$BACKEND" = sshfs ] && target_needs_root "$TARGET"; then
+ echo "$TARGET needs root, and the sshfs backend always runs the wrapped" >&2
+ echo "command as the invoking user (a FUSE mount belongs to its mounter)." >&2
+ echo "There is no --no-root-squash equivalent here, so this combination" >&2
+ echo "cannot measure what the target is for. Try the nfs or loop backend." >&2
+ exit 2
+fi
+
# The target scripts re-derive their own defaults from matrix.sh, but an
# override handed to us must survive into the wrapped child.
export EVAL_UNDER_SRC_DIR
@@ -62,6 +77,7 @@ case "$BACKEND" in
# need an export that does not squash root, and need to keep
# their privileges rather than being dropped to the invoker.
target_needs_root "$TARGET" && opts=(--no-root-squash) ;;
+ sshfs) opts=() ;;
*) echo "unknown backend: $BACKEND" >&2; exit 1 ;;
esac
diff --git a/bin/eval-under-sshfs b/bin/eval-under-sshfs
new file mode 100755
index 0000000..3c952fa
--- /dev/null
+++ b/bin/eval-under-sshfs
@@ -0,0 +1,506 @@
+#!/bin/bash
+# SPDX-FileCopyrightText: 2026 Yaroslav Halchenko
+# SPDX-License-Identifier: MIT
+#
+# Generated with Claude Code
+#
+# eval-under-sshfs: run a command with TMPDIR / DATALAD_TESTS_TEMP_DIR (and
+# optionally HOME) pointing at an sshfs (FUSE-over-SFTP) mount.
+#
+# Backend of the eval-under framework.
+#
+# Two modes:
+#
+# loopback (default) -- brings up a throwaway sshd on a high port, with
+# its own host key and its own authorized_keys, and sshfs-mounts a
+# fresh backing directory back over it. Nothing outside the scratch
+# dir is touched: the user's ~/.ssh is never written to.
+#
+# remote (--host) -- sshfs-mounts a directory on a host you name, using
+# your existing ssh config/agent. For reproducing a reporter's
+# actual setup, where the interesting variable is often their
+# server's SFTP implementation rather than sshfs itself.
+
+set -eu
+
+# Empty MNT means "derive a per-run mountpoint next to the scratch dir"
+# (see the MNT_BASE block below); only an explicit --mount-point /
+# EVAL_UNDER_MOUNT pins the mount to a fixed path.
+MNT="${EVAL_UNDER_MOUNT:-}"
+KEEP="${EVAL_UNDER_KEEP:-0}"
+SET_HOME="${EVAL_UNDER_HOME_ON_MOUNT:-0}"
+
+# Loopback-mode knobs. An empty PORT means "pick a free one": two
+# eval-under runs on the same machine (or two CI cells on one runner)
+# must not fight over a fixed port.
+PORT="${EVAL_UNDER_SSHFS_PORT:-}"
+# Remote mode: unset means loopback.
+HOST="${EVAL_UNDER_SSHFS_HOST:-}"
+REMOTE_USER="${EVAL_UNDER_SSHFS_USER:-}"
+REMOTE_DIR="${EVAL_UNDER_SSHFS_DIR:-}"
+
+# sshfs -o options. EVAL_UNDER_SSHFS_OPTS is a comma-separated string
+# appended after ours, so a caller can override any default we set.
+EXTRA_OPTS="${EVAL_UNDER_SSHFS_OPTS:-}"
+NO_CACHE="${EVAL_UNDER_SSHFS_NO_CACHE:-0}"
+WORKAROUND="${EVAL_UNDER_SSHFS_WORKAROUND:-}"
+
+usage() {
+ cat <<'EOF'
+Usage: eval-under-sshfs [OPTIONS] -- CMD [ARGS...]
+
+Run CMD with TMPDIR / DATALAD_TESTS_TEMP_DIR (and optionally HOME)
+pointing at an sshfs mount.
+
+By default this is a loopback mount: a throwaway sshd is started on a
+high port with its own host key and its own authorized_keys file, and a
+fresh backing directory is sshfs-mounted back over it. Your ~/.ssh is
+not read or written. With --host, mounts a remote directory instead,
+using your normal ssh configuration.
+
+Requires the `sshfs` package (and, for loopback mode, `openssh-server`).
+Install with: apt install sshfs openssh-server
+
+Options (flag / env var / default / purpose):
+
+ --mount-point PATH EVAL_UNDER_MOUNT (next to scratch dir)
+ Where to mount sshfs on the host. Default: a fresh per-run
+ directory next to this run's scratch dir, both under $TMPDIR
+ (/tmp if unset) -- e.g. with TMPDIR=~/.tmp:
+
+ ~/.tmp/eval-under-sshfs-B3UXj.scratch keys, sshd conf, backing
+ ~/.tmp/eval-under-sshfs-B3UXj.sshfs where this run mounts it
+
+ A fixed default would collide between concurrent runs and hide
+ which run owned the mount. Pass this option when a stable path is
+ wanted.
+
+ --set-home EVAL_UNDER_HOME_ON_MOUNT (unset)
+ Also set HOME=/home for the wrapped command.
+
+ --keep EVAL_UNDER_KEEP (unset)
+ Skip teardown; leave the mount (and the throwaway sshd) up for
+ debugging.
+
+ --port N EVAL_UNDER_SSHFS_PORT (first free >=2222)
+ Port for the throwaway sshd (loopback mode only). Left unset, the
+ first free port at or above 2222 is used, and a port lost to a
+ concurrent run between the scan and the bind is retried on the
+ next one up, so concurrent runs do not collide.
+
+ --host HOST EVAL_UNDER_SSHFS_HOST (unset)
+ Mount from a remote host instead of starting a local sshd. Uses
+ your ssh config, keys and agent as-is. Implies that --port refers
+ to that host's sshd.
+
+ --user USER EVAL_UNDER_SSHFS_USER (invoking user)
+ Remote username. Requires --host: the throwaway sshd generated for
+ loopback mode only admits the invoking user, so any other name
+ there could only ever be refused.
+
+ --remote-dir PATH EVAL_UNDER_SSHFS_DIR (remote: required)
+ Directory on the remote host to mount. In loopback mode this
+ defaults to a fresh scratch directory and should be left alone.
+
+ --no-cache EVAL_UNDER_SSHFS_NO_CACHE (unset)
+ Pass `-o cache=no`. sshfs caches attributes and directory listings
+ by default, which papers over exactly the stale-stat behaviour a
+ reporter may be hitting. Turn the cache off to see the raw
+ protocol semantics.
+
+ --workaround LIST EVAL_UNDER_SSHFS_WORKAROUND (unset)
+ Pass `-o workaround=LIST` (sshfs's own compatibility switches,
+ e.g. `rename`, `truncate`, `buflimit`). `rename` is the
+ interesting one here: it makes sshfs emulate rename-over-existing
+ by unlinking the destination first, which is what it has to do
+ against SFTP servers lacking the POSIX-rename extension -- a
+ non-atomic rename, and precisely the semantics git and git-annex
+ assume they can rely on.
+
+ --opt OPTS EVAL_UNDER_SSHFS_OPTS (unset)
+ Extra comma-separated `-o` options, appended last so they win.
+ May be given more than once.
+
+ -h, --help
+ Print this help.
+
+Environment variables set FOR the wrapped command (all backends):
+
+ TMPDIR =
+ DATALAD_TESTS_TEMP_DIR =
+ HOME = /home (only if --set-home)
+
+Exit status: the wrapped command's exit status. Teardown runs on any
+exit (unless --keep).
+
+Examples:
+
+ # git-annex test over loopback sshfs, cache off
+ sudo bin/eval-under sshfs --no-cache --set-home -- \
+ bash -c 'cd "$HOME" && git init t && cd t && git annex init && git annex test'
+
+ # Reproduce non-atomic rename-over-existing
+ sudo bin/eval-under sshfs --workaround rename --set-home -- some-command
+
+ # Against a reporter's own server
+ sudo bin/eval-under sshfs --host store.example.org --remote-dir /data/scratch \
+ --set-home -- git annex fsck
+EOF
+}
+
+while [ $# -gt 0 ]; do
+ case "$1" in
+ --keep) KEEP=1; shift ;;
+ --mount-point) MNT="$2"; shift 2 ;;
+ --set-home) SET_HOME=1; shift ;;
+ --port) PORT="$2"; shift 2 ;;
+ --host) HOST="$2"; shift 2 ;;
+ --user) REMOTE_USER="$2"; shift 2 ;;
+ --remote-dir) REMOTE_DIR="$2"; shift 2 ;;
+ --no-cache) NO_CACHE=1; shift ;;
+ --workaround) WORKAROUND="$2"; shift 2 ;;
+ --opt) EXTRA_OPTS="${EXTRA_OPTS:+$EXTRA_OPTS,}$2"; shift 2 ;;
+ -h|--help) usage; exit 0 ;;
+ --) shift; break ;;
+ *) echo "unknown arg: $1" >&2; usage >&2; exit 2 ;;
+ esac
+done
+
+[ $# -gt 0 ] || { echo "no command given" >&2; usage >&2; exit 2; }
+
+
+# Did the caller pin the port? Decided before start_local_sshd starts
+# reassigning PORT while hunting for a free one.
+PORT_PINNED="${PORT:+1}"
+
+if [ "$(id -u)" -ne 0 ]; then
+ command -v sudo >/dev/null || {
+ echo "must run as root (or have sudo available)" >&2; exit 2; }
+ SUDO=(sudo)
+else
+ SUDO=()
+fi
+
+# The FUSE mount belongs to whoever runs sshfs, and the wrapped command
+# has to be able to read it. Both therefore run as the invoking user --
+# the same treatment bin/eval-under-nfs gives root_squash.
+if [ -n "${SUDO_UID:-}" ]; then
+ INVOKER_UID="$SUDO_UID"; INVOKER_GID="${SUDO_GID:-$SUDO_UID}"
+ INVOKER_USER="${SUDO_USER:-$INVOKER_UID}"
+else
+ INVOKER_UID="$(id -u)"; INVOKER_GID="$(id -g)"
+ INVOKER_USER="$(id -un)"
+fi
+
+# In loopback mode the generated sshd config has AllowUsers , so
+# a different --user could only ever be refused -- with nothing but
+# "Connection reset by peer" to explain it.
+if [ -n "$REMOTE_USER" ] && [ -z "$HOST" ]; then
+ echo "--user needs --host: the throwaway sshd only accepts $INVOKER_USER" >&2
+ exit 2
+fi
+
+# Dropping privileges needs `sudo` itself, so this deliberately does NOT
+# go through ${SUDO[@]}: that array is EMPTY when we are already root,
+# which is exactly the case that needs the drop. Under
+# `sudo eval-under sshfs ...` -- the documented invocation, and the one
+# bin/ci/run-under.sh uses -- id -u is 0 while INVOKER_UID is the real
+# user, so "${SUDO[@]}" -u would expand to a bare `-u`. bin/eval-under-nfs
+# spells out `sudo -u` for the same reason.
+as_invoker() {
+ if [ "$INVOKER_UID" = "$(id -u)" ]; then
+ "$@"
+ else
+ sudo -u "$INVOKER_USER" "$@"
+ fi
+}
+
+# Dropping privileges is only possible with sudo present, whether or not
+# we needed it to become root in the first place.
+if [ "$INVOKER_UID" != "$(id -u)" ]; then
+ command -v sudo >/dev/null || {
+ echo "ERROR: need sudo to run as the invoking user ($INVOKER_USER)" >&2
+ exit 2
+ }
+fi
+
+log() { printf '\nI: %s\n' "$*"; }
+
+command -v sshfs >/dev/null 2>&1 || {
+ echo "ERROR: sshfs not installed. Install with: apt install sshfs" >&2
+ exit 3
+}
+
+# The scratch dir (throwaway host key, sshd config, authorized_keys and
+# the backing directory that gets re-exported) and the mountpoint derive
+# from one mktemp -u base, so a run leaves a self-describing pair of
+# siblings under $TMPDIR -- the convention every backend here follows
+# (see "Adding a new backend" in the README).
+MNT_BASE="$(mktemp -u "${TMPDIR:-/tmp}/eval-under-sshfs-XXXXX")"
+SCRATCH="$MNT_BASE.scratch"
+MNT="${MNT:-$MNT_BASE.sshfs}"
+mkdir -p "$SCRATCH"
+# 755, not mktemp's 700: sshd reads its config and authorized_keys from
+# here as root while the mount and the wrapped command run as the
+# invoking user. The mode is not enough on its own -- root created the
+# directory, so the invoker also has to own it or every as_invoker write
+# into it (the keys, authorized_keys, the backing dir) is EACCES.
+chmod 755 "$SCRATCH"
+chown "$INVOKER_UID:$INVOKER_GID" "$SCRATCH"
+SSHD_PID=""
+
+# Whether teardown created the mountpoint and may therefore remove it.
+MNT_CREATED=0
+
+teardown() {
+ local rc=$?
+ set +e
+ if [ "$KEEP" = 1 ]; then
+ echo "I: --keep set; leaving mount up (exit=$rc)"
+ echo "I: mount: $MNT"
+ echo "I: scratch: $SCRATCH"
+ [ -n "$SSHD_PID" ] && echo "I: sshd pid: $SSHD_PID (kill it yourself)"
+ return "$rc"
+ fi
+ log "teardown"
+ # Kill the wrapped command first. On a signal it is still running, and
+ # it must not survive the filesystem it is working on.
+ pkill -P $$ 2>/dev/null
+ if command -v fuser >/dev/null 2>&1; then
+ "${SUDO[@]}" fuser -k -M "$MNT" 2>/dev/null
+ fi
+ # fusermount as the mount's owner; fall back to a lazy root umount.
+ as_invoker fusermount3 -u "$MNT" 2>/dev/null \
+ || as_invoker fusermount -u "$MNT" 2>/dev/null \
+ || "${SUDO[@]}" umount -l "$MNT" 2>/dev/null
+ # Kill the whole process group: the daemon forks a child per
+ # connection, and killing only the listener leaves those behind
+ # holding the port.
+ [ -n "$SSHD_PID" ] && { "${SUDO[@]}" kill -- "-$SSHD_PID" 2>/dev/null \
+ || "${SUDO[@]}" kill "$SSHD_PID" 2>/dev/null; }
+ # Only remove the mountpoint if this run made it: an explicit
+ # --mount-point may well be a directory that predates us.
+ if [ "$MNT_CREATED" = 1 ]; then
+ "${SUDO[@]}" rm -rf "$MNT" 2>/dev/null
+ fi
+ "${SUDO[@]}" rm -rf "$SCRATCH" 2>/dev/null
+ return "$rc"
+}
+# Teardown on a signal too, not just a clean exit: otherwise a SIGTERM
+# unmounts and rm -rf's the filesystem out from under a wrapped command
+# that is still running, and the orphan keeps writing into a deleted
+# inode. on_signal drops the EXIT trap so teardown runs exactly once.
+on_signal() {
+ trap - EXIT
+ teardown
+ exit 130
+}
+trap teardown EXIT
+trap on_signal INT TERM HUP
+
+# First free TCP port at or above $1 on loopback. `ss` if we have it,
+# otherwise a bash /dev/tcp connect probe: a refused connection means
+# nothing is listening.
+find_free_port() {
+ local p="$1" limit=$(( $1 + 200 ))
+ while [ "$p" -lt "$limit" ]; do
+ if command -v ss >/dev/null 2>&1; then
+ ss -ltnH "sport = :$p" 2>/dev/null | grep -q . || { echo "$p"; return 0; }
+ else
+ if ! (exec 3<>"/dev/tcp/127.0.0.1/$p") 2>/dev/null; then
+ echo "$p"; return 0
+ fi
+ fi
+ p=$(( p + 1 ))
+ done
+ return 1
+}
+
+find_sftp_server() {
+ local p
+ for p in /usr/lib/openssh/sftp-server \
+ /usr/libexec/openssh/sftp-server \
+ /usr/lib/ssh/sftp-server \
+ /usr/libexec/sftp-server; do
+ [ -x "$p" ] && { echo "$p"; return 0; }
+ done
+ return 1
+}
+
+# Bring up an sshd that exists only for this run: own port, own host key,
+# own authorized_keys, own pid file. Deliberately NOT the system sshd --
+# we would otherwise have to append to the user's authorized_keys and
+# hope teardown removes it again.
+start_local_sshd() {
+ local sftp_server sshd_bin
+ sshd_bin="$(command -v sshd || echo /usr/sbin/sshd)"
+ [ -x "$sshd_bin" ] || {
+ echo "ERROR: sshd not found. Install with: apt install openssh-server" >&2
+ exit 3
+ }
+ sftp_server="$(find_sftp_server)" || {
+ echo "ERROR: no sftp-server binary found (openssh-sftp-server?)" >&2
+ exit 3
+ }
+
+ as_invoker ssh-keygen -t ed25519 -N '' -q -f "$SCRATCH/id" -C eval-under-sshfs
+ as_invoker ssh-keygen -t ed25519 -N '' -q -f "$SCRATCH/hostkey" -C eval-under-sshfs-host
+ as_invoker cp "$SCRATCH/id.pub" "$SCRATCH/authorized_keys"
+ chmod 600 "$SCRATCH/authorized_keys"
+
+ # sshd refuses to start without its privilege-separation directory,
+ # which is absent in minimal containers (it is created at boot by the
+ # packaging, not by the package install).
+ "${SUDO[@]}" mkdir -p /run/sshd
+
+ # Finding a free port and binding it are two steps, and a concurrent
+ # run can take the port in between -- with four runs on one machine,
+ # three used to die on "sshd did not start". So treat a failed bind as
+ # a lost race and try the next free port up, unless the caller pinned
+ # one with --port.
+ local tries=0 from=2222
+ while [ "$tries" -lt 5 ]; do
+ tries=$((tries + 1))
+ [ -n "$PORT_PINNED" ] || PORT="$(find_free_port "$from")" || {
+ echo "ERROR: no free port at or above $from" >&2; exit 3; }
+ write_sshd_config "$sftp_server"
+ log "starting throwaway sshd on 127.0.0.1:$PORT"
+ # Run it as root so any AllowUsers target can log in; it is bound to
+ # loopback and accepts a single ephemeral key. -E keeps its log, which
+ # is the only place an auth refusal is ever explained.
+ if "${SUDO[@]}" "$sshd_bin" -f "$SCRATCH/sshd_config" \
+ -E "$SCRATCH/sshd.log" && wait_for_sshd_pid; then
+ log "sshd pid $SSHD_PID (port $PORT)"
+ return 0
+ fi
+ # Reap anything that did start before we move the port: otherwise it
+ # survives teardown, because SSHD_PID is still empty.
+ "${SUDO[@]}" pkill -f "sshd -f $SCRATCH/sshd_config" 2>/dev/null || true
+ [ -z "$PORT_PINNED" ] || break
+ from=$((PORT + 1))
+ done
+
+ echo "ERROR: sshd did not start" >&2
+ dump_sshd_log
+ exit 3
+}
+
+# The config is rewritten per attempt because the Port line changes.
+write_sshd_config() {
+ cat > "$SCRATCH/sshd_config" <&2
+ sed 's/^/E: /' "$SCRATCH/sshd.log" >&2
+}
+
+# Assemble the -o list. Ours first, caller's last so theirs wins.
+build_opts() {
+ local opts="reconnect,ServerAliveInterval=15,ServerAliveCountMax=3"
+ if [ -z "$HOST" ]; then
+ # Loopback: a throwaway host key we just generated is not worth
+ # recording, and would collide on the next run.
+ opts="$opts,IdentityFile=$SCRATCH/id,IdentitiesOnly=yes"
+ opts="$opts,StrictHostKeyChecking=no,UserKnownHostsFile=/dev/null"
+ fi
+ [ "$NO_CACHE" != 0 ] && opts="$opts,cache=no"
+ [ -n "$WORKAROUND" ] && opts="$opts,workaround=$WORKAROUND"
+ [ -n "$EXTRA_OPTS" ] && opts="$opts,$EXTRA_OPTS"
+ echo "$opts"
+}
+
+mount_sshfs() {
+ local spec opts
+ opts="$(build_opts)"
+
+ if [ -z "$HOST" ]; then
+ start_local_sshd
+ REMOTE_DIR="${REMOTE_DIR:-$SCRATCH/backing}"
+ as_invoker mkdir -p "$REMOTE_DIR"
+ spec="${REMOTE_USER:-$INVOKER_USER}@127.0.0.1:$REMOTE_DIR"
+ else
+ [ -n "$REMOTE_DIR" ] || {
+ echo "ERROR: --host requires --remote-dir" >&2; exit 2; }
+ spec="${REMOTE_USER:-$INVOKER_USER}@$HOST:$REMOTE_DIR"
+ fi
+
+ [ -d "$MNT" ] || MNT_CREATED=1
+ "${SUDO[@]}" mkdir -p "$MNT"
+ # Only re-own a directory we made. An explicit --mount-point may be
+ # someone's existing directory, and teardown has no way to put its
+ # ownership back.
+ if [ "$MNT_CREATED" = 1 ]; then
+ "${SUDO[@]}" chown "$INVOKER_UID:$INVOKER_GID" "$MNT"
+ fi
+
+ # Remote mode never called start_local_sshd, so PORT may still be
+ # empty: that is plain ssh, which means 22.
+ log "sshfs -p ${PORT:=22} -o $opts $spec $MNT"
+ if ! as_invoker sshfs -p "$PORT" -o "$opts" "$spec" "$MNT"; then
+ echo "ERROR: sshfs mount failed" >&2
+ dump_sshd_log
+ exit 3
+ fi
+
+ # An sshfs mount can appear in the mount table before the first
+ # request round-trips. Touch it once so a failure surfaces here rather
+ # than inside the wrapped command.
+ as_invoker test -d "$MNT" || { echo "ERROR: mount not usable" >&2; exit 3; }
+ "${SUDO[@]}" mount | grep -F "$MNT" | sed 's/^/I: /'
+}
+
+mount_sshfs
+
+RUN_HOME="$HOME"
+if [ "$SET_HOME" = 1 ]; then
+ RUN_HOME="$MNT/home"
+ as_invoker mkdir -p "$RUN_HOME"
+fi
+
+if [ "$INVOKER_UID" != "$(id -u)" ]; then
+ log "running (as uid=$INVOKER_UID): $*"
+ # Literal sudo, not "${SUDO[@]}" -- same reason as in as_invoker: the
+ # array is empty precisely when we are root and need the drop.
+ sudo -u "$INVOKER_USER" -E \
+ env HOME="$RUN_HOME" TMPDIR="$MNT" DATALAD_TESTS_TEMP_DIR="$MNT" \
+ "$@"
+else
+ log "running: $*"
+ HOME="$RUN_HOME" TMPDIR="$MNT" DATALAD_TESTS_TEMP_DIR="$MNT" "$@"
+fi
diff --git a/evals/known-issues.yaml b/evals/known-issues.yaml
index ef52a2e..af41cbb 100644
--- a/evals/known-issues.yaml
+++ b/evals/known-issues.yaml
@@ -166,3 +166,135 @@ issues:
tests: ["*"]
tags: [harness, needs-triage]
links: ["GOTCHAS.md#loop-git-annex-cells-annexdiskreserve"]
+
+ # --------------------------------------------------------------- sshfs
+ - id: sshfs-git-local-clone-hardlink
+ title: local `git clone` verifies its hardlinks, and sshfs synthesises st_ino
+ backends: [sshfs]
+ targets: [git]
+ tests:
+ - "t0001-init.sh#37"
+ - "t0003-attributes.sh#24-25,29,32-34"
+ - "t0021-conversion.sh#28-30"
+ - "t0033-safe-directory.sh#16"
+ - "t0035-safe-bare-repository.sh#1,13"
+ - "t0410-partial-clone.sh#34,38"
+ - "t0610-reftable-basics.sh#26-28"
+ - "t1013-read-tree-submodule.sh#1-8,10-15,18-28,30-48,51-60,65-68"
+ - "t1060-object-corruption.sh#12"
+ - "t1091-sparse-checkout-builtin.sh#31-32,36-38,49,77"
+ - "t1350-config-hooks-path.sh#4"
+ - "t1423-ref-backend.sh#36"
+ - "t1460-refs-migrate.sh#9,24"
+ - "t1500-rev-parse.sh#77"
+ - "t1507-rev-parse-upstream.sh#1-7,9-11,13-14,17-18,21,23-27"
+ - "t1600-index.sh#6"
+ tags: [fs-divergence]
+ links: ["GOTCHAS.md#sshfs-bineval-under-sshfs"]
+ notes: |
+ SFTP's `ATTRS` carries no inode number, so sshfs synthesises `st_ino`
+ per path. `git clone ` hardlinks each object and then
+ compares `st_mode`/`st_ino`/`st_dev`/`st_size`/`st_uid`/`st_gid`
+ against the source (`builtin/clone.c`), so the check fails and the
+ clone dies with `fatal: hardlink different from source`. **Plain
+ `git clone` of a local path does not work on sshfs at all.**
+
+ Each script here clones, or adds a submodule (which clones), in its
+ setup, so one failed setup cascades through the script; the scripts
+ were attributed by that message appearing in their own logs.
+ `git clone --no-hardlinks`, `git clone file://...` and mounting with
+ `-o disable_hardlink` all work -- with the option, sshfs fails
+ `link()` with `EPERM` instead of pretending, and git falls back to
+ copying.
+
+ Test ids seeded from run 36476334300 (git v2.55.0), 110 assertions
+ across 16 scripts.
+
+ - id: sshfs-git-no-unix-sockets
+ title: git's IPC and credential-cache tests need Unix sockets on the work tree
+ backends: [sshfs]
+ targets: [git]
+ tests:
+ - "t0052-simple-ipc.sh#1-9"
+ - "t0301-credential-cache.sh#2-3,7-8,10-11,13-23,25-26,28,30,32-33,37-38,40-41,43-52"
+ tags: [fs-limitation]
+ links: ["GOTCHAS.md#sshfs-bineval-under-sshfs"]
+ notes: |
+ sshfs has no Unix sockets: `bind()` on the mount fails with
+ `Operation not permitted`, so the credential-cache daemon never
+ starts (`unable to bind to .../credential/socket`) and simple-ipc
+ finds `no server listening`. `unix-socket=no` in
+ `bin/ci/fs-capabilities.sh` predicts both. The same limitation is
+ recorded for vfat as `vfat-git-no-unix-sockets`.
+
+ - id: sshfs-git-untriaged
+ title: remaining git failures on sshfs, not yet attributed
+ backends: [sshfs]
+ targets: [git]
+ tests:
+ # Observed across CI runs 36476334300 and 36485450259. This list
+ # DOCUMENTS a sample; it does not gate -- see the notes.
+ - "t0003-attributes.sh#41,48"
+ - "t0017-env-helper.sh#4"
+ - "t0040-parse-options.sh#37"
+ - "t0061-run-command.sh#6,18"
+ - "t0302-credential-store.sh#57"
+ - "t0450-txt-doc-vs-help.sh#131,647,797"
+ - "t0610-reftable-basics.sh#61"
+ - "t1091-sparse-checkout-builtin.sh#21,43,48"
+ - "t1092-sparse-checkout-compatibility.sh#55"
+ - "t1300-config.sh#194,197,237,285,494"
+ - "t1403-show-ref.sh#9"
+ - "t1430-bad-ref-name.sh#26"
+ - "t1450-fsck.sh#36"
+ - "t1461-refs-list.sh#415"
+ - "t1503-rev-parse-verify.sh#4"
+ - "t1700-split-index.sh#10-12,14-15"
+ tags: [needs-triage]
+ links: ["GOTCHAS.md#sshfs-bineval-under-sshfs"]
+ notes: |
+ **These are flaky, not fixed divergences, and this entry cannot
+ gate the cell.** Two mechanisms above are deterministic; this
+ residue is not. Measured three ways:
+
+ - The same cell on two CI runs of near-identical code
+ (36476334300, then 36485450259) reported 12 and 18 residual
+ failures with **no overlap**: every id the first run flagged
+ passed in the second, and vice versa. The two mechanism entries
+ meanwhile reproduced exactly both times, 110 and 46.
+ - Locally, six runs of `t0003 t0017 t0040 t1700` under sshfs:
+ `t0003`'s six hardlink assertions failed in all six runs, while
+ `t0017#4` failed in one and `t0040#37`, `t1700#10-15` and
+ `t0003#41`/`#48` in none -- although CI has flagged each of them.
+ - `prove --jobs 1` is no cleaner than `--jobs 4`, and 14 runs of
+ `t0017` alone were all clean, so it takes the fuller suite's
+ concurrent load to show up at all.
+
+ Some of these tests touch no filesystem semantics whatsoever --
+ `t0040#37` "OPT_CALLBACK() and OPT_BIT() work" and `t0017#4`
+ "test-tool env-helper --type=ulong" parse arguments and
+ environment variables, and `t0450` compares documentation against
+ `-h` output. What they do share is capturing output into `>out` /
+ `2>err` inside the trash directory on the mount and then grepping
+ it, which points at the mount losing or delaying writes under
+ concurrent load rather than at any semantic divergence.
+
+ So a `script#N` list is the wrong instrument here: each run draws a
+ different sample, and pinning one run's sample is what made this
+ cell report `failing-new` twice.
+
+ Mounting with `-o cache=no` (`--no-cache`) was measured, not
+ guessed: five runs of `t0*.sh` at `--jobs 4`, counting failures
+ beyond the 64 this subset's two mechanisms own. Baseline drew 4, 8
+ and 8; `cache=no` drew 1 and 2, for about 7% more wall clock (107s
+ -> 114s). So the cache is implicated and `cache=no` is worth having
+ as a mitigation -- but it does not fix the gate, because
+ `known_issues.py` sets `failing-new` on the *first* uncovered
+ failure and has no tolerance for a flake budget. A rate of one per
+ run still fails the cell most runs.
+
+ Which leaves a decision rather than a patch: cover the whole cell
+ with `tests: ["*"]` as `vfat-pjdfstest` does (green, but it masks
+ the 110 and 46 findings and any future regression), make the cell
+ non-gating, or give the verdict machinery a flake budget. That last
+ one belongs in the machinery, not in this backend's PR.
diff --git a/evals/matrix.yaml b/evals/matrix.yaml
index 350a674..aad8a17 100644
--- a/evals/matrix.yaml
+++ b/evals/matrix.yaml
@@ -51,7 +51,8 @@ refs:
# than a bare GitHub cancellation.
# loop-size-mb backing-image size for `eval-under loop --size`.
# needs-root the suite is meaningless without privilege, so the
-# NFS backend must export no_root_squash for it.
+# NFS backend must export no_root_squash for it, and a
+# backend marked `no-root` gets no cell for it at all.
# pjdfstest is half privileged-vs-unprivileged
# assertions and refuses to run otherwise; stress-ng's
# chown/mknod stressors need CAP_CHOWN / CAP_MKNOD.
@@ -100,6 +101,19 @@ backends:
- backend: nfs
version: n/a
label: NFS (localhost)
+
+ # `no-root` says this backend cannot hand the wrapped suite privilege:
+ # a FUSE mount belongs to whoever mounted it, and there is no
+ # --no-root-squash equivalent the way there is for NFS. So this row has
+ # no cell for a needs-root target (stress-ng, pjdfstest) -- such a cell
+ # would report on privilege rather than on the filesystem, which is the
+ # same reason bin/ci/run-under.sh refuses the pair outright. The two
+ # cells that remain are the ones that found something: git-annex is red
+ # here by design (invisible hardlinks), git is the control.
+ - backend: sshfs
+ version: n/a
+ label: sshfs (loopback)
+ no-root: true
- backend: loop
version: vfat
label: Loop vfat