Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
20 changes: 20 additions & 0 deletions .github/workflows/ci.yml
Original file line number Diff line number Diff line change
Expand Up @@ -59,6 +59,26 @@ jobs:
uv run pytest -q \
-c benchmarks/frontierchallenge/pyproject.toml \
benchmarks/frontierchallenge/tests
# check_public_leaks.py --allow-empty passes a framework-only snapshot,
# so the "payload stays on Hugging Face" rule needs its own assertion.
found=$(git ls-files 'benchmarks/frontierchallenge/tasks/**' | head -20)
if [ -n "$found" ]; then
echo "::error::task payload belongs on Hugging Face, not in Git:"
echo "$found"
exit 1
fi
echo "no tracked task payload"
uv run python - <<'PY'
import json
registry = json.load(open("benchmarks/frontierchallenge/registry.json"))
assert registry["n_tasks"] == len(registry["tasks"]) == 97, (
f'registry drifted: n_tasks={registry["n_tasks"]}, '
f'tasks={len(registry["tasks"])}, expected 97'
)
ids = {row["id"] for row in registry["tasks"]}
assert len(ids) == 97, f"registry has {97 - len(ids)} duplicate task id(s)"
print("public registry: 97 unique task commitments")
PY
uv run python benchmarks/frontierchallenge/scripts/check_public_leaks.py \
benchmarks/frontierchallenge --allow-empty
uv run python benchmarks/frontierchallenge/scripts/check_restricted_software.py \
Expand Down
86 changes: 60 additions & 26 deletions apodex/clipboard.py
Original file line number Diff line number Diff line change
Expand Up @@ -12,7 +12,6 @@
import json
import os
import secrets
import shlex
import subprocess
import sys
import tempfile
Expand Down Expand Up @@ -117,34 +116,40 @@ def _read_macos_pasteboard(temp_dir: str) -> dict[str, Any]:
return payload


def _looks_like_file_urls(text: str) -> bool:
"""Report whether every non-blank line is a ``file://`` URL, without touching disk.

Used where the text is untrusted and must not be resolved: a purely textual
check leaks nothing about which host paths exist.
"""
lines = [line.strip() for line in text.splitlines() if line.strip()]
return bool(lines) and all(line.startswith("file://") for line in lines)


def _path_text(text: str) -> list[str] | None:
"""Return absolute existing paths represented by clipboard text."""
"""Return local paths represented by explicit ``file://`` URLs.

Plain absolute-path text is deliberately not promoted to an attachment:
copied webpage or chat content must not be able to make the client stage a
readable host file merely because its text happens to name one.

Only ever called on input the local user produced — a real pasteboard read,
or a paste into a TUI running natively on the host. Text arriving over the
broker is container-controlled and must not reach this function; see
``_broker_text_paste``.
"""
raw = text.strip()
if not raw:
return None
lines = [line.strip() for line in raw.splitlines() if line.strip()]
if lines and all(line.startswith("file://") for line in lines):
candidates = []
for line in lines:
parsed = urlparse(line)
if parsed.scheme == "file":
candidates.append(unquote(parsed.path))
elif len(lines) > 1 and all(
Path(line.strip("'\"")).expanduser().is_absolute()
and Path(line.strip("'\"")).expanduser().exists()
for line in lines
):
candidates = [line.strip("'\"") for line in lines]
else:
try:
candidates = shlex.split(raw)
except ValueError:
candidates = [line.strip() for line in raw.splitlines() if line.strip()]
# A single unescaped path may contain spaces. Prefer it when it exists.
if Path(raw).expanduser().is_absolute() and Path(raw).expanduser().exists():
candidates = [raw]
if not candidates:
if not lines or not all(line.startswith("file://") for line in lines):
return None
candidates: list[str] = []
for line in lines:
parsed = urlparse(line)
if parsed.scheme != "file" or parsed.netloc not in {"", "localhost"}:
return None
candidates.append(unquote(parsed.path))
resolved: list[str] = []
for candidate in candidates:
path = Path(candidate).expanduser()
Expand All @@ -157,7 +162,12 @@ def _path_text(text: str) -> list[str] | None:
def capture_macos_clipboard(
manager: AttachmentManager, *, pasted_text: str | None = None,
) -> ClipboardPaste:
"""Capture Finder files, an image, a path string, or ordinary text."""
"""Capture Finder files, an image, a path string, or ordinary text.

``pasted_text`` is trusted here: the only callers are the local user's own
paste, either natively or via the broker's pasteboard read. The broker's
request path deliberately does not route request text through this function.
"""
if pasted_text is not None:
paths = _path_text(pasted_text)
if paths is None:
Expand Down Expand Up @@ -204,6 +214,23 @@ def capture_macos_clipboard(
raise ClipboardError(str(payload.get("message") or "unsupported clipboard content"))


def _broker_text_paste(pasted_text: str) -> ClipboardPaste:
"""Wrap container-supplied paste text, explaining a gesture the bridge drops.

Dragging a file into the terminal pastes ``file://`` URLs, which the native
TUI stages as an attachment. Over the bridge the same text is indistinguishable
from a request forged by container code, so it stays text — but the user gets
told why, plus the Finder-copy route that does work through the bridge.
"""
message = (
"dropped file paths cannot be attached through the container clipboard "
"bridge; copy the file in Finder (Cmd-C) and paste again to attach it"
if _looks_like_file_urls(pasted_text)
else ""
)
return ClipboardPaste("text", text=pasted_text, message=message)


class ClipboardBroker:
"""Loopback-only host service used by the macOS Docker TUI."""

Expand All @@ -230,8 +257,15 @@ def do_POST(self) -> None:
pasted_text = request.get("text") if isinstance(request, dict) else None
if pasted_text is not None and not isinstance(pasted_text, str):
raise ClipboardError("invalid pasted text")
response = capture_macos_clipboard(
broker.manager, pasted_text=pasted_text,
# The bearer token is available to the container so it can
# call this bridge. Never treat request data as a host path:
# doing so would let container code copy arbitrary readable
# host files into its attachment mount. Only a real macOS
# pasteboard read may produce host file attachments.
response = (
_broker_text_paste(pasted_text)
if pasted_text is not None
else capture_macos_clipboard(broker.manager)
)
body = json.dumps(asdict(response)).encode()
self.send_response(200)
Expand Down
72 changes: 68 additions & 4 deletions apodex/tests/test_clipboard.py
Original file line number Diff line number Diff line change
Expand Up @@ -10,6 +10,7 @@
_BROKER_URL_ENV,
ClipboardBroker,
ClipboardPaste,
_looks_like_file_urls,
_path_text,
capture_macos_clipboard,
paste_from_clipboard,
Expand All @@ -22,17 +23,19 @@ def _manager(monkeypatch, tmp_path: Path) -> AttachmentManager:
return AttachmentManager(str(tmp_path), "session")


def test_path_text_accepts_quoted_and_file_url_paths(tmp_path: Path) -> None:
def test_path_text_requires_explicit_local_file_urls(tmp_path: Path) -> None:
source = tmp_path / "policy wording.pdf"
source.write_text("policy")
second = tmp_path / "claim photo.png"
second.write_bytes(b"png")

assert _path_text(f'"{source}"') == [str(source.resolve())]
assert _path_text(source.as_uri()) == [str(source.resolve())]
assert _path_text(f"{source}\n{second}") == [
assert _path_text(f"{source.as_uri()}\n{second.as_uri()}") == [
str(source.resolve()), str(second.resolve()),
]
assert _path_text(f'"{source}"') is None
assert _path_text(f"{source}\n{second}") is None
assert _path_text("file://evil.test/etc/passwd") is None
assert _path_text("ordinary clipboard text") is None


Expand All @@ -43,7 +46,7 @@ def test_capture_path_text_attaches_instead_of_inserting(
source.write_bytes(b"claim")
manager = _manager(monkeypatch, tmp_path)

result = capture_macos_clipboard(manager, pasted_text=str(source))
result = capture_macos_clipboard(manager, pasted_text=source.as_uri())

assert result == ClipboardPaste("attachments", ("claim.pdf",))
assert (manager.staging_dir / "claim.pdf").read_bytes() == b"claim"
Expand Down Expand Up @@ -88,3 +91,64 @@ def test_broker_round_trip_supports_clipboard_and_pasted_text(
assert paste_from_clipboard(manager, pasted_text="pasted").text == "pasted"
finally:
broker.close()


def test_broker_never_resolves_request_text_as_a_host_path(
monkeypatch, tmp_path: Path,
) -> None:
source = tmp_path / "host-secret.txt"
source.write_text("host-only")
manager = _manager(monkeypatch, tmp_path)
try:
broker = ClipboardBroker(manager)
except PermissionError:
pytest.skip("test sandbox does not allow loopback listeners")
broker.start()
monkeypatch.setenv(_BROKER_URL_ENV, f"http://127.0.0.1:{broker.port}")
monkeypatch.setenv(_BROKER_TOKEN_ENV, broker.token)
try:
result = paste_from_clipboard(manager, pasted_text=str(source))
finally:
broker.close()

assert result == ClipboardPaste("text", text=str(source))
assert manager.list() == []


def test_looks_like_file_urls_is_textual_only(tmp_path: Path) -> None:
missing = tmp_path / "absent.pdf"

# No disk access: a path that does not exist still reads as a file URL, so
# the broker cannot be used as an "does this host file exist" oracle.
assert _looks_like_file_urls(missing.as_uri()) is True
assert _looks_like_file_urls(f"{missing.as_uri()}\n{tmp_path.as_uri()}") is True
assert _looks_like_file_urls(str(missing)) is False
assert _looks_like_file_urls("ordinary clipboard text") is False
assert _looks_like_file_urls("") is False


def test_broker_explains_a_dropped_file_instead_of_silently_pasting_text(
monkeypatch, tmp_path: Path,
) -> None:
source = tmp_path / "claim.pdf"
source.write_bytes(b"claim")
manager = _manager(monkeypatch, tmp_path)
try:
broker = ClipboardBroker(manager)
except PermissionError:
pytest.skip("test sandbox does not allow loopback listeners")
broker.start()
monkeypatch.setenv(_BROKER_URL_ENV, f"http://127.0.0.1:{broker.port}")
monkeypatch.setenv(_BROKER_TOKEN_ENV, broker.token)
try:
dropped = paste_from_clipboard(manager, pasted_text=source.as_uri())
ordinary = paste_from_clipboard(manager, pasted_text="just text")
finally:
broker.close()

# The same gesture attaches natively, so the bridge says why it could not.
assert dropped.kind == "text"
assert dropped.text == source.as_uri()
assert "Finder" in dropped.message
assert manager.list() == []
assert ordinary == ClipboardPaste("text", text="just text")
4 changes: 4 additions & 0 deletions apodex/tui/app.py
Original file line number Diff line number Diff line change
Expand Up @@ -1027,6 +1027,10 @@ async def _paste_clipboard(self, pasted_text: str | None = None) -> None:
)
elif result.kind == "text":
self._insert_pasted_text(prompt, result.text)
if result.message:
# e.g. the container bridge cannot attach dropped file paths, so
# the paste stayed text — say why rather than leaving raw URLs.
self.notify(result.message, severity="warning")
else:
self.notify(result.message or "clipboard is empty or unsupported", severity="warning")
self._focus_prompt()
Expand Down
31 changes: 23 additions & 8 deletions benchmarks/frontierchallenge/scripts/check_restricted_software.py
Original file line number Diff line number Diff line change
Expand Up @@ -9,12 +9,13 @@
"""
from __future__ import annotations

import argparse
import re
import subprocess
import sys
from pathlib import Path

ROOT = Path(__file__).resolve().parents[1]
DEFAULT_ROOT = Path(__file__).resolve().parents[1]

FORBIDDEN_PATHS = {
"shared_images/Dockerfile.orca",
Expand All @@ -35,25 +36,39 @@
)


def tracked_files() -> list[str]:
def tracked_files(root: Path) -> list[str]:
try:
out = subprocess.check_output(
["git", "ls-files", "-z"], cwd=ROOT, stderr=subprocess.DEVNULL
["git", "ls-files", "-z"], cwd=root, stderr=subprocess.DEVNULL
).decode("utf-8", errors="surrogateescape")
return [p for p in out.split("\0") if p]
except subprocess.CalledProcessError:
# A history-free public export intentionally has no .git directory.
# Audit every materialized file there instead of skipping the gate.
return sorted(
path.relative_to(ROOT).as_posix()
for path in ROOT.rglob("*")
path.relative_to(root).as_posix()
for path in root.rglob("*")
if path.is_file()
)


def main() -> int:
def main(argv: list[str] | None = None) -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument(
"root",
nargs="?",
default=DEFAULT_ROOT,
type=Path,
help="tree to audit (default: the FrontierChallenge root beside this script)",
)
args = parser.parse_args(argv)
root: Path = args.root.resolve()
if not root.is_dir():
print(f"not a directory: {root}", file=sys.stderr)
return 2

problems: list[str] = []
tracked = tracked_files()
tracked = tracked_files(root)

for relative in tracked:
lower = relative.lower()
Expand All @@ -64,7 +79,7 @@ def main() -> int:
if lower.startswith("shared_images/") and "orca" in Path(lower).name:
problems.append(f"ORCA build artifact must not live in shared_images: {relative}")

path = ROOT / relative
path = root / relative
if not path.is_file() or path.suffix == ".fcref":
continue
try:
Expand Down
22 changes: 19 additions & 3 deletions benchmarks/public/judges/widesearch.py
Original file line number Diff line number Diff line change
Expand Up @@ -137,10 +137,26 @@ def metric_exact_match(response, target, criterion=None):
return (1.0, "match") if response.lower() == target.lower() else (0.0, "no match")


# Deliberately permissive so IDN/non-ASCII hosts and paths are captured too;
# prose punctuation that a URL cannot end with is trimmed afterwards instead of
# being excluded from the class (a URL body legitimately contains "." and ",").
_URL_RE = re.compile(r"https?://[^\s<>\"']+")
_URL_TRAILING_PUNCTUATION = ".,;:!?'\")]}>"


def _url_netlocs(text):
"""Collect the hostnames of every URL in ``text``, ignoring prose punctuation."""
found = set()
for raw in _URL_RE.findall(text):
trimmed = raw.rstrip(_URL_TRAILING_PUNCTUATION)
if trimmed:
found.add(urlparse(trimmed).netloc)
return found
Comment thread
zhanghanduo marked this conversation as resolved.


def metric_url_match(response, target, criterion=None):
pat = re.compile(r"http[s]?://(?:[a-zA-Z]|[0-9]|[$-_@.&+]|[!*\\(\\),]|(?:%[0-9a-fA-F][0-9a-fA-F]))+")
rd = {urlparse(u).netloc for u in pat.findall(response)}
td = {urlparse(u).netloc for u in pat.findall(target)}
rd = _url_netlocs(response)
td = _url_netlocs(target)
return (1.0, "match") if rd == td else (0.0, "no match")


Expand Down
Loading
Loading