C内核按库功能拆解,编译结果缓存区构建,编译过程与已有缓存结果对照功能实现

This commit is contained in:
lujingze committed 2026-09-12 05:24:48 +00:00
1 parent aa4951b14e
commit 151e6e4b97
30 files changed
+4856 -604

No files matched your search

@@ -0,0 +1,453 @@
"""Leases and bounded LRU eviction for immutable native cache entries.
Builds publish and validate their own artifacts. Readers hold a shared lease
before checking an entry and until they finish using it. Eviction only removes
an entry while holding the corresponding nonblocking exclusive lease.
"""
from __future__ import annotations
from contextlib import suppress
from dataclasses import dataclass
import errno
import os
from pathlib import Path
import re
import shutil
import stat
import weakref
MODEL_LIMIT_BYTES = 256 * 1024**2
OBJECT_LIMIT_BYTES = 128 * 1024**2
LOCK_SHARDS = 128
_KINDS = frozenset(("models", "objects"))
_KEY = re.compile(r"[0-9a-f]{64}\Z")
# tempfile.mkdtemp uses eight lowercase letters, digits or underscores. Older
# stages lacking a lease key are deliberately outside automatic cleanup.
_STAGE = re.compile(r"building-([0-9a-f]{64})-[a-z0-9_]{8}\Z")
if os.name == "nt":
import ctypes
from ctypes import wintypes
import msvcrt
class _Offset(ctypes.Structure):
_fields_ = [("Offset", wintypes.DWORD), ("OffsetHigh", wintypes.DWORD)]
class _OffsetUnion(ctypes.Union):
_fields_ = [("offset", _Offset), ("Pointer", ctypes.c_void_p)]
class _Overlapped(ctypes.Structure):
_fields_ = [
("Internal", ctypes.c_size_t), ("InternalHigh", ctypes.c_size_t),
("offset", _OffsetUnion), ("hEvent", wintypes.HANDLE),
]
_kernel32 = ctypes.WinDLL("kernel32", use_last_error=True)
_lock_file = _kernel32.LockFileEx
_lock_file.argtypes = [
wintypes.HANDLE, wintypes.DWORD, wintypes.DWORD,
wintypes.DWORD, wintypes.DWORD, ctypes.POINTER(_Overlapped),
]
_lock_file.restype = wintypes.BOOL
_unlock_file = _kernel32.UnlockFileEx
_unlock_file.argtypes = [
wintypes.HANDLE, wintypes.DWORD, wintypes.DWORD,
wintypes.DWORD, ctypes.POINTER(_Overlapped),
]
_unlock_file.restype = wintypes.BOOL
else:
import fcntl
def _validate(kind: str, key: str) -> None:
if kind not in _KINDS or not isinstance(key, str) or _KEY.fullmatch(key) is None:
raise ValueError("Native cache leases require models/objects and a lowercase SHA-256 key.")
def _reparse(info: os.stat_result) -> bool:
# Junctions as well as symbolic links are excluded on Windows.
return stat.S_ISLNK(info.st_mode) or bool(
getattr(info, "st_file_attributes", 0)
& getattr(stat, "FILE_ATTRIBUTE_REPARSE_POINT", 0x400)
)
def _directory(path: Path) -> os.stat_result | None:
try:
info = path.lstat()
except FileNotFoundError:
return None
if _reparse(info) or not stat.S_ISDIR(info.st_mode):
return None
return info
def _require_directory(path: Path, *, create: bool = False) -> None:
if create:
path.mkdir(parents=True, exist_ok=True)
if _directory(path) is None:
raise RuntimeError(f"Native cache directory must be a real directory: {path}")
def _lock(fd: int, *, exclusive: bool, blocking: bool) -> bool:
if os.name == "nt":
# os.open supplies a synchronous handle. LockFileEx therefore really
# waits without FAIL_IMMEDIATELY; unlike msvcrt.locking it also supports
# overlapping shared leases, including multiple builds in one process.
# The one-byte lock may extend beyond EOF, so lock files remain empty.
overlapped = _Overlapped()
flags = (2 if exclusive else 0) | (0 if blocking else 1)
if _lock_file(msvcrt.get_osfhandle(fd), flags, 0, 1, 0, ctypes.byref(overlapped)):
return True
code = ctypes.get_last_error()
if not blocking and code == 33: # ERROR_LOCK_VIOLATION
return False
raise ctypes.WinError(code)
operation = fcntl.LOCK_EX if exclusive else fcntl.LOCK_SH
if not blocking:
operation |= fcntl.LOCK_NB
while True:
try:
fcntl.flock(fd, operation)
return True
except InterruptedError:
continue
except OSError as exc:
if not blocking and exc.errno in (errno.EACCES, errno.EAGAIN):
return False
raise
def _release(fd: int, owner_pid: int) -> None:
# A forked child's finalizer must not explicitly unlock its parent's flock.
# Closing the inherited descriptor alone leaves the parent's lease intact.
with suppress(OSError):
if os.getpid() == owner_pid:
if os.name == "nt":
overlapped = _Overlapped()
_unlock_file(msvcrt.get_osfhandle(fd), 0, 1, 0, ctypes.byref(overlapped))
else:
fcntl.flock(fd, fcntl.LOCK_UN)
with suppress(OSError):
os.close(fd)
class CacheLease:
"""An acquired cache-use lock; close explicitly or use as a context manager."""
__slots__ = ("path", "kind", "key", "exclusive", "_finalizer", "__weakref__")
def __init__(self, fd: int, path: Path, kind: str, key: str, exclusive: bool):
self.path = path
self.kind = kind
self.key = key
self.exclusive = exclusive
self._finalizer = weakref.finalize(self, _release, fd, os.getpid())
@property
def closed(self) -> bool:
return not self._finalizer.alive
def close(self) -> None:
self._finalizer()
def __enter__(self) -> CacheLease:
if self.closed:
raise RuntimeError("Cannot reuse a closed native cache lease.")
return self
def __exit__(self, *_: object) -> None:
self.close()
def acquire_cache_lease(
cache_dir: Path, kind: str, key: str, *, exclusive: bool = False,
blocking: bool = True,
) -> CacheLease | None:
"""Acquire a shared-use/exclusive-eviction lease; return None only if busy.
Locks use 128 stable shards per namespace, bounding metadata as model keys
accumulate. A collision can defer eviction but cannot admit unsafe deletion.
Lock files must never be unlinked while the application is running: waiters
may already hold the old inode. There are at most 256 empty lock files.
"""
_validate(kind, key)
shard = int(key[:8], 16) % LOCK_SHARDS
return _acquire_lock_file(
Path(cache_dir).absolute(), kind, key, f"{kind}-{shard:03d}.lock",
exclusive=exclusive, blocking=blocking,
)
def _acquire_lock_file(
cache: Path, kind: str, key: str, lock_name: str, *, exclusive: bool,
blocking: bool,
) -> CacheLease | None:
_require_directory(cache, create=True)
lock_dir = cache / ".locks"
_require_directory(lock_dir, create=True)
lock_path = lock_dir / lock_name
try:
info = lock_path.lstat()
except FileNotFoundError:
info = None
if info is not None and (_reparse(info) or not stat.S_ISREG(info.st_mode)):
raise RuntimeError(f"Native cache lock must be a regular file: {lock_path}")
flags = os.O_RDWR | os.O_CREAT | getattr(os, "O_NOFOLLOW", 0) | getattr(os, "O_BINARY", 0)
fd = os.open(lock_path, flags, 0o600)
try:
if not stat.S_ISREG(os.fstat(fd).st_mode):
raise RuntimeError(f"Native cache lock must be a regular file: {lock_path}")
os.set_inheritable(fd, False)
if not _lock(fd, exclusive=exclusive, blocking=blocking):
os.close(fd)
return None
return CacheLease(fd, cache / kind / key, kind, key, exclusive)
except BaseException:
os.close(fd)
raise
def touch_cache_entry(cache_dir: Path, kind: str, key: str) -> None:
"""Mark an immutable entry as recently used while its caller holds a lease."""
_validate(kind, key)
cache = Path(cache_dir).absolute()
_require_directory(cache)
_require_directory(cache / kind)
entry = cache / kind / key
_require_directory(entry)
# Directory mtime is separate from the checksummed artifact/manifest bytes.
os.utime(entry, None, follow_symlinks=False)
@dataclass(frozen=True)
class _Entry:
path: Path
size: int
mtime_ns: int
device: int
inode: int
def _tree_size(path: Path) -> int | None:
"""Count regular file bytes without reading files or following any links."""
size = 0
try:
with os.scandir(path) as children:
for child in children:
info = child.stat(follow_symlinks=False)
if _reparse(info):
return None
if stat.S_ISREG(info.st_mode):
size += info.st_size
elif stat.S_ISDIR(info.st_mode):
nested = _tree_size(Path(child.path))
if nested is None:
return None
size += nested
else:
return None
except FileNotFoundError:
return None # A concurrent sweep may already have removed this entry.
return size
def _scan(namespace: Path) -> tuple[list[_Entry], int]:
entries = []
skipped = 0
if _directory(namespace) is None:
try:
namespace.lstat()
except FileNotFoundError:
return entries, 0
return entries, 1
with os.scandir(namespace) as children:
for child in children:
if _KEY.fullmatch(child.name) is None:
skipped += 1
continue
path = Path(child.path)
info = _directory(path)
if info is None:
skipped += 1
continue
size = _tree_size(path)
if size is None:
skipped += 1
continue
entries.append(_Entry(path, size, info.st_mtime_ns, info.st_dev, info.st_ino))
return entries, skipped
def _empty_report(limit: int) -> dict:
return {
"limitBytes": limit, "beforeBytes": 0, "afterBytes": 0,
"removedBytes": 0, "removedEntries": 0, "skippedInUse": 0,
"skippedChanged": 0, "skippedUnmanaged": 0, "oversizedEntries": 0,
"overLimitBytes": 0, "skippedConcurrentSweep": False, "errors": [],
}
def _prune_kind(cache: Path, kind: str, limit: int) -> dict:
report = _empty_report(limit)
namespace = cache / kind
try:
entries, skipped = _scan(namespace)
except OSError as exc:
report["errors"].append(str(exc))
return report
report["skippedUnmanaged"] = skipped
remaining = report["beforeBytes"] = sum(entry.size for entry in entries)
remaining_count = len(entries)
for entry in sorted(entries, key=lambda item: (item.mtime_ns, item.path.name)):
if remaining <= limit:
break
# Keep the last entry if it alone exceeds the budget. Older oversized
# entries are still evictable, so they cannot accumulate unboundedly.
if remaining_count == 1 and entry.size > limit:
continue
try:
lease = acquire_cache_lease(cache, kind, entry.path.name, exclusive=True, blocking=False)
if lease is None:
report["skippedInUse"] += 1
continue
with lease:
info = _directory(entry.path)
if info is None:
report["skippedChanged"] += 1
continue
if (info.st_dev, info.st_ino, info.st_mtime_ns) != (
entry.device, entry.inode, entry.mtime_ns,
):
# It was republished or reused after sorting; defer it to
# the next sweep instead of evicting a freshly used build.
report["skippedChanged"] += 1
continue
size = _tree_size(entry.path)
if size is None:
report["skippedUnmanaged"] += 1
continue
shutil.rmtree(entry.path)
remaining -= entry.size
remaining_count -= 1
report["removedBytes"] += size
report["removedEntries"] += 1
except (OSError, RuntimeError) as exc:
# Cleanup is best effort. A locked executable, permissions or a
# damaged unrelated cache entry must not fail a valid simulation.
report["errors"].append(str(exc))
try:
current, _ = _scan(namespace)
report["afterBytes"] = sum(entry.size for entry in current)
report["oversizedEntries"] = sum(entry.size > limit for entry in current)
except OSError as exc:
report["errors"].append(str(exc))
report["afterBytes"] = remaining
report["overLimitBytes"] = max(0, report["afterBytes"] - limit)
return report
def _empty_orphan_report() -> dict:
return {
"orphanStagesRemoved": 0, "orphanStagesBytes": 0,
"skippedInUse": 0, "skippedUnmanaged": 0,
"skippedConcurrentSweep": False, "errors": [],
}
def _prune_orphan_stages(cache: Path) -> dict:
"""Remove only named stages whose builder's lease is no longer held.
Callers acquire the embedded key's shared lease BEFORE creating the stage
and retain it until publication or cleanup. The OS releases that protection
if the builder is killed; no age threshold or unreliable PID test is needed.
Root/models stages use models leases; objects stages use objects leases.
"""
report = _empty_orphan_report()
for namespace, kind in ((cache, "models"), (cache / "models", "models"),
(cache / "objects", "objects")):
try:
if _directory(namespace) is None:
continue
with os.scandir(namespace) as children:
candidates = []
for child in children:
match = _STAGE.fullmatch(child.name)
if match is not None:
candidates.append((Path(child.path), match.group(1)))
elif child.name.startswith("building-"):
report["skippedUnmanaged"] += 1
except OSError as exc:
report["errors"].append(f"{namespace}: {exc}")
continue
for stage, key in candidates:
try:
lease = acquire_cache_lease(cache, kind, key, exclusive=True, blocking=False)
if lease is None:
report["skippedInUse"] += 1
continue
with lease:
if _directory(stage) is None:
report["skippedUnmanaged"] += 1
continue
size = _tree_size(stage)
if size is None:
report["skippedUnmanaged"] += 1
continue
shutil.rmtree(stage)
report["orphanStagesRemoved"] += 1
report["orphanStagesBytes"] += size
except (OSError, RuntimeError) as exc:
report["errors"].append(f"{stage}: {exc}")
return report
def prune_cache(
cache_dir: Path, *, model_limit_bytes: int = MODEL_LIMIT_BYTES,
object_limit_bytes: int = OBJECT_LIMIT_BYTES,
) -> dict:
"""Apply separate soft LRU byte budgets to managed models and objects.
In-use entries (including conservative shard collisions), the final entry
when it alone exceeds the budget, unknown names, links and legacy root-level
builds are retained. Returned counts explain any remaining overage. The
cache may transiently exceed its budgets while builds or solvers are active.
Strictly named stages abandoned by killed builders are removed under their
matching lease and counted separately in the orphanStages report.
"""
for limit in (model_limit_bytes, object_limit_bytes):
if isinstance(limit, bool) or not isinstance(limit, int) or limit < 0:
raise ValueError("Native cache byte limits must be nonnegative integers.")
cache = Path(cache_dir).absolute()
empty = {
"models": _empty_report(model_limit_bytes),
"objects": _empty_report(object_limit_bytes),
"orphanStages": _empty_orphan_report(),
}
try:
if _directory(cache) is None:
# Do not create a missing cache or inspect an alias to a directory.
return empty
# One additional stable lock file (257 total) serializes sweep decisions
# without serializing readers/builds. Otherwise simultaneous sweeps can
# each remove a different entry and incorrectly discard the final one.
sweep = _acquire_lock_file(
cache, "models", "0" * 64, "prune.lock", exclusive=True, blocking=False,
)
if sweep is None:
for report in empty.values():
report["skippedConcurrentSweep"] = True
return empty
with sweep:
orphan_stages = _prune_orphan_stages(cache)
return {
"models": _prune_kind(cache, "models", model_limit_bytes),
"objects": _prune_kind(cache, "objects", object_limit_bytes),
"orphanStages": orphan_stages,
}
except (OSError, RuntimeError) as exc:
for report in empty.values():
report["errors"].append(str(exc))
return empty