收窄项目写锁同进程复用判据为同线程重入
Project CI / Repository checks (push) Successful in 3m21s
Project CI / Frontend tests (push) Successful in 4m15s
Project CI / Backend tests (push) Successful in 6m13s
Project CI / Native shell tests (push) Successful in 18m37s

- 修复 write_lock.rs:advisory 复用判据从「本进程持锁」收窄为「同一线程重入或自主流水线」,新增按锁路径登记真实持锁线程的 PROJECT_WRITE_LOCK_THREAD_OWNERS,登记在 create_new 成功处、在 guard Drop 里按路径注销(guard 会被移到别的线程再 Drop)
- 恢复本进程其它线程写通道的串行化:一致快照读、project.diff / action_history、command.output_read、steer 序号分配、项目 revision 侧车、pending sidecar 复核与恢复安装重新等待并保持终态占用
- 调整 tests/project_tools.rs 的 agent_runtime_file_write_lock_failure_redacts_project_path:改为在另一条线程持锁,保持「别的写者持锁时 file.write 失败关闭且脱敏」的断言语义
- 调整 ui_editor/persistence.rs 的 recovery_install_respects_the_project_write_lock:同样改为在另一条线程持锁,保持占用失败断言
- 同步技术方案「2026-09-14 项目客户端占用锁收敛」段落,以及里程碑和实施计划的目标、验收标准、修改顺序、验证命令与未决事项
- 在 decision-log 记录复用判据收窄为同线程重入,并在 pitfalls 记录「同进程复用判据不能只看 pid」的现场、原因、处理与易错点
- 回答 AGC 生命周期文档「同 PID advisory guard 是否放过并行写」的未决项:确认会放过并行写,已收窄

Co-authored-by: DotCraft <273930855+dotcraft-ai@users.noreply.github.com>
This commit is contained in:
2026-09-14 15:41:17 +08:00
parent 7092689b36
commit 4e553bb15d
9 changed files with 138 additions and 27 deletions
@@ -14,16 +14,59 @@ const PROJECT_WRITE_LOCK_UNWRITTEN_GRACE_SECONDS: u64 = 30;
const PROJECT_WRITE_LOCK_PID_REUSE_TOLERANCE_SECONDS: u64 = 5;
const PROJECT_WRITE_LOCK_MAX_BYTES: u64 = 4 * 1024;
/// 本进程内真正落盘持有项目写锁的线程登记表。
///
/// `.agent/project.lock` 的 `pid` 只能证明“锁由本进程的某条写通道持有”,它分不清
/// 两种完全不同的局面:
/// - **同一条调用链再次取锁**:持锁方就是自己,必须放行,否则每次嵌套项目写入都要
/// 白等一个等待预算再报“项目正在被其他写操作占用”;
/// - **本进程另一条写通道正在写**:项目 revision 侧车、steer 序号、一致快照读、
/// pending sidecar 复核和恢复安装都靠这把锁串行化,必须照旧等待。
///
/// 复用判据因此不能停在 `pid`:只有**当前线程**就是真实持锁线程时才返回 advisory
/// guard,本进程其余争用继续走有界等待与终态占用。登记按路径进行、按路径注销:
/// guard 可能被移到别的线程再 Drop(例如写入路径把锁交给阻塞线程池的持有者),
/// 按线程注销会漏项,让后续的重入判断失真。
static PROJECT_WRITE_LOCK_THREAD_OWNERS: std::sync::Mutex<Vec<(PathBuf, std::thread::ThreadId)>> =
std::sync::Mutex::new(Vec::new());
fn project_write_lock_thread_owners(
) -> std::sync::MutexGuard<'static, Vec<(PathBuf, std::thread::ThreadId)>> {
// 登记表只是复用判据的加速器:中毒时继续用内部值,不能让一次取锁失败升级成
// 整个进程再也写不了项目。
PROJECT_WRITE_LOCK_THREAD_OWNERS
.lock()
.unwrap_or_else(|poisoned| poisoned.into_inner())
}
fn register_project_write_lock_thread_owner(path: &Path) {
let mut owners = project_write_lock_thread_owners();
if owners.iter().any(|(owner, _)| owner == path) {
return;
}
owners.push((path.to_path_buf(), std::thread::current().id()));
}
fn unregister_project_write_lock_thread_owner(path: &Path) {
project_write_lock_thread_owners().retain(|(owner, _)| owner != path);
}
/// 当前线程是否就是这条锁路径上真实落盘的持有者(同线程重入)。
fn project_write_lock_reentered_by_current_thread(path: &Path) -> bool {
let thread = std::thread::current().id();
project_write_lock_thread_owners()
.iter()
.any(|(owner, owner_thread)| owner == path && *owner_thread == thread)
}
#[derive(Debug)]
pub(crate) struct ProjectWriteLock {
path: PathBuf,
content: String,
/// In the free-form autonomous lane a single Runtime process may have
/// several specialist actions in flight at once. A file lock is still
/// useful across processes, but making same-process contenders fail turns
/// ordinary parallel work into a dead run (and can deadlock nested tool
/// calls). Such a contender receives an in-process/advisory guard instead
/// of deleting the real holder's lock on drop.
/// 两种“本进程持锁但不必自等”的争用会拿到 advisory guard:同一线程重入(同一条
/// 调用链再次取锁)和自主游戏构建流水线(它有意让并行专家动作同时在飞)。这两种
/// 情况下争用是进程内重叠而不是另一个客户端在改项目,返回的 guard 不拥有
/// `.agent/project.lock`,Drop 时也不得删除真实持有者的锁。
bypassed_same_process: bool,
}
@@ -47,6 +90,7 @@ impl Drop for ProjectWriteLock {
if self.bypassed_same_process {
return;
}
unregister_project_write_lock_thread_owner(&self.path);
if fs::read_to_string(&self.path).is_ok_and(|content| content == self.content) {
let _ = fs::remove_file(&self.path);
}
@@ -815,6 +859,7 @@ pub(crate) fn acquire_project_write_lock_failure(
path.display()
)));
}
register_project_write_lock_thread_owner(&path);
return Ok(ProjectWriteLock {
path,
content: content.clone(),
@@ -860,11 +905,15 @@ pub(crate) fn acquire_project_write_lock_failure(
}
}
}
if project_write_lock_is_owned_by_current_process(&path) {
// A project lock is the client-use lock. Nested calls in
// the same client process must reuse that ownership instead
// of waiting on their own durable marker. Cross-process
// contenders still take the normal retryable path.
if project_write_lock_is_owned_by_current_process(&path)
&& (crate::agent::autonomous_game_build_root_run_active_at(root)
|| project_write_lock_reentered_by_current_thread(&path))
{
// 持锁方就是本进程自己时必须区分重入与并发:同一条调用链(同一
// 线程)再次取锁,以及自主流水线有意并行专家动作,返回 advisory
// guard、不自等、不动真实锁;本进程**其它线程**正在写则继续走
// 有界等待,保住 revision 侧车、steer 序号、一致快照读与恢复安装
// 的串行化。
return Ok(ProjectWriteLock {
path,
content: String::new(),
@@ -5818,8 +5818,27 @@ async fn agent_runtime_file_write_lock_failure_redacts_project_path() {
},
)
.expect("allow direct file write");
let lock = acquire_project_write_lock(&root, "persistent-writer")
.expect("acquire persistent project writer");
// 持锁方必须是**另一条线程**:本用例验证的是“别的写通道正在写时 file.write 必须
// 走满等待预算并失败关闭”,同一条调用链自持锁属于重入复用,不会失败。
let holder_root = root.clone();
let (release_sender, release_receiver) = mpsc::channel::<()>();
let holder = std::thread::spawn(move || {
let lock = acquire_project_write_lock(&holder_root, "persistent-writer")
.expect("acquire persistent project writer");
let _ = release_receiver.recv();
drop(lock);
});
let lock_path = root.join(PROJECT_WRITE_LOCK_PATH);
for _ in 0..400 {
if lock_path.is_file() {
break;
}
std::thread::sleep(Duration::from_millis(5));
}
assert!(
lock_path.is_file(),
"persistent writer must hold the project write lock"
);
let observation = execute_game_creator_agent_runtime_tool_action(
&root,
@@ -5837,7 +5856,8 @@ async fn agent_runtime_file_write_lock_failure_redacts_project_path() {
)
.await;
drop(lock);
let _ = release_sender.send(());
holder.join().expect("join persistent project writer");
assert_eq!(observation.status, "failed");
assert!(!observation
.summary
@@ -1214,8 +1214,28 @@ mod tests {
.expect("resolve primary");
fs::write(&primary, b"{broken").expect("corrupt primary");
let project_lock = acquire_project_write_lock(directory.path(), "test.concurrent-save")
.expect("hold project write lock");
// 持锁方必须是**另一条线程**:本用例验证的是“另一个写者持锁时恢复安装必须失败
// 关闭”,同一条调用链自持锁属于重入复用,不再产生占用失败。
let holder_root = directory.path().to_path_buf();
let (release_sender, release_receiver) = std::sync::mpsc::channel::<()>();
let holder = std::thread::spawn(move || {
let lock = acquire_project_write_lock(&holder_root, "test.concurrent-save")
.expect("hold project write lock");
let _ = release_receiver.recv();
drop(lock);
});
let lock_path = resolve_local_project_path(directory.path(), PROJECT_WRITE_LOCK_PATH)
.expect("resolve project write lock path");
for _ in 0..400 {
if lock_path.is_file() {
break;
}
std::thread::sleep(std::time::Duration::from_millis(5));
}
assert!(
lock_path.is_file(),
"concurrent writer must hold the project write lock"
);
let error = load_ui_design_state_at(LoadUiDesignStateInput {
project_path: directory.path().to_string_lossy().into_owned(),
expected_project_id: PROJECT_ID.to_string(),
@@ -1224,7 +1244,8 @@ mod tests {
.expect_err("recovery must not install while another writer holds the lock");
assert!(error.contains("项目正在被其他写操作占用"));
assert!(read_ui_design_document_path(&primary).is_err());
drop(project_lock);
let _ = release_sender.send(());
holder.join().expect("join concurrent writer");
let recovered = load_ui_design_state_at(LoadUiDesignStateInput {
project_path: directory.path().to_string_lossy().into_owned(),