refactor: improve proxy lifetime management

This commit is contained in:
zhom
2026-08-02 00:11:43 +04:00
parent 4b048f9702
commit be9b892786
8 changed files with 955 additions and 108 deletions
+14 -5
View File
@@ -1763,13 +1763,22 @@ pub fn run_with_builder(
.collect();
for config in list_proxy_configs() {
let has_running_browser = config
.profile_id
.as_ref()
.is_some_and(|pid| running_profile_ids.contains(pid));
// Prefer the owner identity the worker itself recorded: it pins the
// browser to one exact process, so a recycled PID cannot make a dead
// browser look alive and strand this worker for good. Configs written
// before that field existed fall back to the profile's stored PID,
// which is all the information those workers have.
let has_running_browser = if config.browser_pid_start_time.is_some() {
crate::proxy_storage::browser_owner_is_alive(&config)
} else {
config
.profile_id
.as_ref()
.is_some_and(|pid| running_profile_ids.contains(pid))
};
if has_running_browser {
log::info!(
"Startup: preserving proxy worker {} (profile browser still running)",
"Startup: preserving proxy worker {} (browser still running)",
config.id
);
continue;
+20 -2
View File
@@ -1426,8 +1426,17 @@ impl ProfileManager {
if let Err(e) = self.save_profile(&merged) {
log::warn!("Warning: Failed to update profile with new PID: {e}");
}
// Re-point the worker at the browser's NEW identity. update_proxy_pid
// persists it, so a browser that re-execs mid-session doesn't leave
// the detached worker watching a process that no longer exists.
if let Some(prev) = old_pid {
let _ = crate::proxy_manager::PROXY_MANAGER.update_proxy_pid(prev, pid);
} else {
// First sighting this session (e.g. the GUI restarted while the
// browser kept running): nothing to re-key, but the worker still
// needs a live owner recorded or nothing will ever reap it.
crate::proxy_manager::PROXY_MANAGER
.set_browser_pid_for_profile(&merged.id.to_string(), pid);
}
}
} else if merged.process_id.is_some() {
@@ -1493,8 +1502,17 @@ impl ProfileManager {
if let Err(e) = self.save_profile(&latest) {
log::warn!("Warning: Failed to update Wayfern profile with process info: {e}");
}
if let (Some(prev), Some(new)) = (old_pid, wayfern_process.processId) {
let _ = crate::proxy_manager::PROXY_MANAGER.update_proxy_pid(prev, new);
// Same contract as the Camoufox path: the worker reaps itself off
// the identity on disk, so a PID change must reach disk too.
match (old_pid, wayfern_process.processId) {
(Some(prev), Some(new)) => {
let _ = crate::proxy_manager::PROXY_MANAGER.update_proxy_pid(prev, new);
}
(None, Some(new)) => {
crate::proxy_manager::PROXY_MANAGER
.set_browser_pid_for_profile(&latest.id.to_string(), new);
}
_ => {}
}
// Emit profile update event to frontend
+249 -64
View File
@@ -160,6 +160,37 @@ pub fn is_launch_placeholder_pid(pid: u32) -> bool {
pid >= LAUNCH_PLACEHOLDER_PID_MIN
}
/// Write the exact browser process identity (PID plus its start time) onto a
/// worker's on-disk config. Both halves matter: the PID says which process to
/// watch, the start time pins it to THAT process so a recycled PID can never
/// read as "my browser is still alive".
///
/// Rejects launch placeholders and 0 — neither is a real browser — so a caller
/// can treat `false` as "this worker has no verified owner" and abort rather
/// than leave a worker nothing will reap.
pub(crate) fn persist_browser_identity(proxy_id: &str, browser_pid: u32) -> bool {
if browser_pid == 0 || is_launch_placeholder_pid(browser_pid) {
return false;
}
let Some(mut cfg) = crate::proxy_storage::get_proxy_config(proxy_id) else {
return false;
};
let Some(start_time) = crate::proxy_storage::resolve_process_start_time(browser_pid) else {
log::warn!("Could not resolve start time for browser PID {browser_pid} (proxy {proxy_id})");
return false;
};
cfg.browser_pid = Some(browser_pid);
cfg.browser_pid_start_time = Some(start_time);
if crate::proxy_storage::update_proxy_config(&cfg) {
log::info!("Recorded browser PID {browser_pid} on proxy config {proxy_id} for self-reaping");
true
} else {
log::warn!("Failed to persist browser_pid {browser_pid} to proxy config {proxy_id}");
false
}
}
impl StoredProxy {
pub fn new(name: String, proxy_settings: ProxySettings) -> Self {
let sync_enabled = crate::sync::is_sync_configured();
@@ -1154,15 +1185,14 @@ impl ProxyManager {
.map_err(|e| e.to_string());
let ip_result = match proxy_start_result {
Ok(mut proxy_config) => {
Ok(proxy_config) => {
let local_url = format!("http://127.0.0.1:{}", proxy_config.local_port.unwrap_or(0));
let config_id = proxy_config.id.clone();
// Tie the check worker's lifetime to this GUI process: the worker's
// PID watchdog self-exits when browser_pid dies, so if the app is
// owner watchdog self-exits when that identity dies, so if the app is
// killed mid-check the worker follows instead of idling until the
// next app launch.
proxy_config.browser_pid = Some(std::process::id());
if !crate::proxy_storage::update_proxy_config(&proxy_config) {
if !persist_browser_identity(&config_id, std::process::id()) {
log::warn!("Failed to tag check worker {config_id} with app PID for self-expiry");
}
// Wrap in a timeout so the check worker doesn't stay alive indefinitely
@@ -1987,14 +2017,35 @@ impl ProxyManager {
}
// Update the PID mapping for an existing proxy
/// Re-key the in-memory map when a profile's browser PID changes, and rewrite
/// the worker's on-disk owner identity to match.
///
/// Persisting is not optional bookkeeping. The detached worker reaps itself by
/// watching the identity on disk, so a re-key that only touched memory left
/// the worker watching the PREVIOUS process: once that PID was recycled the
/// worker saw a live "browser" forever and outlived both its browser and the
/// GUI. Callers on the status-sync path (`profile::manager`) hit this every
/// time a browser re-execs or restarts itself.
pub fn update_proxy_pid(&self, old_pid: u32, new_pid: u32) -> Result<(), String> {
let mut proxies = self.active_proxies.lock().unwrap();
if let Some(proxy_info) = proxies.remove(&old_pid) {
proxies.insert(new_pid, proxy_info);
Ok(())
} else {
Err(format!("No proxy found for PID {old_pid}"))
let proxy_id = {
let mut proxies = self.active_proxies.lock().unwrap();
match proxies.remove(&old_pid) {
Some(proxy_info) => {
let id = proxy_info.id.clone();
proxies.insert(new_pid, proxy_info);
id
}
None => return Err(format!("No proxy found for PID {old_pid}")),
}
};
if !persist_browser_identity(&proxy_id, new_pid) {
log::warn!(
"Re-keyed proxy {proxy_id} to browser PID {new_pid} in memory but could not persist it; \
the detached worker is still watching the previous process"
);
}
Ok(())
}
/// Persist the real browser PID onto the worker's on-disk config so the
@@ -2010,24 +2061,33 @@ impl ProxyManager {
if browser_pid == 0 {
return false;
}
let proxy_id = {
let map = self.profile_active_proxy_ids.lock().unwrap();
match map.get(profile_id) {
Some(id) => id.clone(),
None => return false,
}
};
let Some(mut cfg) = crate::proxy_storage::get_proxy_config(&proxy_id) else {
let Some(proxy_id) = self.resolve_proxy_id_for_profile(profile_id) else {
return false;
};
cfg.browser_pid = Some(browser_pid);
if crate::proxy_storage::update_proxy_config(&cfg) {
log::info!("Recorded browser PID {browser_pid} on proxy config {proxy_id} for self-reaping");
true
} else {
log::warn!("Failed to persist browser_pid {browser_pid} to proxy config {proxy_id}");
false
persist_browser_identity(&proxy_id, browser_pid)
}
/// Find the worker serving a profile. Prefers the in-memory map, then falls
/// back to the newest matching config on disk: after a GUI restart the map is
/// empty, but a browser (and its worker) launched by the PREVIOUS GUI can
/// still be running, and that worker's owner identity must stay refreshable.
fn resolve_proxy_id_for_profile(&self, profile_id: &str) -> Option<String> {
if let Some(id) = self
.profile_active_proxy_ids
.lock()
.unwrap()
.get(profile_id)
.cloned()
{
return Some(id);
}
// Smallest age = most recently created worker for this profile.
crate::proxy_storage::list_proxy_configs()
.into_iter()
.filter(|config| config.profile_id.as_deref() == Some(profile_id))
.min_by_key(|config| crate::proxy_storage::proxy_config_age_secs(&config.id))
.map(|config| config.id)
}
// Clean up proxies for dead browser processes
@@ -2047,7 +2107,6 @@ impl ProxyManager {
// The user doesn't care if proxy processes run indefinitely as long as they're not consuming CPU
let orphaned_configs = {
use crate::proxy_storage::{is_process_running, list_proxy_configs};
use std::time::{SystemTime, UNIX_EPOCH};
let all_configs = list_proxy_configs();
let tracked_proxy_ids: std::collections::HashSet<String> = {
@@ -2055,12 +2114,6 @@ impl ProxyManager {
proxies.values().map(|p| p.id.clone()).collect()
};
// Get current time for grace period check
let now = SystemTime::now()
.duration_since(UNIX_EPOCH)
.unwrap()
.as_secs();
all_configs
.into_iter()
.filter(|config| {
@@ -2069,15 +2122,9 @@ impl ProxyManager {
return false;
}
// Extract creation time from proxy ID (format: proxy_{timestamp}_{random})
// This gives us a grace period for newly created proxies
let proxy_age = config
.id
.strip_prefix("proxy_")
.and_then(|s| s.split('_').next())
.and_then(|s| s.parse::<u64>().ok())
.map(|created_at| now.saturating_sub(created_at))
.unwrap_or(0);
// Creation time comes from the proxy ID (format: proxy_{timestamp}_{random}),
// giving newly created proxies a grace period.
let proxy_age = crate::proxy_storage::proxy_config_age_secs(&config.id);
// Grace period: don't clean up proxies created in the last 120 seconds
// This prevents race conditions during startup (increased from 60 to 120 for safety)
@@ -2140,12 +2187,6 @@ impl ProxyManager {
// proxies for running browsers (due to launcher-vs-browser PID mismatch).
{
use crate::proxy_storage::{is_process_running, list_proxy_configs};
use std::time::{SystemTime, UNIX_EPOCH};
let now = SystemTime::now()
.duration_since(UNIX_EPOCH)
.unwrap()
.as_secs();
let all_configs = list_proxy_configs();
for config in all_configs {
@@ -2161,13 +2202,7 @@ impl ProxyManager {
}
// Check age: only kill if older than 5 minutes
let proxy_age = config
.id
.strip_prefix("proxy_")
.and_then(|s| s.split('_').next())
.and_then(|s| s.parse::<u64>().ok())
.map(|created_at| now.saturating_sub(created_at))
.unwrap_or(0);
let proxy_age = crate::proxy_storage::proxy_config_age_secs(&config.id);
if proxy_age > 300 {
log::info!(
@@ -2184,12 +2219,13 @@ impl ProxyManager {
// Kill proxy workers whose browser process has died.
//
// active_proxies is keyed by the EXACT browser PID that was recorded in
// update_proxy_pid(). Checking that PID against a single process-table
// snapshot is deterministic: either the PID refers to a live process or
// it doesn't. This avoids the fuzzy launcher-vs-browser detection used
// by check_browser_status (which historically had false negatives on
// Linux and was the reason profile-associated workers were left alone
// in the other cleanup branches).
// update_proxy_pid(). That PID is matched against a single process-table
// snapshot AND, when the worker's config records one, against the browser's
// start time — existence alone would let a recycled PID keep a dead
// browser's worker alive indefinitely. This avoids the fuzzy
// launcher-vs-browser detection used by check_browser_status (which
// historically had false negatives on Linux and was the reason
// profile-associated workers were left alone in the other cleanup branches).
//
// Without this, every time a user closes their browser via the window's
// X button (bypassing Donut's stop flow) or the browser crashes, the
@@ -2228,10 +2264,18 @@ impl ProxyManager {
if browser_pid == 0 || is_launch_placeholder_pid(browser_pid) {
continue;
}
if system
.process(sysinfo::Pid::from_u32(browser_pid))
.is_some()
{
// Reuse the single scan rather than re-querying per PID, but hold the
// entry to the recorded identity when the worker has one.
let expected_start_time = crate::proxy_storage::get_proxy_config(&proxy_id)
.and_then(|config| config.browser_pid_start_time);
let still_the_same_browser = match system.process(sysinfo::Pid::from_u32(browser_pid)) {
Some(process) => expected_start_time.is_none_or(|expected| {
// Same PID, different process: the browser died and its PID was reused.
process.start_time() == expected
}),
None => false,
};
if still_the_same_browser {
alive_pids.push(browser_pid);
} else {
dead_candidates.push((browser_pid, proxy_id, profile_id));
@@ -2845,6 +2889,143 @@ mod tests {
assert_eq!(info.profile_id.as_deref(), Some("prof_a"));
}
/// Save a worker config for a profile in an isolated cache dir and return its id.
fn saved_worker_config(profile_id: Option<&str>) -> String {
let id = crate::proxy_storage::generate_proxy_id();
let config = crate::proxy_storage::ProxyConfig::new(id.clone(), "DIRECT".to_string(), Some(0))
.with_profile_id(profile_id.map(str::to_string));
crate::proxy_storage::save_proxy_config(&config).unwrap();
id
}
#[test]
fn browser_identity_records_the_pid_and_its_start_time() {
let temp = tempfile::tempdir().unwrap();
let _cache_guard = crate::app_dirs::set_test_cache_dir(temp.path().to_path_buf());
let id = saved_worker_config(Some("prof_identity"));
let pid = std::process::id();
assert!(persist_browser_identity(&id, pid));
let saved = crate::proxy_storage::get_proxy_config(&id).unwrap();
assert_eq!(saved.browser_pid, Some(pid));
assert_eq!(
saved.browser_pid_start_time,
crate::proxy_storage::process_start_time(pid),
"the start time must pin the PID to this exact process"
);
assert!(crate::proxy_storage::browser_owner_is_alive(&saved));
}
#[test]
fn browser_identity_refuses_owners_that_are_not_real_browsers() {
let temp = tempfile::tempdir().unwrap();
let _cache_guard = crate::app_dirs::set_test_cache_dir(temp.path().to_path_buf());
let id = saved_worker_config(Some("prof_reject"));
// 0 and launch placeholders mean "no browser reported yet". Recording one
// would leave a worker whose owner can never die, so it must fail loudly
// enough for the caller to abort the launch.
assert!(!persist_browser_identity(&id, 0));
assert!(!persist_browser_identity(
&id,
next_launch_placeholder_pid()
));
// A PID with no process behind it can't be pinned to an identity either.
assert!(!persist_browser_identity(&id, u32::MAX));
// An unknown worker is not silently created.
assert!(!persist_browser_identity(
"proxy_1_missing",
std::process::id()
));
let saved = crate::proxy_storage::get_proxy_config(&id).unwrap();
assert_eq!(saved.browser_pid, None);
assert_eq!(saved.browser_pid_start_time, None);
}
/// The regression behind the orphaned-worker reports: the status synchronizer
/// re-keys a profile's browser PID whenever the browser re-execs, and that
/// change has to reach the worker's config. When it only landed in memory the
/// detached worker kept watching the PREVIOUS process — and once that PID was
/// recycled it saw a live "browser" forever and outlived everything.
#[test]
fn remapping_a_browser_pid_rewrites_the_workers_on_disk_owner() {
let temp = tempfile::tempdir().unwrap();
let _cache_guard = crate::app_dirs::set_test_cache_dir(temp.path().to_path_buf());
let id = saved_worker_config(Some("prof_remap"));
let pm = ProxyManager::new();
pm.insert_active_proxy(100, make_proxy_info(&id, 9010, Some("prof_remap")));
let stale_start_time = 1;
let mut seeded = crate::proxy_storage::get_proxy_config(&id).unwrap();
seeded.browser_pid = Some(100);
seeded.browser_pid_start_time = Some(stale_start_time);
assert!(crate::proxy_storage::update_proxy_config(&seeded));
let live_pid = std::process::id();
pm.update_proxy_pid(100, live_pid).unwrap();
let saved = crate::proxy_storage::get_proxy_config(&id).unwrap();
assert_eq!(saved.browser_pid, Some(live_pid));
assert_ne!(saved.browser_pid_start_time, Some(stale_start_time));
assert!(
crate::proxy_storage::browser_owner_is_alive(&saved),
"the worker must now be watching the browser that is actually running"
);
}
/// After a GUI restart the profile→proxy map is empty, but a browser launched
/// by the previous GUI can still be running with its worker attached. That
/// worker's owner still has to be refreshable, so the lookup falls back to disk.
#[test]
fn the_worker_for_a_profile_is_found_on_disk_when_memory_is_empty() {
let temp = tempfile::tempdir().unwrap();
let _cache_guard = crate::app_dirs::set_test_cache_dir(temp.path().to_path_buf());
let id = saved_worker_config(Some("prof_restart"));
let _other = saved_worker_config(Some("prof_unrelated"));
let pm = ProxyManager::new();
assert_eq!(
pm.resolve_proxy_id_for_profile("prof_restart").as_deref(),
Some(id.as_str())
);
assert!(pm.resolve_proxy_id_for_profile("prof_absent").is_none());
let live_pid = std::process::id();
assert!(pm.set_browser_pid_for_profile("prof_restart", live_pid));
let saved = crate::proxy_storage::get_proxy_config(&id).unwrap();
assert_eq!(saved.browser_pid, Some(live_pid));
assert!(crate::proxy_storage::browser_owner_is_alive(&saved));
}
#[test]
fn the_in_memory_mapping_wins_over_the_on_disk_scan() {
let temp = tempfile::tempdir().unwrap();
let _cache_guard = crate::app_dirs::set_test_cache_dir(temp.path().to_path_buf());
let stale = saved_worker_config(Some("prof_both"));
let current = saved_worker_config(Some("prof_both"));
let pm = ProxyManager::new();
pm.profile_active_proxy_ids
.lock()
.unwrap()
.insert("prof_both".to_string(), current.clone());
assert_eq!(
pm.resolve_proxy_id_for_profile("prof_both").as_deref(),
Some(current.as_str())
);
assert!(pm.set_browser_pid_for_profile("prof_both", std::process::id()));
assert_eq!(
crate::proxy_storage::get_proxy_config(&stale)
.unwrap()
.browser_pid,
None,
"only the tracked worker may be re-pointed"
);
}
#[test]
fn test_update_proxy_pid_error_for_unknown_pid() {
let pm = ProxyManager::new();
@@ -3057,6 +3238,7 @@ mod tests {
dns_allowlist_mode: false,
local_protocol: None,
browser_pid: None,
browser_pid_start_time: None,
};
let dead_config = ProxyConfig {
id: dead_id.clone(),
@@ -3071,6 +3253,7 @@ mod tests {
dns_allowlist_mode: false,
local_protocol: None,
browser_pid: None,
browser_pid_start_time: None,
};
save_proxy_config(&live_config).unwrap();
@@ -3113,6 +3296,7 @@ mod tests {
dns_allowlist_mode: false,
local_protocol: None,
browser_pid: None,
browser_pid_start_time: None,
};
// Save
@@ -3540,6 +3724,7 @@ mod tests {
dns_allowlist_mode: false,
local_protocol: None,
browser_pid: None,
browser_pid_start_time: None,
};
save_proxy_config(&config).unwrap();
+51 -24
View File
@@ -1562,11 +1562,12 @@ pub async fn run_proxy_server(config: ProxyConfig) -> Result<(), Box<dyn std::er
// Self-reaping supervisor. The worker is a detached process that outlives the
// GUI, so it cannot rely on the GUI's in-memory death-monitor (which is lost
// when the GUI restarts). Once the GUI records the browser PID this worker
// serves, poll it and exit when that browser is gone — never while it is
// alive, and never before a PID is recorded (covers the launch window and
// pre-upgrade configs lacking the field). A 2-miss debounce avoids exiting on
// a transient sysinfo false-negative under load / sleep-wake.
// when the GUI restarts). Once the GUI records the browser PID and start time
// this worker serves, poll that exact process identity and exit when it is
// gone — never while it is alive. A 2-miss debounce avoids exiting on a
// transient sysinfo false-negative under load / sleep-wake. The decision table
// itself lives in `proxy_storage::supervisor_verdict` so every branch is unit
// tested without spawning browsers.
//
// This runs on a DEDICATED OS THREAD, not a tokio task. If the worker's
// accept/dial path ever busy-loops (e.g. a client retry-storm against a
@@ -1578,29 +1579,39 @@ pub async fn run_proxy_server(config: ProxyConfig) -> Result<(), Box<dyn std::er
// reaps itself. Every call here is synchronous and safe off the runtime.
{
let watch_id = config.id.clone();
let poll_interval = watchdog_poll_interval();
std::thread::spawn(move || {
use crate::proxy_storage::SupervisorVerdict;
let mut consecutive_misses: u32 = 0;
loop {
std::thread::sleep(std::time::Duration::from_secs(15));
match crate::proxy_storage::get_proxy_config(&watch_id) {
Some(cfg) => match cfg.browser_pid {
Some(bpid) if bpid != 0 => {
if crate::proxy_storage::is_process_running(bpid) {
consecutive_misses = 0;
} else {
consecutive_misses += 1;
if consecutive_misses >= 2 {
log::info!("Browser PID {bpid} for config {watch_id} is gone; worker exiting");
crate::proxy_storage::delete_proxy_config(&watch_id);
std::process::exit(0);
}
}
std::thread::sleep(poll_interval);
let cfg = crate::proxy_storage::get_proxy_config(&watch_id);
let verdict = crate::proxy_storage::supervisor_verdict(
cfg.as_ref(),
crate::proxy_storage::proxy_config_age_secs(&watch_id),
crate::proxy_storage::browser_owner_is_alive,
);
match verdict {
SupervisorVerdict::Keep => consecutive_misses = 0,
SupervisorVerdict::ExitOwnerGone => {
consecutive_misses += 1;
if consecutive_misses >= 2 {
let owner = cfg.as_ref().and_then(|c| c.browser_pid).unwrap_or(0);
log::info!("Browser PID {owner} for config {watch_id} is gone; worker exiting");
crate::proxy_storage::delete_proxy_config(&watch_id);
std::process::exit(0);
}
// No browser PID recorded yet (launch window / old config): keep running.
_ => consecutive_misses = 0,
},
// Our own config was removed (e.g. GUI stopped us): nothing to serve.
None => {
}
SupervisorVerdict::ExitNeverClaimed => {
log::info!(
"Config {watch_id} was never claimed by a browser within the launch window; worker exiting"
);
crate::proxy_storage::delete_proxy_config(&watch_id);
std::process::exit(0);
}
SupervisorVerdict::ExitConfigRemoved => {
log::info!("Proxy config {watch_id} was removed; worker exiting");
std::process::exit(0);
}
@@ -1751,6 +1762,22 @@ async fn handle_connect_from_buffer(
Ok(())
}
/// How often the self-reaping supervisor re-checks its owner. Overridable via
/// `DONUT_PROXY_WATCHDOG_INTERVAL_MS` so lifecycle tests can observe a real
/// worker reaping itself in seconds instead of a minute; the floor keeps a
/// mistyped value from turning the supervisor into a spin loop.
const WATCHDOG_POLL_INTERVAL: std::time::Duration = std::time::Duration::from_secs(15);
const WATCHDOG_POLL_INTERVAL_FLOOR: std::time::Duration = std::time::Duration::from_millis(100);
fn watchdog_poll_interval() -> std::time::Duration {
std::env::var("DONUT_PROXY_WATCHDOG_INTERVAL_MS")
.ok()
.and_then(|raw| raw.trim().parse::<u64>().ok())
.map(std::time::Duration::from_millis)
.map(|interval| interval.max(WATCHDOG_POLL_INTERVAL_FLOOR))
.unwrap_or(WATCHDOG_POLL_INTERVAL)
}
/// Upper bound on concurrent connection handlers per worker. A real browser
/// never holds anywhere near this many simultaneous tunnels; the cap stops a
/// client retry-storm from spawning unbounded tasks (each of which parks a
+229 -1
View File
@@ -29,9 +29,16 @@ pub struct ProxyConfig {
/// launch. The detached worker watches this and self-terminates when the
/// browser dies, so it dies with its browser even if the GUI has exited or
/// restarted. `None` until launch completes (the worker keeps running while
/// it is `None`).
/// it is `None`, up to `UNCLAIMED_WORKER_GRACE_SECS`).
#[serde(default)]
pub browser_pid: Option<u32>,
/// Start time of `browser_pid`, pinning it to one exact process. Without it a
/// recycled PID reads as "my browser is still alive" and the worker outlives
/// its browser forever — the orphan users report. `None` on configs written
/// before this field existed; those fall back to a bare existence check so an
/// upgrade never reaps a worker whose browser is still running.
#[serde(default)]
pub browser_pid_start_time: Option<u64>,
}
impl ProxyConfig {
@@ -49,6 +56,7 @@ impl ProxyConfig {
dns_allowlist_mode: false,
local_protocol: None,
browser_pid: None,
browser_pid_start_time: None,
}
}
@@ -222,6 +230,98 @@ pub fn process_identity_matches(pid: u32, expected_start_time: Option<u64>) -> b
expected_start_time.is_some_and(|expected| process_start_time(pid) == Some(expected))
}
/// Read a just-spawned process's start time, retrying briefly. A process is not
/// always visible in the process table the instant `spawn` returns, and giving
/// up would leave the caller unable to pin the PID to an identity.
pub fn resolve_process_start_time(pid: u32) -> Option<u64> {
for _ in 0..25 {
if let Some(start_time) = process_start_time(pid) {
return Some(start_time);
}
std::thread::sleep(std::time::Duration::from_millis(10));
}
None
}
/// Seconds since a worker config's id was minted. Ids are
/// `proxy_{unix_secs}_{rand}` (see `generate_proxy_id`); anything else, or a
/// timestamp in the future, reads as age 0 so callers stay conservative.
pub fn proxy_config_age_secs(id: &str) -> u64 {
let Ok(now) = std::time::SystemTime::now().duration_since(std::time::UNIX_EPOCH) else {
return 0;
};
id.strip_prefix("proxy_")
.and_then(|rest| rest.split('_').next())
.and_then(|secs| secs.parse::<u64>().ok())
.map(|created_at| now.as_secs().saturating_sub(created_at))
.unwrap_or(0)
}
/// How long a worker may run without the GUI claiming it by recording a browser
/// PID. Covers the launch window (worker starts before the browser); past it the
/// GUI that spawned this worker is gone and nothing will ever claim it.
pub const UNCLAIMED_WORKER_GRACE_SECS: u64 = 300;
/// Is the browser this worker serves still the same live process?
///
/// Identity-checked whenever a start time was recorded. Configs written before
/// `browser_pid_start_time` existed fall back to a bare existence check, so
/// upgrading never reaps a worker whose browser is still running.
pub fn browser_owner_is_alive(config: &ProxyConfig) -> bool {
let Some(pid) = config.browser_pid.filter(|pid| *pid != 0) else {
return false;
};
match config.browser_pid_start_time {
Some(start_time) => process_identity_matches(pid, Some(start_time)),
None => is_process_running(pid),
}
}
/// What the detached worker's self-reaping supervisor should do this tick.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum SupervisorVerdict {
/// Keep serving.
Keep,
/// Our config is gone — the GUI stopped us, or the worker was superseded.
ExitConfigRemoved,
/// The browser we were started for is gone (debounced by the caller).
ExitOwnerGone,
/// No browser was ever recorded and the launch window has long passed.
ExitNeverClaimed,
}
/// Pure decision table for the worker supervisor. `owner_alive` is injected so
/// every branch is testable without spawning browsers; production passes
/// `browser_owner_is_alive`.
pub fn supervisor_verdict(
config: Option<&ProxyConfig>,
age_secs: u64,
owner_alive: impl Fn(&ProxyConfig) -> bool,
) -> SupervisorVerdict {
let Some(config) = config else {
return SupervisorVerdict::ExitConfigRemoved;
};
match config.browser_pid {
Some(pid) if pid != 0 => {
if owner_alive(config) {
SupervisorVerdict::Keep
} else {
SupervisorVerdict::ExitOwnerGone
}
}
// Never claimed. Keep serving through the launch window — the browser may
// still be starting — then give up rather than idling forever.
_ => {
if age_secs >= UNCLAIMED_WORKER_GRACE_SECS {
SupervisorVerdict::ExitNeverClaimed
} else {
SupervisorVerdict::Keep
}
}
}
}
#[cfg(test)]
mod tests {
use super::*;
@@ -298,6 +398,134 @@ mod tests {
);
}
fn owned_config(browser_pid: Option<u32>, browser_pid_start_time: Option<u64>) -> ProxyConfig {
let mut config = ProxyConfig::new("proxy_1_2".to_string(), "DIRECT".to_string(), Some(1080));
config.browser_pid = browser_pid;
config.browser_pid_start_time = browser_pid_start_time;
config
}
#[test]
fn owner_liveness_is_pinned_to_the_exact_process_that_was_recorded() {
let pid = std::process::id();
let start_time = process_start_time(pid).expect("current process should be visible");
assert!(browser_owner_is_alive(&owned_config(
Some(pid),
Some(start_time)
)));
// Same PID, different process: what a recycled PID looks like. Treating
// this as alive is what stranded workers forever.
assert!(!browser_owner_is_alive(&owned_config(
Some(pid),
Some(start_time.saturating_add(1))
)));
// Written before start times were recorded: existence is all we have, and
// an upgrade must not reap a worker whose browser is still running.
assert!(browser_owner_is_alive(&owned_config(Some(pid), None)));
assert!(!browser_owner_is_alive(&owned_config(
Some(u32::MAX),
Some(start_time)
)));
assert!(!browser_owner_is_alive(&owned_config(None, None)));
assert!(!browser_owner_is_alive(&owned_config(Some(0), None)));
}
#[test]
fn supervisor_keeps_serving_a_live_owner_and_exits_a_dead_one() {
let alive = |_: &ProxyConfig| true;
let dead = |_: &ProxyConfig| false;
let claimed = owned_config(Some(4321), Some(99));
assert_eq!(
supervisor_verdict(Some(&claimed), 0, alive),
SupervisorVerdict::Keep
);
assert_eq!(
supervisor_verdict(Some(&claimed), 0, dead),
SupervisorVerdict::ExitOwnerGone
);
// A live owner outranks age: a long-running browser is not an orphan.
assert_eq!(
supervisor_verdict(Some(&claimed), UNCLAIMED_WORKER_GRACE_SECS * 10, alive),
SupervisorVerdict::Keep
);
}
#[test]
fn supervisor_exits_when_its_config_is_gone() {
assert_eq!(
supervisor_verdict(None, 0, |_| true),
SupervisorVerdict::ExitConfigRemoved
);
}
#[test]
fn supervisor_gives_up_on_a_worker_no_browser_ever_claimed() {
// The GUI records the owner right after launch. Inside the window the
// browser may still be starting, so keep serving...
for unclaimed in [owned_config(None, None), owned_config(Some(0), None)] {
assert_eq!(
supervisor_verdict(Some(&unclaimed), 0, |_| false),
SupervisorVerdict::Keep
);
assert_eq!(
supervisor_verdict(
Some(&unclaimed),
UNCLAIMED_WORKER_GRACE_SECS.saturating_sub(1),
|_| false
),
SupervisorVerdict::Keep
);
// ...past it the GUI that spawned this worker is gone and nothing will
// ever claim it, so it must not idle forever.
assert_eq!(
supervisor_verdict(Some(&unclaimed), UNCLAIMED_WORKER_GRACE_SECS, |_| false),
SupervisorVerdict::ExitNeverClaimed
);
}
}
#[test]
fn config_age_is_read_from_the_generated_id() {
let fresh = generate_proxy_id();
assert!(
proxy_config_age_secs(&fresh) <= 1,
"a just-generated id should read as brand new, got {}",
proxy_config_age_secs(&fresh)
);
let now = std::time::SystemTime::now()
.duration_since(std::time::UNIX_EPOCH)
.unwrap()
.as_secs();
assert_eq!(
proxy_config_age_secs(&format!("proxy_{}_123", now.saturating_sub(600))),
600
);
// Unparsable and future-dated ids read as brand new so callers stay
// conservative rather than reaping something they can't date.
assert_eq!(proxy_config_age_secs("not-a-proxy-id"), 0);
assert_eq!(proxy_config_age_secs("proxy_abc_123"), 0);
assert_eq!(proxy_config_age_secs(&format!("proxy_{}_1", now + 600)), 0);
}
#[test]
fn configs_written_before_owner_start_times_still_load() {
let legacy = serde_json::json!({
"id": "proxy_1_2",
"upstream_url": "DIRECT",
"local_port": 1080,
"ignore_proxy_certificate": null,
"local_url": "http://127.0.0.1:1080",
"pid": 42,
"browser_pid": 4242
});
let config: ProxyConfig = serde_json::from_value(legacy).unwrap();
assert_eq!(config.browser_pid, Some(4242));
assert_eq!(config.browser_pid_start_time, None);
}
#[test]
fn test_is_process_running_returns_false_for_nonexistent_pid() {
// PID 0 is the "System Idle Process" on Windows and sysinfo reports it as running,
+2 -11
View File
@@ -1,7 +1,7 @@
use crate::proxy_runner::find_sidecar_executable;
#[cfg(unix)]
use crate::proxy_storage::is_process_running;
use crate::proxy_storage::{process_identity_matches, process_start_time};
use crate::proxy_storage::{process_identity_matches, resolve_process_start_time};
use crate::xray::{build_client_config_json, parse_vless_uri, XrayClientRuntime};
use crate::xray_worker_storage::{
create_xray_worker_log, delete_xray_worker_config, generate_xray_worker_id,
@@ -21,16 +21,6 @@ const READY_CHECK_TIMEOUT: Duration = Duration::from_millis(750);
static XRAY_BINARY_VERIFIED: AtomicBool = AtomicBool::new(false);
static XRAY_START_LOCK: tokio::sync::Mutex<()> = tokio::sync::Mutex::const_new(());
fn resolve_process_start_time(pid: u32) -> Option<u64> {
for _ in 0..25 {
if let Some(start_time) = process_start_time(pid) {
return Some(start_time);
}
std::thread::sleep(Duration::from_millis(10));
}
None
}
fn structured_error(code: &str) -> Box<dyn std::error::Error> {
serde_json::json!({ "code": code }).to_string().into()
}
@@ -659,6 +649,7 @@ pub async fn run_xray_worker(config_path: &Path) -> Result<(), Box<dyn std::erro
#[cfg(test)]
mod tests {
use super::*;
use crate::proxy_storage::process_start_time;
#[cfg(unix)]
fn short_lived_child() -> Child {