geph5-app 0.3.10

The Geph5 desktop app: the `geph` CLI and its supervising daemon
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
//! Windows full-tunnel VPN backend: a manager-owned WinTUN device, a `/1` split default
//! route into it, a DNS sentinel, a fail-closed kill switch (see [`firewall`]),
//! and a stdio packet pump between the device and the engine child.
//!
//! All host network configuration — the tun addresses, the split-default routes,
//! and physical-interface discovery — is done in-process through the IP Helper
//! API (`iphlpapi`) keyed by the WinTUN interface LUID, *not* by shelling out to
//! `netsh`/`powershell`. Each of those is a process spawn costing hundreds of
//! milliseconds, and a server switch reasserts the whole set under the manager
//! lock; doing it via direct syscalls keeps the reconnect gap short (the same
//! near-instant reconfiguration Linux/macOS get from netlink / route sockets).
//!
//! Loop prevention is entirely per-process: the engine binds its own outbound
//! sockets to the physical interface via `IP_UNICAST_IF` (driven by the
//! `GEPH_VPN_BIND_IF4/6` env vars the manager sets — see `geph5-client`'s
//! `bound_dialer`), so its bridge/exit traffic leaves the real NIC. The broker's
//! own HTTP clients (`reqwest`/`aws-sdk`) connect through an in-process loopback
//! forwarder whose upstream is dialed the same way (see `geph5-client`'s
//! `broker::bind_forward`). Loop prevention is therefore strictly per-process: any
//! other app reaching the same (intentionally shared, domain-fronted) broker IP
//! still routes into the tunnel.
//!
//! Unlike Linux (where the engine reads the tun fd directly), the engine has no
//! native Windows tun ingestion, so the manager owns the WinTUN device and pumps
//! packets to `geph5-client --stdio-vpn` over its stdin/stdout.

mod firewall;

use std::{
    ffi::c_void,
    io::{Read, Write},
    net::{IpAddr, Ipv4Addr, Ipv6Addr},
    process::{ChildStdin, ChildStdout},
    sync::{
        Arc, Condvar, Mutex,
        atomic::{AtomicBool, Ordering},
    },
    thread::JoinHandle,
};

use anyhow::{Context, bail};
use windows_sys::Win32::{
    Devices::DeviceAndDriverInstallation::{
        CM_LOCATE_DEVNODE_PHANTOM, CM_Locate_DevNodeW, CM_NOTIFY_ACTION_DEVICEINSTANCEREMOVED,
        CM_NOTIFY_FILTER, CM_NOTIFY_FILTER_0, CM_NOTIFY_FILTER_0_1,
        CM_NOTIFY_FILTER_TYPE_DEVICEINSTANCE, CM_Register_Notification, CM_Uninstall_DevNode,
        CM_Unregister_Notification, CR_NO_SUCH_DEVNODE, CR_SUCCESS, HCMNOTIFICATION,
    },
    Foundation::{
        ERROR_BUFFER_OVERFLOW, ERROR_NOT_FOUND, ERROR_OBJECT_ALREADY_EXISTS, ERROR_SUCCESS,
        NO_ERROR,
    },
    NetworkManagement::IpHelper::{
        ConvertInterfaceGuidToLuid, CreateIpForwardEntry2, CreateUnicastIpAddressEntry,
        DNS_INTERFACE_SETTINGS, DNS_INTERFACE_SETTINGS_VERSION1, DNS_SETTING_IPV6,
        DNS_SETTING_NAMESERVER, DeleteIpForwardEntry2, FreeMibTable, GAA_FLAG_SKIP_ANYCAST,
        GAA_FLAG_SKIP_MULTICAST, GAA_FLAG_SKIP_UNICAST, GetAdaptersAddresses, GetIpForwardTable2,
        IP_ADAPTER_ADDRESSES_LH, InitializeIpForwardEntry, InitializeUnicastIpAddressEntry,
        MIB_IPFORWARD_ROW2, MIB_IPFORWARD_TABLE2, MIB_UNICASTIPADDRESS_ROW,
        SetInterfaceDnsSettings,
    },
    Networking::WinSock::{
        ADDRESS_FAMILY, AF_INET, AF_INET6, AF_UNSPEC, IN_ADDR, IN_ADDR_0, IN6_ADDR, IN6_ADDR_0,
        SOCKADDR, SOCKADDR_IN, SOCKADDR_IN6, SOCKADDR_INET,
    },
};
use wintun::{Adapter, Session, Wintun};

use firewall::{Firewall, WfpKillSwitch};

/// WinTUN adapter alias (also the interface name shown in the OS).
const TUN_NAME: &str = "Geph";
/// Fixed adapter GUID so we recognise our own device across runs.
const TUN_GUID: u128 = 0x6765_7068_0000_0000_0000_0000_0000_0001;
const TUN_DEVICE_INSTANCE_ID: &str = r"SWD\WINTUN\{67657068-0000-0000-0000-000000000001}";

const TUN_V4_ADDR: Ipv4Addr = Ipv4Addr::new(100, 64, 0, 1);
const TUN_V4_PREFIX: u8 = 10; // 255.192.0.0
const TUN_V6_ADDR: Ipv6Addr = Ipv6Addr::new(0xfd00, 0x6765, 0, 0, 0, 0, 0, 1);
const TUN_V6_PREFIX: u8 = 64;
const TUN_MTU: usize = 16384;

/// Ring-buffer capacity for the WinTUN session (4 MiB; a power of two between
/// wintun's MIN and MAX capacities).
const RING_CAPACITY: u32 = 0x40_0000;

/// The `/1` split that captures all traffic into WinTUN without deleting the
/// physical default route (`redirect-gateway def1`). These are `/1`, never `/0`,
/// so physical-interface discovery (which looks only at exact default routes)
/// never mistakes them for the real default route.
const V4_SPLIT: [(Ipv4Addr, u8); 2] = [
    (Ipv4Addr::new(0, 0, 0, 0), 1),
    (Ipv4Addr::new(128, 0, 0, 0), 1),
];
const V6_SPLIT: [(Ipv6Addr, u8); 2] = [
    (Ipv6Addr::new(0, 0, 0, 0, 0, 0, 0, 0), 1),
    (Ipv6Addr::new(0x8000, 0, 0, 0, 0, 0, 0, 0), 1),
];

fn sentinel_dns() -> [IpAddr; 2] {
    [
        IpAddr::V4(Ipv4Addr::new(1, 1, 1, 1)),
        IpAddr::V6(Ipv6Addr::new(0x2606, 0x4700, 0x4700, 0, 0, 0, 0, 0x1111)),
    ]
}

/// The physical default-route interface(s) the engine's own traffic must use
/// (passed to the engine as `GEPH_VPN_BIND_IF4/6` for `IP_UNICAST_IF`).
#[derive(Clone, Debug)]
pub(super) struct PhysIface {
    index4: u32,
    index6: u32,
}

impl PhysIface {
    pub(super) fn bind_indices(&self) -> (u32, u32) {
        (self.index4, self.index6)
    }
}

/// Persistent VPN state, held by the manager across engine-child restarts: the
/// WinTUN adapter (keeps the device + its `/1` routes alive), the kill switch,
/// and enough state to tear routing back down. The packet pump is *not* here —
/// it is re-created per child (see [`Pump`]).
pub(super) struct VpnHandle {
    // `wintun` must outlive `adapter` (the adapter borrows the loaded DLL); keep
    // it alive for the handle's lifetime even though we don't touch it again.
    #[allow(dead_code)]
    wintun: Wintun,
    adapter: Arc<Adapter>,
    phys: PhysIface,
    firewall: WfpKillSwitch,
}

struct AdapterRemovalContext {
    removed: Mutex<bool>,
    changed: Condvar,
}

/// A native PnP notification registered before hard task termination. Waiting
/// for the actual device-instance removal avoids both an arbitrary delay and the
/// WinTUN create-vs-removal race.
pub(crate) struct AdapterRemovalNotification {
    notification: HCMNOTIFICATION,
    context: Box<AdapterRemovalContext>,
}

impl AdapterRemovalNotification {
    pub(crate) fn prepare() -> anyhow::Result<Option<Self>> {
        unsafe extern "system" fn callback(
            _notification: HCMNOTIFICATION,
            context: *const c_void,
            action: i32,
            _event_data: *const windows_sys::Win32::Devices::DeviceAndDriverInstallation::CM_NOTIFY_EVENT_DATA,
            _event_data_size: u32,
        ) -> u32 {
            if action == CM_NOTIFY_ACTION_DEVICEINSTANCEREMOVED {
                let context = unsafe { &*(context.cast::<AdapterRemovalContext>()) };
                let mut removed = context.removed.lock().unwrap();
                *removed = true;
                context.changed.notify_all();
            }
            ERROR_SUCCESS
        }

        let instance_id: Vec<u16> = TUN_DEVICE_INSTANCE_ID
            .encode_utf16()
            .chain(std::iter::once(0))
            .collect();
        let mut device = 0;
        match unsafe { CM_Locate_DevNodeW(&mut device, instance_id.as_ptr(), 0) } {
            CR_SUCCESS => {}
            CR_NO_SUCH_DEVNODE => return Ok(None),
            error => anyhow::bail!("locating the WinTUN device instance failed ({error})"),
        }

        let mut filter: CM_NOTIFY_FILTER = unsafe { std::mem::zeroed() };
        filter.cbSize = std::mem::size_of::<CM_NOTIFY_FILTER>() as u32;
        filter.FilterType = CM_NOTIFY_FILTER_TYPE_DEVICEINSTANCE;
        let mut filter_instance_id = [0u16; 200];
        filter_instance_id[..instance_id.len()].copy_from_slice(&instance_id);
        filter.u = CM_NOTIFY_FILTER_0 {
            DeviceInstance: CM_NOTIFY_FILTER_0_1 {
                InstanceId: filter_instance_id,
            },
        };

        let context = Box::new(AdapterRemovalContext {
            removed: Mutex::new(false),
            changed: Condvar::new(),
        });
        let mut notification = std::ptr::null_mut();
        let result = unsafe {
            CM_Register_Notification(
                &filter,
                std::ptr::from_ref(context.as_ref()).cast(),
                Some(callback),
                &mut notification,
            )
        };
        if result != CR_SUCCESS {
            if unsafe { CM_Locate_DevNodeW(&mut device, instance_id.as_ptr(), 0) }
                == CR_NO_SUCH_DEVNODE
            {
                return Ok(None);
            }
            anyhow::bail!("registering WinTUN removal notification failed ({result})");
        }
        let registration = Self {
            notification,
            context,
        };

        // Check again after registering so a device removed concurrently with
        // registration cannot leave this process waiting for a missed event.
        match unsafe { CM_Locate_DevNodeW(&mut device, instance_id.as_ptr(), 0) } {
            CR_SUCCESS => Ok(Some(registration)),
            CR_NO_SUCH_DEVNODE => Ok(None),
            error => anyhow::bail!("locating the WinTUN device instance failed ({error})"),
        }
    }

    pub(crate) fn wait(&self) -> anyhow::Result<()> {
        let mut removed = self.context.removed.lock().unwrap();
        while !*removed {
            removed = self.context.changed.wait(removed).unwrap();
        }
        drop(removed);

        uninstall_orphaned_adapter()
    }
}

fn uninstall_orphaned_adapter() -> anyhow::Result<()> {
    // A process killed while owning an HSWDEVICE cannot run
    // WintunCloseAdapter's device-uninstall path. PnP first reports the
    // device as removed but retains a phantom devnode whose requested
    // network GUID still collides with immediate recreation. Remove that
    // nonpresent instance's persistent state before reusing the GUID.
    let instance_id: Vec<u16> = TUN_DEVICE_INSTANCE_ID
        .encode_utf16()
        .chain(std::iter::once(0))
        .collect();
    let mut device = 0;
    match unsafe {
        CM_Locate_DevNodeW(&mut device, instance_id.as_ptr(), CM_LOCATE_DEVNODE_PHANTOM)
    } {
        CR_SUCCESS => {}
        CR_NO_SUCH_DEVNODE => return Ok(()),
        error => anyhow::bail!("locating the removed WinTUN device instance failed ({error})"),
    }
    let result = unsafe { CM_Uninstall_DevNode(device, 0) };
    if result != CR_SUCCESS {
        anyhow::bail!("uninstalling the orphaned WinTUN device failed ({result})");
    }
    Ok(())
}

impl Drop for AdapterRemovalNotification {
    fn drop(&mut self) {
        unsafe {
            let _ = CM_Unregister_Notification(self.notification);
        }
    }
}

impl VpnHandle {
    /// The physical interface indices the engine child should pin its sockets to
    /// (passed through as `GEPH_VPN_BIND_IF4/6`).
    pub(super) fn bind_indices(&self) -> (u32, u32) {
        (self.phys.index4, self.phys.index6)
    }

    /// The physical adapter's own DNS resolvers (its DHCP/ISP servers), for the
    /// engine to resolve over the physical NIC via `GEPH_PHYS_DNS`.
    pub(super) fn phys_dns(&self) -> Vec<IpAddr> {
        physical_dns_servers(&[self.phys.index4, self.phys.index6])
    }

    /// The WinTUN interface LUID, the key every IP Helper call below is scoped to.
    fn luid(&self) -> u64 {
        unsafe { self.adapter.get_luid().Value }
    }

    /// Open a fresh WinTUN session on the (persistent) adapter for a newly-spawned
    /// engine child's pump.
    pub(super) fn start_session(&self) -> anyhow::Result<Arc<Session>> {
        Ok(Arc::new(
            self.adapter
                .start_session(RING_CAPACITY)
                .context("start wintun session")?,
        ))
    }

    fn cleanup(&mut self) {
        self.firewall.remove();
        let _ = set_tun_dns(self.adapter.get_guid(), &[]);
        delete_split_routes(self.luid());
    }
}

/// Tear down a live VPN and remove its routes, DNS override, and kill switch.
pub(super) fn cleanup(mut handle: VpnHandle) {
    handle.cleanup();
}

#[derive(Clone)]
pub(super) struct NetworkSnapshot {
    bind_indices: (u32, u32),
}

pub(super) fn network_snapshot(handle: &VpnHandle) -> NetworkSnapshot {
    NetworkSnapshot {
        bind_indices: handle.bind_indices(),
    }
}

pub(super) fn network_check(snapshot: &NetworkSnapshot) -> super::NetworkAction {
    match physical_iface() {
        Ok(current) if current.bind_indices() != snapshot.bind_indices => {
            super::NetworkAction::Reconcile
        }
        _ => super::NetworkAction::Healthy,
    }
}

/// Discover the physical default-route interface(s) via the IP Helper route
/// table. The uniform VPN monitor compares this against the connected route and
/// triggers full in-place reconciliation if it changes.
pub(super) fn physical_iface() -> anyhow::Result<PhysIface> {
    let index4 = default_route_ifindex(AF_INET)
        .context("could not find the IPv4 default-route interface")?;
    // IPv6 may be disabled; fall back to the IPv4 interface index like before.
    let index6 = default_route_ifindex(AF_INET6).unwrap_or(index4);
    Ok(PhysIface { index4, index6 })
}

/// Bring up the WinTUN device, routing, DNS, and the kill switch. Manager
/// initialization owns prior-run cleanup; this function rolls back only the
/// state created by its own failed setup attempt.
pub(super) fn setup(phys: PhysIface, allow_lan: bool) -> anyhow::Result<VpnHandle> {
    tracing::debug!("starting Windows VPN setup");
    let wintun = unsafe { wintun::load() }.map_err(|e| anyhow::anyhow!("load wintun.dll: {e}"))?;
    tracing::debug!("loaded wintun.dll for VPN setup");
    let rollback = scopeguard::guard((), |_| cleanup_stale());

    let adapter = match Adapter::open(&wintun, TUN_NAME) {
        Ok(existing) => {
            tracing::debug!("opened existing WinTUN adapter");
            existing
        }
        Err(_) => {
            tracing::debug!("creating WinTUN adapter");
            Adapter::create(&wintun, TUN_NAME, TUN_NAME, Some(TUN_GUID))
                .map_err(|e| anyhow::anyhow!("create wintun adapter: {e}"))?
        }
    };
    tracing::debug!("WinTUN adapter is ready");
    let luid = unsafe { adapter.get_luid().Value };

    // Addresses + MTU + DNS sentinel.
    set_unicast_address(luid, IpAddr::V4(TUN_V4_ADDR), TUN_V4_PREFIX)
        .context("assign tun IPv4 address")?;
    let _ = set_unicast_address(luid, IpAddr::V6(TUN_V6_ADDR), TUN_V6_PREFIX);
    let _ = adapter.set_mtu(TUN_MTU);
    let _ = set_tun_dns(adapter.get_guid(), &sentinel_dns());
    tracing::debug!("configured WinTUN addressing and DNS");

    // Kill switch before capture routes. (The engine's own broker/bridge/exit sockets reach the
    // physical NIC per-process via IP_UNICAST_IF + the loopback forwarder, so
    // there are no destination bypass routes to punch in here.)
    let mut firewall = WfpKillSwitch::new();
    firewall.preflight().context("kill switch preflight")?;
    tracing::debug!("completed WFP kill-switch preflight");
    firewall
        .install(&geph_app_ids(), luid, allow_lan)
        .context("install kill switch")?;
    tracing::debug!("installed WFP kill switch");

    ensure_split_routes(luid)?;
    tracing::debug!("installed WinTUN capture routes");

    scopeguard::ScopeGuard::into_inner(rollback);
    Ok(VpnHandle {
        wintun,
        adapter,
        phys,
        firewall,
    })
}

/// Reassert WinTUN, DNS, routes, and the complete WFP policy without replacing
/// the live adapter. WFP is committed first, so route repair remains fail-closed.
pub(super) fn reconcile(
    handle: &mut VpnHandle,
    phys: PhysIface,
    allow_lan: bool,
) -> anyhow::Result<()> {
    let luid = handle.luid();
    set_unicast_address(luid, IpAddr::V4(TUN_V4_ADDR), TUN_V4_PREFIX)
        .context("reassert tun IPv4 address")?;
    let _ = set_unicast_address(luid, IpAddr::V6(TUN_V6_ADDR), TUN_V6_PREFIX);
    let _ = handle.adapter.set_mtu(TUN_MTU);
    let _ = set_tun_dns(handle.adapter.get_guid(), &sentinel_dns());
    handle
        .firewall
        .replace(&geph_app_ids(), luid, allow_lan)
        .context("reconcile kill switch")?;
    ensure_split_routes(luid)?;
    handle.phys = phys;
    Ok(())
}

/// Ensure the `/1` capture routes point our WinTUN interface. Idempotent: an
/// identical route already present is treated as success, so no delete/re-add
/// churn is needed. IPv4 is fatal (the tunnel cannot capture without it); IPv6
/// is best-effort since the host may have IPv6 disabled.
fn ensure_split_routes(luid: u64) -> anyhow::Result<()> {
    for (addr, prefix) in V4_SPLIT {
        add_route(luid, IpAddr::V4(addr), prefix)
            .with_context(|| format!("add tun route {addr}/{prefix}"))?;
    }
    for (addr, prefix) in V6_SPLIT {
        let _ = add_route(luid, IpAddr::V6(addr), prefix);
    }
    Ok(())
}

/// Remove the `/1` capture routes from our WinTUN interface. Best-effort.
fn delete_split_routes(luid: u64) {
    for (addr, prefix) in V4_SPLIT {
        let _ = delete_route(luid, IpAddr::V4(addr), prefix);
    }
    for (addr, prefix) in V6_SPLIT {
        let _ = delete_route(luid, IpAddr::V6(addr), prefix);
    }
}

/// Startup purge of VPN state a prior crashed manager left behind. The kill switch
/// is now non-dynamic (it stays installed across a crash so the machine remains
/// fail-closed), so a manager that restarts *disconnected* must delete it here or
/// the host stays blackholed. Best-effort; safe to call when nothing is stranded.
pub(super) fn cleanup_stale() {
    // A Task Scheduler restart after hard process termination does not pass
    // through register_manager's pre-registered removal barrier. Join the
    // already-started PnP removal here, then remove the nonpresent devnode so
    // the fixed requested GUID can be reused immediately.
    match AdapterRemovalNotification::prepare() {
        Ok(Some(removal)) => {
            if let Err(error) = removal.wait() {
                tracing::warn!(%error, "could not finish stale WinTUN device removal");
            }
        }
        Ok(None) => {
            if let Err(error) = uninstall_orphaned_adapter() {
                tracing::warn!(%error, "could not remove orphaned WinTUN device");
            }
        }
        Err(error) => {
            tracing::warn!(%error, "could not observe stale WinTUN device removal");
        }
    }

    // Do not open the adapter through WinTUN merely to clean it. Immediately
    // opening the same adapter again for setup can wedge inside
    // WintunOpenAdapter after a hard manager replacement. The adapter GUID is
    // fixed, so IP Helper can resolve its LUID directly without acquiring a
    // WinTUN adapter handle.
    let guid = windows_sys::core::GUID::from_u128(TUN_GUID);
    let mut luid: windows_sys::Win32::NetworkManagement::Ndis::NET_LUID_LH =
        unsafe { std::mem::zeroed() };
    let _ = set_tun_dns(TUN_GUID, &[]);
    if unsafe { ConvertInterfaceGuidToLuid(&guid, &mut luid) } == NO_ERROR {
        delete_split_routes(unsafe { luid.Value });
    }
    // Delete any leftover kill-switch sublayer (and its filters) by GUID, in case
    // a previous manager used a non-dynamic session or otherwise didn't clean up.
    let _ = firewall::purge_stale();
}

/// Full image paths the kill switch permits to egress the physical NIC: the
/// manager itself and the engine child.
fn geph_app_ids() -> Vec<std::path::PathBuf> {
    let mut ids = Vec::new();
    if let Ok(exe) = std::env::current_exe() {
        ids.push(exe);
    }
    // Must be the *same* full path the engine is actually spawned from, or the
    // kill switch's app-id permit won't match and the engine blocks itself.
    ids.push(crate::platform::engine_bin_path());
    ids
}

// ---- IP Helper (iphlpapi) host-network configuration ----
//
// Everything here is keyed by the WinTUN interface LUID and runs in-process, in
// place of the `netsh`/`powershell` subprocesses the reconnect path used to spawn.

/// Set (or clear, with an empty list) the tun adapter's DNS servers via a direct,
/// correctly-formed `SetInterfaceDnsSettings` call — instant and validation-free.
/// NOT wintun's `set_dns_servers`: its own `SetInterfaceDnsSettings` binding has
/// `NameServer`/`Domain` swapped, so it always fails and falls back to
/// `netsh set dns` without `validate=no`; netsh then probes the server, which
/// blocks ~12s whenever the tunnel isn't carrying packets — i.e. on every single
/// exit switch, dominating the reconnect gap.
fn set_tun_dns(guid: u128, servers: &[IpAddr]) -> anyhow::Result<()> {
    let guid = windows_sys::core::GUID::from_u128(guid);
    let v4: Vec<String> = servers
        .iter()
        .filter(|ip| ip.is_ipv4())
        .map(|ip| ip.to_string())
        .collect();
    let v6: Vec<String> = servers
        .iter()
        .filter(|ip| ip.is_ipv6())
        .map(|ip| ip.to_string())
        .collect();
    // One call per family; an empty NameServer string clears that family.
    for (list, v6_flag) in [(v4, 0u64), (v6, DNS_SETTING_IPV6 as u64)] {
        let joined: Vec<u16> = list
            .join(",")
            .encode_utf16()
            .chain(std::iter::once(0))
            .collect();
        let settings = DNS_INTERFACE_SETTINGS {
            Version: DNS_INTERFACE_SETTINGS_VERSION1,
            Flags: DNS_SETTING_NAMESERVER as u64 | v6_flag,
            Domain: std::ptr::null_mut(),
            NameServer: joined.as_ptr() as *mut u16,
            SearchList: std::ptr::null_mut(),
            RegistrationEnabled: 0,
            RegisterAdapterName: 0,
            EnableLLMNR: 0,
            QueryAdapterName: 0,
            ProfileNameServer: std::ptr::null_mut(),
        };
        let code = unsafe { SetInterfaceDnsSettings(guid, &settings) };
        if code != NO_ERROR {
            bail!(
                "SetInterfaceDnsSettings(ipv6={}) failed: 0x{code:08X}",
                v6_flag != 0
            );
        }
    }
    Ok(())
}

/// Fill a `SOCKADDR_INET` in place with `ip` (address family + address; port and
/// scope left zeroed). Writing a union field is safe in Rust; only reads are not.
fn write_sockaddr_inet(dst: &mut SOCKADDR_INET, ip: IpAddr) {
    match ip {
        IpAddr::V4(v4) => {
            let mut sa: SOCKADDR_IN = unsafe { std::mem::zeroed() };
            sa.sin_family = AF_INET;
            // `from_ne_bytes` lays the octets out in memory order, i.e. network
            // byte order, which is exactly what `S_addr` expects.
            sa.sin_addr = IN_ADDR {
                S_un: IN_ADDR_0 {
                    S_addr: u32::from_ne_bytes(v4.octets()),
                },
            };
            dst.Ipv4 = sa;
        }
        IpAddr::V6(v6) => {
            let mut sa: SOCKADDR_IN6 = unsafe { std::mem::zeroed() };
            sa.sin6_family = AF_INET6;
            sa.sin6_addr = IN6_ADDR {
                u: IN6_ADDR_0 { Byte: v6.octets() },
            };
            dst.Ipv6 = sa;
        }
    }
}

/// The unspecified address of `ip`'s family, used as the on-link route next hop.
fn unspecified_like(ip: IpAddr) -> IpAddr {
    match ip {
        IpAddr::V4(_) => IpAddr::V4(Ipv4Addr::UNSPECIFIED),
        IpAddr::V6(_) => IpAddr::V6(Ipv6Addr::UNSPECIFIED),
    }
}

/// Assign a unicast address to the interface. Replaces
/// `netsh interface ip set/add address`. An already-present identical address is
/// treated as success so reconciles are idempotent.
fn set_unicast_address(luid: u64, ip: IpAddr, prefix: u8) -> anyhow::Result<()> {
    let mut row: MIB_UNICASTIPADDRESS_ROW = unsafe { std::mem::zeroed() };
    unsafe { InitializeUnicastIpAddressEntry(&mut row) };
    row.InterfaceLuid.Value = luid;
    write_sockaddr_inet(&mut row.Address, ip);
    row.OnLinkPrefixLength = prefix;
    let code = unsafe { CreateUnicastIpAddressEntry(&row) };
    if code == NO_ERROR || code == ERROR_OBJECT_ALREADY_EXISTS {
        Ok(())
    } else {
        bail!("CreateUnicastIpAddressEntry({ip}/{prefix}) failed: 0x{code:08X}");
    }
}

/// Add an on-link route pointing our WinTUN interface. Replaces
/// `netsh interface ip add route`. `ERROR_OBJECT_ALREADY_EXISTS` (the exact route
/// is already present) is success. The `/1` splits are more specific than the
/// physical `/0`, so longest-prefix match captures traffic regardless of metric.
fn add_route(luid: u64, dest: IpAddr, prefix: u8) -> anyhow::Result<()> {
    let mut row: MIB_IPFORWARD_ROW2 = unsafe { std::mem::zeroed() };
    unsafe { InitializeIpForwardEntry(&mut row) };
    row.InterfaceLuid.Value = luid;
    write_sockaddr_inet(&mut row.DestinationPrefix.Prefix, dest);
    row.DestinationPrefix.PrefixLength = prefix;
    write_sockaddr_inet(&mut row.NextHop, unspecified_like(dest));
    let code = unsafe { CreateIpForwardEntry2(&row) };
    if code == NO_ERROR || code == ERROR_OBJECT_ALREADY_EXISTS {
        Ok(())
    } else {
        bail!("CreateIpForwardEntry2({dest}/{prefix}) failed: 0x{code:08X}");
    }
}

/// Delete one of our on-link routes. Replaces `netsh interface ip delete route`.
/// `ERROR_NOT_FOUND` (already gone) is success. The key is
/// (InterfaceLuid, DestinationPrefix, NextHop).
fn delete_route(luid: u64, dest: IpAddr, prefix: u8) -> anyhow::Result<()> {
    let mut row: MIB_IPFORWARD_ROW2 = unsafe { std::mem::zeroed() };
    row.InterfaceLuid.Value = luid;
    write_sockaddr_inet(&mut row.DestinationPrefix.Prefix, dest);
    row.DestinationPrefix.PrefixLength = prefix;
    write_sockaddr_inet(&mut row.NextHop, unspecified_like(dest));
    let code = unsafe { DeleteIpForwardEntry2(&row) };
    if code == NO_ERROR || code == ERROR_NOT_FOUND {
        Ok(())
    } else {
        bail!("DeleteIpForwardEntry2({dest}/{prefix}) failed: 0x{code:08X}");
    }
}

/// Interface index of the lowest-(route-)metric default route for `family`,
/// ignoring loopback. Replaces the `Get-NetRoute` PowerShell query. Our own
/// capture routes are `/1`, so the exact `/0` filter never selects them.
fn default_route_ifindex(family: ADDRESS_FAMILY) -> Option<u32> {
    let mut table: *mut MIB_IPFORWARD_TABLE2 = std::ptr::null_mut();
    let code = unsafe { GetIpForwardTable2(family, &mut table) };
    if code != NO_ERROR || table.is_null() {
        return None;
    }
    let mut best: Option<(u32, u32)> = None; // (route metric, interface index)
    unsafe {
        let count = (*table).NumEntries as usize;
        let rows = std::ptr::addr_of!((*table).Table) as *const MIB_IPFORWARD_ROW2;
        for i in 0..count {
            let row = &*rows.add(i);
            if row.DestinationPrefix.PrefixLength != 0 {
                continue; // not a default route
            }
            if row.DestinationPrefix.Prefix.si_family != family {
                continue;
            }
            if row.Loopback != 0 {
                continue;
            }
            if best.map(|(m, _)| row.Metric < m).unwrap_or(true) {
                best = Some((row.Metric, row.InterfaceIndex));
            }
        }
        FreeMibTable(table as *const c_void);
    }
    best.map(|(_, idx)| idx)
}

/// The DNS servers configured on the physical adapter(s) whose interface index is
/// in `indices` (the physical default-route interface we pin the engine to). Read
/// via `GetAdaptersAddresses`. We only ever change the *tun* adapter's DNS (the
/// sentinel), so the physical adapter still carries its real DHCP/ISP resolvers.
fn physical_dns_servers(indices: &[u32]) -> Vec<IpAddr> {
    let flags = GAA_FLAG_SKIP_UNICAST | GAA_FLAG_SKIP_ANYCAST | GAA_FLAG_SKIP_MULTICAST;
    let mut size: u32 = 16 * 1024;
    let mut buf: Vec<u8> = Vec::new();
    let mut out: Vec<IpAddr> = Vec::new();
    // Size-probe / retry loop: GetAdaptersAddresses reports ERROR_BUFFER_OVERFLOW
    // and updates `size` when the buffer is too small.
    for _ in 0..3 {
        buf.resize(size as usize, 0);
        let ret = unsafe {
            GetAdaptersAddresses(
                AF_UNSPEC as u32,
                flags,
                std::ptr::null(),
                buf.as_mut_ptr() as *mut IP_ADAPTER_ADDRESSES_LH,
                &mut size,
            )
        };
        if ret == ERROR_BUFFER_OVERFLOW {
            continue;
        }
        if ret != ERROR_SUCCESS {
            return out;
        }
        let mut adapter = buf.as_ptr() as *const IP_ADAPTER_ADDRESSES_LH;
        while !adapter.is_null() {
            let a = unsafe { &*adapter };
            let if4 = unsafe { a.Anonymous1.Anonymous.IfIndex };
            if indices.contains(&if4) || indices.contains(&a.Ipv6IfIndex) {
                let mut dns = a.FirstDnsServerAddress;
                while !dns.is_null() {
                    let d = unsafe { &*dns };
                    if let Some(ip) = unsafe { sockaddr_to_ip(d.Address.lpSockaddr) }
                        && !out.contains(&ip)
                    {
                        out.push(ip);
                    }
                    dns = d.Next;
                }
            }
            adapter = a.Next;
        }
        return out;
    }
    out
}

/// Decode a Win32 `SOCKADDR` (v4 or v6) into an `IpAddr`.
unsafe fn sockaddr_to_ip(sa: *const SOCKADDR) -> Option<IpAddr> {
    if sa.is_null() {
        return None;
    }
    match unsafe { (*sa).sa_family } {
        AF_INET => {
            let sin = sa as *const SOCKADDR_IN;
            let bytes = unsafe { (*sin).sin_addr.S_un.S_addr }.to_ne_bytes();
            Some(IpAddr::V4(Ipv4Addr::from(bytes)))
        }
        AF_INET6 => {
            let sin6 = sa as *const SOCKADDR_IN6;
            let bytes = unsafe { (*sin6).sin6_addr.u.Byte };
            Some(IpAddr::V6(Ipv6Addr::from(bytes)))
        }
        _ => None,
    }
}

/// The stdio packet pump bridging the WinTUN session and one engine child's
/// stdin/stdout (16-bit big-endian length framing, matching the engine's
/// `--stdio-vpn`). Dropping it stops both directions and joins the threads.
pub(super) struct Pump {
    session: Arc<Session>,
    stop: Arc<AtomicBool>,
    threads: Vec<JoinHandle<()>>,
}

impl Pump {
    pub(super) fn start(
        session: Arc<Session>,
        child_stdin: ChildStdin,
        child_stdout: ChildStdout,
    ) -> Pump {
        let stop = Arc::new(AtomicBool::new(false));
        let up = {
            let session = session.clone();
            let stop = stop.clone();
            std::thread::spawn(move || pump_up(session, child_stdin, stop))
        };
        let dn = {
            let session = session.clone();
            let stop = stop.clone();
            std::thread::spawn(move || pump_down(session, child_stdout, stop))
        };
        Pump {
            session,
            stop,
            threads: vec![up, dn],
        }
    }
}

impl Drop for Pump {
    fn drop(&mut self) {
        self.stop.store(true, Ordering::SeqCst);
        // Unblock the up-thread parked in `receive_blocking`.
        let _ = self.session.shutdown();
        for handle in self.threads.drain(..) {
            let _ = handle.join();
        }
    }
}

/// WinTUN -> engine: read IP packets off the device, length-prefix them, write to
/// the child's stdin.
fn pump_up(session: Arc<Session>, mut child_stdin: ChildStdin, stop: Arc<AtomicBool>) {
    while !stop.load(Ordering::SeqCst) {
        let packet = match session.receive_blocking() {
            Ok(p) => p,
            Err(_) => break,
        };
        let bytes = packet.bytes();
        let len = std::cmp::min(bytes.len(), u16::MAX as usize);
        if child_stdin.write_all(&(len as u16).to_be_bytes()).is_err()
            || child_stdin.write_all(&bytes[..len]).is_err()
            || child_stdin.flush().is_err()
        {
            break;
        }
    }
}

/// Engine -> WinTUN: read length-prefixed IP packets off the child's stdout and
/// inject them into the device.
fn pump_down(session: Arc<Session>, mut child_stdout: ChildStdout, stop: Arc<AtomicBool>) {
    let mut len_buf = [0u8; 2];
    while !stop.load(Ordering::SeqCst) {
        if child_stdout.read_exact(&mut len_buf).is_err() {
            break;
        }
        let len = u16::from_be_bytes(len_buf) as usize;
        if len == 0 {
            continue;
        }
        let mut buf = vec![0u8; len];
        if child_stdout.read_exact(&mut buf).is_err() {
            break;
        }
        match session.allocate_send_packet(len as u16) {
            Ok(mut packet) => {
                packet.bytes_mut().copy_from_slice(&buf);
                session.send_packet(packet);
            }
            Err(_) => break,
        }
    }
}