use std::{ fs::{create_dir_all, File, FileTimes, OpenOptions}, io::{self, BufRead, BufReader, Read, Seek, SeekFrom, Write}, path::Path, time::{Duration, SystemTime}, }; use anyhow::{anyhow, Result}; use indicatif::{ProgressBar, ProgressStyle}; use ntfs::{ structured_values::{NtfsFileNamespace, NtfsStandardInformation}, Ntfs, NtfsAttributeType, NtfsTime, }; // --------------------------------------------------------------------------- // Constants // --------------------------------------------------------------------------- const SECTOR_SIZE: u64 = 512; const BUF_SIZE: usize = 256 * 1024; // VHD format const VHD_COOKIE: &[u8; 8] = b"conectix"; const VHD_TYPE_FIXED: u32 = 2; const VHD_TYPE_DYNAMIC: u32 = 3; const VHD_TYPE_DIFFERENCING: u32 = 4; const VHD_FOOTER_DISK_TYPE_OFFSET: usize = 0x3C; const VHD_FOOTER_DATA_OFFSET: usize = 0x10; /// VHD footer Unique Id (GUID), 16 bytes at offset 68 (0x44) — identifies this VHD. const VHD_FOOTER_UNIQUE_ID_OFFSET: usize = 0x44; // Dynamic/differencing VHD header const DYNAMIC_HEADER_COOKIE: &[u8; 8] = b"cxsparse"; const DYNAMIC_HEADER_SIZE: usize = 1024; const DYNAMIC_BAT_OFFSET_FIELD: usize = 0x10; const DYNAMIC_MAX_ENTRIES_FIELD: usize = 0x18; const DYNAMIC_BLOCK_SIZE_FIELD: usize = 0x20; /// Dynamic header Parent Unique ID (GUID), 16 bytes at offset 40 (0x28) — only /// meaningful for differencing VHDs; points at the parent VHD's footer Unique Id. const DYNAMIC_PARENT_UNIQUE_ID_OFFSET: usize = 0x28; const BAT_UNUSED: u32 = 0xFFFFFFFF; pub type VhdGuid = [u8; 16]; /// Chain-linking info read from a VHD. `parent_id` is `Some` only for differencing /// disks (type 4), and points at the parent VHD's `own_id`. #[derive(Debug, Clone)] pub struct VhdGuidInfo { pub own_id: VhdGuid, pub parent_id: Option, /// Kept for diagnostics / future validation; not every caller inspects it. #[allow(dead_code)] pub disk_type: u32, } /// Read a VHD's Unique Id and (for differencing VHDs) its Parent Unique ID. /// Cheap — only reads the 512-byte footer plus, if differencing, the 1024-byte /// dynamic header. Used to build chains by matching child.parent_id -> parent.own_id. pub fn read_vhd_guid_info(path: &Path) -> Result { let mut f = File::open(path)?; let size = f.seek(SeekFrom::End(0))?; if size < SECTOR_SIZE { return Err(VhdError::InvalidCookie); } f.seek(SeekFrom::Start(size - SECTOR_SIZE))?; let mut footer = [0u8; SECTOR_SIZE as usize]; f.read_exact(&mut footer)?; if &footer[..8] != VHD_COOKIE { return Err(VhdError::InvalidCookie); } let own_id: VhdGuid = footer[VHD_FOOTER_UNIQUE_ID_OFFSET..VHD_FOOTER_UNIQUE_ID_OFFSET + 16] .try_into() .unwrap(); let disk_type = read_be_u32(&footer, VHD_FOOTER_DISK_TYPE_OFFSET); let parent_id = if disk_type == VHD_TYPE_DIFFERENCING { let header_offset = read_be_u64(&footer, VHD_FOOTER_DATA_OFFSET); f.seek(SeekFrom::Start(header_offset))?; let mut hdr = [0u8; DYNAMIC_HEADER_SIZE]; f.read_exact(&mut hdr)?; if &hdr[..8] != DYNAMIC_HEADER_COOKIE { return Err(VhdError::InvalidDynamicHeader); } let guid: VhdGuid = hdr [DYNAMIC_PARENT_UNIQUE_ID_OFFSET..DYNAMIC_PARENT_UNIQUE_ID_OFFSET + 16] .try_into() .unwrap(); Some(guid) } else { None }; Ok(VhdGuidInfo { own_id, parent_id, disk_type }) } // MBR const MBR_SIGNATURE: [u8; 2] = [0x55, 0xAA]; const MBR_PARTITION_TABLE_OFFSET: usize = 0x1BE; const MBR_PARTITION_ENTRY_SIZE: usize = 16; const MBR_MAX_PARTITIONS: usize = 4; const NTFS_PARTITION_TYPE: u8 = 0x07; // NTFS boot sector magic const NTFS_MAGIC: [u8; 4] = [0xEB, 0x52, 0x90, 0x4E]; /// Common virtual offsets where NTFS boot sector might start. const NTFS_PROBE_OFFSETS: [u64; 4] = [0, 32_256, 1_048_576, 512]; // Progress bar const PROGRESS_STYLE: &str = "{prefix} [{bar:20!.bright.yellow/dim.white}] {bytes:>8} [{elapsed}<{eta}, {bytes_per_sec}]"; // Windows epoch -> Unix epoch offset (100ns intervals) const WINDOWS_EPOCH_OFFSET: u64 = 116_444_736_000_000_000; // --------------------------------------------------------------------------- // VHD error type // --------------------------------------------------------------------------- #[derive(Debug, thiserror::Error)] pub enum VhdError { #[error(transparent)] Io(#[from] io::Error), #[error("Not a valid VHD file")] InvalidCookie, #[error("Unsupported VHD type {0}")] UnsupportedType(u32), #[error("Invalid dynamic VHD header")] InvalidDynamicHeader, #[error("No NTFS partition found in VHD")] NoNtfsPartition, } // --------------------------------------------------------------------------- // VHD layout: how to map virtual offsets to file offsets // --------------------------------------------------------------------------- enum VhdLayout { /// Data is contiguous from offset 0 to (file_size - 512). Fixed, /// Data is in blocks addressed via a Block Allocation Table. /// Used for both dynamic (type 3) and differencing (type 4) VHDs. Sparse { bat: Vec, block_size: u64 }, } impl VhdLayout { /// Parse the dynamic/differencing header and BAT. fn parse_sparse(inner: &mut R, footer: &[u8]) -> Result<(Self, u64), VhdError> { let header_offset = read_be_u64(footer, VHD_FOOTER_DATA_OFFSET); inner.seek(SeekFrom::Start(header_offset))?; let mut hdr = [0u8; DYNAMIC_HEADER_SIZE]; inner.read_exact(&mut hdr)?; if &hdr[..8] != DYNAMIC_HEADER_COOKIE { return Err(VhdError::InvalidDynamicHeader); } let bat_offset = read_be_u64(&hdr, DYNAMIC_BAT_OFFSET_FIELD); let max_entries = read_be_u32(&hdr, DYNAMIC_MAX_ENTRIES_FIELD) as usize; let block_size = read_be_u32(&hdr, DYNAMIC_BLOCK_SIZE_FIELD) as u64; inner.seek(SeekFrom::Start(bat_offset))?; let mut raw = vec![0u8; max_entries * 4]; inner.read_exact(&mut raw)?; let bat: Vec = (0..max_entries).map(|i| read_be_u32(&raw, i * 4)).collect(); Ok((VhdLayout::Sparse { bat, block_size }, max_entries as u64 * block_size)) } /// Read bytes from a virtual offset according to this layout. fn read_at( &self, inner: &mut R, virt_off: u64, virtual_size: u64, buf: &mut [u8], ) -> io::Result { if virt_off >= virtual_size { return Ok(0); } let cap = std::cmp::min(buf.len() as u64, virtual_size - virt_off) as usize; match self { VhdLayout::Fixed => { inner.seek(SeekFrom::Start(virt_off))?; inner.read(&mut buf[..cap]) } VhdLayout::Sparse { bat, block_size } => { let bi = (virt_off / block_size) as usize; let bo = virt_off % block_size; let n = std::cmp::min(cap, (block_size - bo) as usize); if bi >= bat.len() || bat[bi] == BAT_UNUSED { buf[..n].fill(0); Ok(n) } else { // Each block: bitmap sector + data. Skip bitmap. let file_off = bat[bi] as u64 * SECTOR_SIZE + SECTOR_SIZE + bo; inner.seek(SeekFrom::Start(file_off))?; inner.read(&mut buf[..n]) } } } } /// Read 4 bytes from a virtual offset (for magic-byte probing). fn read_magic( &self, inner: &mut R, offset: u64, ) -> io::Result<[u8; 4]> { let mut buf = [0u8; 4]; match self { VhdLayout::Fixed => { inner.seek(SeekFrom::Start(offset))?; inner.read_exact(&mut buf)?; } VhdLayout::Sparse { bat, block_size } => { let bi = (offset / block_size) as usize; if bi < bat.len() && bat[bi] != BAT_UNUSED { let file_off = bat[bi] as u64 * SECTOR_SIZE + SECTOR_SIZE + offset % block_size; inner.seek(SeekFrom::Start(file_off))?; inner.read_exact(&mut buf)?; } } } Ok(buf) } } // --------------------------------------------------------------------------- // VHD reader (single VHD) // --------------------------------------------------------------------------- /// Transparently presents the NTFS partition within a fixed or dynamic VHD. pub struct VhdReader { inner: R, layout: VhdLayout, ntfs_offset: u64, virtual_size: u64, pos: u64, } impl VhdReader { pub fn new(mut inner: R) -> Result { let file_size = inner.seek(SeekFrom::End(0))?; if file_size < SECTOR_SIZE { return Err(VhdError::InvalidCookie); } inner.seek(SeekFrom::Start(file_size - SECTOR_SIZE))?; let mut footer = [0u8; SECTOR_SIZE as usize]; inner.read_exact(&mut footer)?; if &footer[..8] != VHD_COOKIE { return Err(VhdError::InvalidCookie); } let disk_type = read_be_u32(&footer, VHD_FOOTER_DISK_TYPE_OFFSET); let (layout, virtual_size) = match disk_type { VHD_TYPE_FIXED => (VhdLayout::Fixed, file_size - SECTOR_SIZE), VHD_TYPE_DYNAMIC | VHD_TYPE_DIFFERENCING => { VhdLayout::parse_sparse(&mut inner, &footer)? } t => return Err(VhdError::UnsupportedType(t)), }; let ntfs_offset = find_ntfs_offset(&mut inner, &layout, virtual_size)?; Ok(Self { inner, layout, ntfs_offset, virtual_size, pos: 0 }) } fn ntfs_size(&self) -> u64 { self.virtual_size - self.ntfs_offset } } impl Read for VhdReader { fn read(&mut self, buf: &mut [u8]) -> io::Result { let remaining = self.ntfs_size().saturating_sub(self.pos); if remaining == 0 { return Ok(0); } let cap = std::cmp::min(buf.len() as u64, remaining) as usize; let n = self.layout.read_at( &mut self.inner, self.ntfs_offset + self.pos, self.virtual_size, &mut buf[..cap], )?; self.pos += n as u64; Ok(n) } } impl Seek for VhdReader { fn seek(&mut self, pos: SeekFrom) -> io::Result { let target = match pos { SeekFrom::Start(o) => o as i64, SeekFrom::Current(o) => self.pos as i64 + o, SeekFrom::End(o) => self.ntfs_size() as i64 + o, }; if target < 0 { return Err(io::Error::new(io::ErrorKind::InvalidInput, "seek before start")); } self.pos = target as u64; Ok(self.pos) } } // --------------------------------------------------------------------------- // Chained VHD reader (base + N deltas overlaid, no on-disk merge needed) // --------------------------------------------------------------------------- /// One layer of a VHD chain: a file handle plus its parsed layout. struct VhdLayer { inner: R, layout: VhdLayout, } /// Reads from a chain of VHDs where `layers[0]` is the base (dynamic/fixed) /// and `layers[1..]` are differencing VHDs in parent→child order. /// /// For each read, walks layers from top delta down to base. At each layer, /// if the sector is present-and-modified (BAT allocated + bitmap bit set), /// that layer's bytes win; otherwise the read falls through to the layer /// below. The base layer's own `read_at` handles zero-fill for unallocated /// dynamic blocks. pub struct ChainedVhdReader { layers: Vec>, ntfs_offset: u64, virtual_size: u64, pos: u64, } impl ChainedVhdReader { /// Build a chain reader. `readers` must be ordered base-first, top-most delta last. pub fn new(readers: Vec) -> Result { if readers.is_empty() { return Err(VhdError::InvalidCookie); } let mut layers: Vec> = Vec::with_capacity(readers.len()); let mut virtual_size = 0u64; for (idx, mut r) in readers.into_iter().enumerate() { let file_size = r.seek(SeekFrom::End(0))?; if file_size < SECTOR_SIZE { return Err(VhdError::InvalidCookie); } r.seek(SeekFrom::Start(file_size - SECTOR_SIZE))?; let mut footer = [0u8; SECTOR_SIZE as usize]; r.read_exact(&mut footer)?; if &footer[..8] != VHD_COOKIE { return Err(VhdError::InvalidCookie); } let disk_type = read_be_u32(&footer, VHD_FOOTER_DISK_TYPE_OFFSET); let (layout, vsize) = match disk_type { VHD_TYPE_FIXED => (VhdLayout::Fixed, file_size - SECTOR_SIZE), VHD_TYPE_DYNAMIC | VHD_TYPE_DIFFERENCING => { VhdLayout::parse_sparse(&mut r, &footer)? } t => return Err(VhdError::UnsupportedType(t)), }; if idx == 0 { virtual_size = vsize; } layers.push(VhdLayer { inner: r, layout }); } let ntfs_offset = find_ntfs_offset_chain(&mut layers, virtual_size)?; Ok(Self { layers, ntfs_offset, virtual_size, pos: 0 }) } fn ntfs_size(&self) -> u64 { self.virtual_size - self.ntfs_offset } } /// If `layer` has the sector for `virt_off` present AND marked modified in /// its bitmap, read from it and return `Some(bytes_read)`. Otherwise `None` /// signals "fall through to the layer below". /// /// Reads are capped at the current sector boundary. The VHD bitmap is /// per-sector: a single block can have a mixed 1/0 pattern, so a larger read /// might cross a sector that belongs to a different layer. The Read /// implementation loops until `buf` is filled, amortising the extra calls. fn try_read_from_layer( layer: &mut VhdLayer, virt_off: u64, buf: &mut [u8], ) -> io::Result> { match &layer.layout { VhdLayout::Fixed => Ok(None), // Fixed deltas make no sense; fall through. VhdLayout::Sparse { bat, block_size } => { let bi = (virt_off / block_size) as usize; let bo = virt_off % block_size; if bi >= bat.len() || bat[bi] == BAT_UNUSED { return Ok(None); } // Cap at the current sector to honour per-sector bitmap semantics. let sector_remaining = (SECTOR_SIZE - (virt_off % SECTOR_SIZE)) as usize; let n = std::cmp::min(buf.len(), sector_remaining); let block_file_offset = bat[bi] as u64 * SECTOR_SIZE; // Read the block's bitmap sector. layer.inner.seek(SeekFrom::Start(block_file_offset))?; let mut bitmap = [0u8; SECTOR_SIZE as usize]; layer.inner.read_exact(&mut bitmap)?; let sector_in_block = (bo / SECTOR_SIZE) as usize; let bitmap_byte = bitmap[sector_in_block / 8]; let bitmap_bit = 7 - (sector_in_block % 8); // MSB first if (bitmap_byte >> bitmap_bit) & 1 == 0 { return Ok(None); } let file_off = block_file_offset + SECTOR_SIZE + bo; layer.inner.seek(SeekFrom::Start(file_off))?; let got = layer.inner.read(&mut buf[..n])?; Ok(Some(got)) } } } /// Walk layers top-to-bottom; first layer that owns the sector wins. /// The base layer (index 0) always answers (possibly with zeros for /// unallocated dynamic blocks). fn read_chain( layers: &mut [VhdLayer], vsize: u64, virt_off: u64, buf: &mut [u8], ) -> io::Result { if virt_off >= vsize { return Ok(0); } let cap = std::cmp::min(buf.len() as u64, vsize - virt_off) as usize; // Try deltas from top (last) down to just above base (index 1). for i in (1..layers.len()).rev() { if let Some(n) = try_read_from_layer(&mut layers[i], virt_off, &mut buf[..cap])? { return Ok(n); } } // Fall through to base. let base = &mut layers[0]; base.layout.read_at(&mut base.inner, virt_off, vsize, &mut buf[..cap]) } impl Read for ChainedVhdReader { fn read(&mut self, buf: &mut [u8]) -> io::Result { let remaining = self.ntfs_size().saturating_sub(self.pos); if remaining == 0 { return Ok(0); } let cap = std::cmp::min(buf.len() as u64, remaining) as usize; let virt_off = self.ntfs_offset + self.pos; let n = read_chain(&mut self.layers, self.virtual_size, virt_off, &mut buf[..cap])?; self.pos += n as u64; Ok(n) } } impl Seek for ChainedVhdReader { fn seek(&mut self, pos: SeekFrom) -> io::Result { let target = match pos { SeekFrom::Start(o) => o as i64, SeekFrom::Current(o) => self.pos as i64 + o, SeekFrom::End(o) => self.ntfs_size() as i64 + o, }; if target < 0 { return Err(io::Error::new(io::ErrorKind::InvalidInput, "seek before start")); } self.pos = target as u64; Ok(self.pos) } } // --------------------------------------------------------------------------- // NTFS partition detection (shared between single and merged readers) // --------------------------------------------------------------------------- /// Find NTFS offset in a single VHD. fn find_ntfs_offset( inner: &mut R, layout: &VhdLayout, vsize: u64, ) -> Result { // Try MBR if vsize >= SECTOR_SIZE { let mut mbr = [0u8; SECTOR_SIZE as usize]; let _ = layout.read_at(inner, 0, vsize, &mut mbr); if mbr[510..512] == MBR_SIGNATURE { for i in 0..MBR_MAX_PARTITIONS { let eo = MBR_PARTITION_TABLE_OFFSET + i * MBR_PARTITION_ENTRY_SIZE; if mbr[eo + 4] == NTFS_PARTITION_TYPE { let lba = u32::from_le_bytes(mbr[eo + 8..eo + 12].try_into().unwrap()); let offset = lba as u64 * SECTOR_SIZE; if offset + 4 <= vsize && layout.read_magic(inner, offset)? == NTFS_MAGIC { return Ok(offset); } } } } } // Probe common offsets for offset in NTFS_PROBE_OFFSETS { if offset + 4 <= vsize && layout.read_magic(inner, offset)? == NTFS_MAGIC { return Ok(offset); } } Err(VhdError::NoNtfsPartition) } /// Find NTFS offset in a chained view (base + N deltas). fn find_ntfs_offset_chain( layers: &mut [VhdLayer], vsize: u64, ) -> Result { // If the chain has only a base, defer to the single-VHD finder — it's simpler // and avoids the bitmap machinery for a pure dynamic/fixed disk. if layers.len() == 1 { let base = &mut layers[0]; return find_ntfs_offset(&mut base.inner, &base.layout, vsize); } let read_magic = |layers: &mut [VhdLayer], offset: u64| -> io::Result<[u8; 4]> { let mut buf = [0u8; 4]; read_chain(layers, vsize, offset, &mut buf)?; Ok(buf) }; // Try MBR from merged view. if vsize >= SECTOR_SIZE { let mut mbr = [0u8; SECTOR_SIZE as usize]; read_chain(layers, vsize, 0, &mut mbr)?; if mbr[510..512] == MBR_SIGNATURE { for i in 0..MBR_MAX_PARTITIONS { let eo = MBR_PARTITION_TABLE_OFFSET + i * MBR_PARTITION_ENTRY_SIZE; if mbr[eo + 4] == NTFS_PARTITION_TYPE { let lba = u32::from_le_bytes(mbr[eo + 8..eo + 12].try_into().unwrap()); let offset = lba as u64 * SECTOR_SIZE; if offset + 4 <= vsize && read_magic(layers, offset)? == NTFS_MAGIC { return Ok(offset); } } } } } for offset in NTFS_PROBE_OFFSETS { if offset + 4 <= vsize && read_magic(layers, offset)? == NTFS_MAGIC { return Ok(offset); } } Err(VhdError::NoNtfsPartition) } // --------------------------------------------------------------------------- // NTFS extraction (shared logic) // --------------------------------------------------------------------------- fn is_ntfs_system_entry(name: &str) -> bool { name.starts_with('$') || name == "." || name == ".." || name == "System Volume Information" } /// Whether a directory-index entry should be skipped when extracting. /// /// Besides NTFS system metadata, this skips DOS (8.3) short-name aliases: a file /// that has a separate short name appears in the index *twice* — once with its /// Win32 long name and once with the `Dos` short name. Without this, every such /// entry is extracted a second time under its mangled `NAME~1.EXT` name (and /// short-named directories get their whole subtree duplicated). fn skip_index_entry(namespace: NtfsFileNamespace, name: &str) -> bool { namespace == NtfsFileNamespace::Dos || is_ntfs_system_entry(name) } fn ntfs_time_to_system_time(t: NtfsTime) -> SystemTime { let nanos = (t.nt_timestamp() - WINDOWS_EPOCH_OFFSET) * 100; SystemTime::UNIX_EPOCH + Duration::from_nanos(nanos) } fn set_ntfs_timestamps(fs: &mut T, file: &ntfs::NtfsFile, path: &Path) { let mut attrs = file.attributes(); while let Some(Ok(attr)) = attrs.next(fs) { if let Ok(attr) = attr.to_attribute() { if let Ok(NtfsAttributeType::StandardInformation) = attr.ty() { if let Ok(info) = attr.resident_structured_value::() { let _ = OpenOptions::new().write(true).open(path).and_then(|h| { h.set_times( FileTimes::new() .set_accessed(ntfs_time_to_system_time(info.access_time())) .set_modified(ntfs_time_to_system_time(info.modification_time())), ) }); } break; } } } } fn extract_ntfs_dir( ntfs: &Ntfs, fs: &mut T, dir: &ntfs::NtfsFile, out: &Path, pb: &ProgressBar, ) -> Result<()> { let index = dir.directory_index(fs)?; let mut iter = index.entries(); while let Some(entry) = iter.next(fs) { let entry = entry?; let key = entry.key().ok_or_else(|| anyhow!("missing key"))??; let name = key.name().to_string_lossy(); if skip_index_entry(key.namespace(), &name) { continue; } let file = entry.to_file(ntfs, fs)?; let dest = out.join(&*name); if key.is_directory() { create_dir_all(&dest)?; extract_ntfs_dir(ntfs, fs, &file, &dest, pb)?; set_ntfs_timestamps(fs, &file, &dest); } else if let Some(data) = file.data(fs, "") { let data_item = data?; let attr = data_item.to_attribute()?; let mut reader = BufReader::with_capacity(BUF_SIZE, attr.value(fs)?.attach(fs)); let mut out_file = File::create(&dest)?; loop { let buf = reader.fill_buf()?; if buf.is_empty() { break; } out_file.write_all(buf)?; let n = buf.len(); reader.consume(n); pb.inc(n as u64); } out_file.flush()?; drop(reader); set_ntfs_timestamps(fs, &file, &dest); } } Ok(()) } fn calculate_ntfs_size( ntfs: &Ntfs, fs: &mut T, dir: &ntfs::NtfsFile, ) -> Result { let mut total = 0u64; let index = dir.directory_index(fs)?; let mut iter = index.entries(); while let Some(entry) = iter.next(fs) { let entry = entry?; let key = entry.key().ok_or_else(|| anyhow!("missing key"))??; if skip_index_entry(key.namespace(), key.name().to_string_lossy().as_ref()) { continue; } let file = entry.to_file(ntfs, fs)?; if key.is_directory() { total += calculate_ntfs_size(ntfs, fs, &file)?; } else if let Some(data) = file.data(fs, "") { total += data?.to_attribute()?.value_length(); } } Ok(total) } /// Shared extraction logic: given an NTFS-bearing Read+Seek, extract to output_dir. fn extract_ntfs_to_dir(fs: &mut T, output_dir: &Path, prefix: &str) -> Result<()> { let mut ntfs = Ntfs::new(fs)?; ntfs.read_upcase_table(fs)?; let root = ntfs.root_directory(fs)?; let total = calculate_ntfs_size(&ntfs, fs, &root)?; let pb = ProgressBar::new(total) .with_style(ProgressStyle::default_bar().template(PROGRESS_STYLE)?); pb.set_prefix(prefix.to_string()); create_dir_all(output_dir)?; let root = ntfs.root_directory(fs)?; extract_ntfs_dir(&ntfs, fs, &root, output_dir, &pb)?; pb.finish(); Ok(()) } // --------------------------------------------------------------------------- // Public API // --------------------------------------------------------------------------- /// Extract all files from a single VHD's NTFS filesystem, then delete the VHD. /// The `output_dir` is where files are extracted to. pub fn extract_vhd(vhd_path: &Path, output_dir: &Path) -> Result<()> { println!("Extracting VHD: {}", vhd_path.display()); let mut vhd = VhdReader::new(File::open(vhd_path)?).map_err(|e| anyhow!(e))?; let prefix = output_dir.file_name().unwrap_or_default().to_string_lossy().to_string(); extract_ntfs_to_dir(&mut vhd, output_dir, &prefix)?; println!("Extracted to: {}", output_dir.display()); drop(vhd); if let Err(e) = std::fs::remove_file(vhd_path) { println!("WARNING: Could not delete VHD: {e}"); } Ok(()) } /// Extract files from a chained view of a base + N differencing VHDs. /// /// `chain` must be ordered base-first, top-most delta last. A chain of length 1 /// is equivalent to extracting just the base. Unlike [`extract_vhd`], this does /// **not** delete the inputs — the caller is responsible, since a single VHD /// in a chain is typically consumed by multiple extractions (one per patch /// level) and must not be removed until all of them have completed. pub fn extract_chained_vhd(chain: &[&Path], output_dir: &Path) -> Result<()> { if chain.is_empty() { return Err(anyhow!("extract_chained_vhd: empty chain")); } let paths_disp = chain .iter() .map(|p| p.display().to_string()) .collect::>() .join(" + "); println!("Extracting chained VHD: {paths_disp}"); let readers: Vec = chain .iter() .map(|p| File::open(p)) .collect::>()?; let mut reader = ChainedVhdReader::new(readers).map_err(|e| anyhow!(e))?; let prefix = output_dir.file_name().unwrap_or_default().to_string_lossy().to_string(); extract_ntfs_to_dir(&mut reader, output_dir, &prefix)?; println!("Extracted to: {}", output_dir.display()); Ok(()) } // --------------------------------------------------------------------------- // Helpers // --------------------------------------------------------------------------- fn read_be_u32(buf: &[u8], offset: usize) -> u32 { u32::from_be_bytes(buf[offset..offset + 4].try_into().unwrap()) } fn read_be_u64(buf: &[u8], offset: usize) -> u64 { u64::from_be_bytes(buf[offset..offset + 8].try_into().unwrap()) } #[cfg(test)] mod tests { use super::skip_index_entry; use ntfs::structured_values::NtfsFileNamespace; #[test] fn skips_dos_short_name_aliases() { // 8.3 aliases duplicate a Win32 entry and must not be extracted again. assert!(skip_index_entry(NtfsFileNamespace::Dos, "OXGETH~1.EXE")); assert!(skip_index_entry(NtfsFileNamespace::Dos, "PROGRA~1")); } #[test] fn keeps_long_and_native_names() { assert!(!skip_index_entry(NtfsFileNamespace::Win32, "oxGetHwInfo.exe")); // A name that is its own short name (no separate Dos entry) is kept. assert!(!skip_index_entry(NtfsFileNamespace::Win32AndDos, "game.bat")); assert!(!skip_index_entry(NtfsFileNamespace::Posix, "readme")); } #[test] fn skips_system_entries_regardless_of_namespace() { assert!(skip_index_entry(NtfsFileNamespace::Win32, "$MFT")); assert!(skip_index_entry(NtfsFileNamespace::Win32, "System Volume Information")); assert!(skip_index_entry(NtfsFileNamespace::Win32, ".")); assert!(skip_index_entry(NtfsFileNamespace::Win32, "..")); } }