mirror of
https://gitea.tendokyu.moe/beerpsi/fsdecrypt.git
synced 2026-09-29 02:08:03 +03:00
When a file or directory has a separate 8.3 short name, it appears in the NTFS directory index twice: once under its Win32 long name and once under the Dos short name. The extractor iterated all index entries, so every such item was written a second time under its mangled NAME~1.EXT alias — and short-named directories had their entire subtree re-extracted. Skip index entries in the Dos namespace (the file is still extracted via its Win32 entry). Add unit tests for the skip decision. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
794 lines
28 KiB
Rust
794 lines
28 KiB
Rust
use std::{
|
|
fs::{create_dir_all, File, FileTimes, OpenOptions},
|
|
io::{self, BufRead, BufReader, Read, Seek, SeekFrom, Write},
|
|
path::Path,
|
|
time::{Duration, SystemTime},
|
|
};
|
|
|
|
use anyhow::{anyhow, Result};
|
|
use indicatif::{ProgressBar, ProgressStyle};
|
|
use ntfs::{
|
|
structured_values::{NtfsFileNamespace, NtfsStandardInformation},
|
|
Ntfs, NtfsAttributeType, NtfsTime,
|
|
};
|
|
|
|
// ---------------------------------------------------------------------------
|
|
// Constants
|
|
// ---------------------------------------------------------------------------
|
|
|
|
const SECTOR_SIZE: u64 = 512;
|
|
const BUF_SIZE: usize = 256 * 1024;
|
|
|
|
// VHD format
|
|
const VHD_COOKIE: &[u8; 8] = b"conectix";
|
|
const VHD_TYPE_FIXED: u32 = 2;
|
|
const VHD_TYPE_DYNAMIC: u32 = 3;
|
|
const VHD_TYPE_DIFFERENCING: u32 = 4;
|
|
const VHD_FOOTER_DISK_TYPE_OFFSET: usize = 0x3C;
|
|
const VHD_FOOTER_DATA_OFFSET: usize = 0x10;
|
|
/// VHD footer Unique Id (GUID), 16 bytes at offset 68 (0x44) — identifies this VHD.
|
|
const VHD_FOOTER_UNIQUE_ID_OFFSET: usize = 0x44;
|
|
|
|
// Dynamic/differencing VHD header
|
|
const DYNAMIC_HEADER_COOKIE: &[u8; 8] = b"cxsparse";
|
|
const DYNAMIC_HEADER_SIZE: usize = 1024;
|
|
const DYNAMIC_BAT_OFFSET_FIELD: usize = 0x10;
|
|
const DYNAMIC_MAX_ENTRIES_FIELD: usize = 0x18;
|
|
const DYNAMIC_BLOCK_SIZE_FIELD: usize = 0x20;
|
|
/// Dynamic header Parent Unique ID (GUID), 16 bytes at offset 40 (0x28) — only
|
|
/// meaningful for differencing VHDs; points at the parent VHD's footer Unique Id.
|
|
const DYNAMIC_PARENT_UNIQUE_ID_OFFSET: usize = 0x28;
|
|
const BAT_UNUSED: u32 = 0xFFFFFFFF;
|
|
|
|
pub type VhdGuid = [u8; 16];
|
|
|
|
/// Chain-linking info read from a VHD. `parent_id` is `Some` only for differencing
|
|
/// disks (type 4), and points at the parent VHD's `own_id`.
|
|
#[derive(Debug, Clone)]
|
|
pub struct VhdGuidInfo {
|
|
pub own_id: VhdGuid,
|
|
pub parent_id: Option<VhdGuid>,
|
|
/// Kept for diagnostics / future validation; not every caller inspects it.
|
|
#[allow(dead_code)]
|
|
pub disk_type: u32,
|
|
}
|
|
|
|
/// Read a VHD's Unique Id and (for differencing VHDs) its Parent Unique ID.
|
|
/// Cheap — only reads the 512-byte footer plus, if differencing, the 1024-byte
|
|
/// dynamic header. Used to build chains by matching child.parent_id -> parent.own_id.
|
|
pub fn read_vhd_guid_info(path: &Path) -> Result<VhdGuidInfo, VhdError> {
|
|
let mut f = File::open(path)?;
|
|
let size = f.seek(SeekFrom::End(0))?;
|
|
if size < SECTOR_SIZE {
|
|
return Err(VhdError::InvalidCookie);
|
|
}
|
|
f.seek(SeekFrom::Start(size - SECTOR_SIZE))?;
|
|
let mut footer = [0u8; SECTOR_SIZE as usize];
|
|
f.read_exact(&mut footer)?;
|
|
if &footer[..8] != VHD_COOKIE {
|
|
return Err(VhdError::InvalidCookie);
|
|
}
|
|
let own_id: VhdGuid = footer[VHD_FOOTER_UNIQUE_ID_OFFSET..VHD_FOOTER_UNIQUE_ID_OFFSET + 16]
|
|
.try_into()
|
|
.unwrap();
|
|
let disk_type = read_be_u32(&footer, VHD_FOOTER_DISK_TYPE_OFFSET);
|
|
|
|
let parent_id = if disk_type == VHD_TYPE_DIFFERENCING {
|
|
let header_offset = read_be_u64(&footer, VHD_FOOTER_DATA_OFFSET);
|
|
f.seek(SeekFrom::Start(header_offset))?;
|
|
let mut hdr = [0u8; DYNAMIC_HEADER_SIZE];
|
|
f.read_exact(&mut hdr)?;
|
|
if &hdr[..8] != DYNAMIC_HEADER_COOKIE {
|
|
return Err(VhdError::InvalidDynamicHeader);
|
|
}
|
|
let guid: VhdGuid = hdr
|
|
[DYNAMIC_PARENT_UNIQUE_ID_OFFSET..DYNAMIC_PARENT_UNIQUE_ID_OFFSET + 16]
|
|
.try_into()
|
|
.unwrap();
|
|
Some(guid)
|
|
} else {
|
|
None
|
|
};
|
|
|
|
Ok(VhdGuidInfo { own_id, parent_id, disk_type })
|
|
}
|
|
|
|
// MBR
|
|
const MBR_SIGNATURE: [u8; 2] = [0x55, 0xAA];
|
|
const MBR_PARTITION_TABLE_OFFSET: usize = 0x1BE;
|
|
const MBR_PARTITION_ENTRY_SIZE: usize = 16;
|
|
const MBR_MAX_PARTITIONS: usize = 4;
|
|
const NTFS_PARTITION_TYPE: u8 = 0x07;
|
|
|
|
// NTFS boot sector magic
|
|
const NTFS_MAGIC: [u8; 4] = [0xEB, 0x52, 0x90, 0x4E];
|
|
/// Common virtual offsets where NTFS boot sector might start.
|
|
const NTFS_PROBE_OFFSETS: [u64; 4] = [0, 32_256, 1_048_576, 512];
|
|
|
|
// Progress bar
|
|
const PROGRESS_STYLE: &str =
|
|
"{prefix} [{bar:20!.bright.yellow/dim.white}] {bytes:>8} [{elapsed}<{eta}, {bytes_per_sec}]";
|
|
|
|
// Windows epoch -> Unix epoch offset (100ns intervals)
|
|
const WINDOWS_EPOCH_OFFSET: u64 = 116_444_736_000_000_000;
|
|
|
|
// ---------------------------------------------------------------------------
|
|
// VHD error type
|
|
// ---------------------------------------------------------------------------
|
|
|
|
#[derive(Debug, thiserror::Error)]
|
|
pub enum VhdError {
|
|
#[error(transparent)]
|
|
Io(#[from] io::Error),
|
|
#[error("Not a valid VHD file")]
|
|
InvalidCookie,
|
|
#[error("Unsupported VHD type {0}")]
|
|
UnsupportedType(u32),
|
|
#[error("Invalid dynamic VHD header")]
|
|
InvalidDynamicHeader,
|
|
#[error("No NTFS partition found in VHD")]
|
|
NoNtfsPartition,
|
|
}
|
|
|
|
// ---------------------------------------------------------------------------
|
|
// VHD layout: how to map virtual offsets to file offsets
|
|
// ---------------------------------------------------------------------------
|
|
|
|
enum VhdLayout {
|
|
/// Data is contiguous from offset 0 to (file_size - 512).
|
|
Fixed,
|
|
/// Data is in blocks addressed via a Block Allocation Table.
|
|
/// Used for both dynamic (type 3) and differencing (type 4) VHDs.
|
|
Sparse { bat: Vec<u32>, block_size: u64 },
|
|
}
|
|
|
|
impl VhdLayout {
|
|
/// Parse the dynamic/differencing header and BAT.
|
|
fn parse_sparse<R: Read + Seek>(inner: &mut R, footer: &[u8]) -> Result<(Self, u64), VhdError> {
|
|
let header_offset = read_be_u64(footer, VHD_FOOTER_DATA_OFFSET);
|
|
|
|
inner.seek(SeekFrom::Start(header_offset))?;
|
|
let mut hdr = [0u8; DYNAMIC_HEADER_SIZE];
|
|
inner.read_exact(&mut hdr)?;
|
|
if &hdr[..8] != DYNAMIC_HEADER_COOKIE {
|
|
return Err(VhdError::InvalidDynamicHeader);
|
|
}
|
|
|
|
let bat_offset = read_be_u64(&hdr, DYNAMIC_BAT_OFFSET_FIELD);
|
|
let max_entries = read_be_u32(&hdr, DYNAMIC_MAX_ENTRIES_FIELD) as usize;
|
|
let block_size = read_be_u32(&hdr, DYNAMIC_BLOCK_SIZE_FIELD) as u64;
|
|
|
|
inner.seek(SeekFrom::Start(bat_offset))?;
|
|
let mut raw = vec![0u8; max_entries * 4];
|
|
inner.read_exact(&mut raw)?;
|
|
let bat: Vec<u32> = (0..max_entries).map(|i| read_be_u32(&raw, i * 4)).collect();
|
|
|
|
Ok((VhdLayout::Sparse { bat, block_size }, max_entries as u64 * block_size))
|
|
}
|
|
|
|
/// Read bytes from a virtual offset according to this layout.
|
|
fn read_at<R: Read + Seek>(
|
|
&self,
|
|
inner: &mut R,
|
|
virt_off: u64,
|
|
virtual_size: u64,
|
|
buf: &mut [u8],
|
|
) -> io::Result<usize> {
|
|
if virt_off >= virtual_size {
|
|
return Ok(0);
|
|
}
|
|
let cap = std::cmp::min(buf.len() as u64, virtual_size - virt_off) as usize;
|
|
|
|
match self {
|
|
VhdLayout::Fixed => {
|
|
inner.seek(SeekFrom::Start(virt_off))?;
|
|
inner.read(&mut buf[..cap])
|
|
}
|
|
VhdLayout::Sparse { bat, block_size } => {
|
|
let bi = (virt_off / block_size) as usize;
|
|
let bo = virt_off % block_size;
|
|
let n = std::cmp::min(cap, (block_size - bo) as usize);
|
|
|
|
if bi >= bat.len() || bat[bi] == BAT_UNUSED {
|
|
buf[..n].fill(0);
|
|
Ok(n)
|
|
} else {
|
|
// Each block: bitmap sector + data. Skip bitmap.
|
|
let file_off = bat[bi] as u64 * SECTOR_SIZE + SECTOR_SIZE + bo;
|
|
inner.seek(SeekFrom::Start(file_off))?;
|
|
inner.read(&mut buf[..n])
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
/// Read 4 bytes from a virtual offset (for magic-byte probing).
|
|
fn read_magic<R: Read + Seek>(
|
|
&self,
|
|
inner: &mut R,
|
|
offset: u64,
|
|
) -> io::Result<[u8; 4]> {
|
|
let mut buf = [0u8; 4];
|
|
match self {
|
|
VhdLayout::Fixed => {
|
|
inner.seek(SeekFrom::Start(offset))?;
|
|
inner.read_exact(&mut buf)?;
|
|
}
|
|
VhdLayout::Sparse { bat, block_size } => {
|
|
let bi = (offset / block_size) as usize;
|
|
if bi < bat.len() && bat[bi] != BAT_UNUSED {
|
|
let file_off = bat[bi] as u64 * SECTOR_SIZE + SECTOR_SIZE + offset % block_size;
|
|
inner.seek(SeekFrom::Start(file_off))?;
|
|
inner.read_exact(&mut buf)?;
|
|
}
|
|
}
|
|
}
|
|
Ok(buf)
|
|
}
|
|
}
|
|
|
|
// ---------------------------------------------------------------------------
|
|
// VHD reader (single VHD)
|
|
// ---------------------------------------------------------------------------
|
|
|
|
/// Transparently presents the NTFS partition within a fixed or dynamic VHD.
|
|
pub struct VhdReader<R> {
|
|
inner: R,
|
|
layout: VhdLayout,
|
|
ntfs_offset: u64,
|
|
virtual_size: u64,
|
|
pos: u64,
|
|
}
|
|
|
|
impl<R: Read + Seek> VhdReader<R> {
|
|
pub fn new(mut inner: R) -> Result<Self, VhdError> {
|
|
let file_size = inner.seek(SeekFrom::End(0))?;
|
|
if file_size < SECTOR_SIZE {
|
|
return Err(VhdError::InvalidCookie);
|
|
}
|
|
|
|
inner.seek(SeekFrom::Start(file_size - SECTOR_SIZE))?;
|
|
let mut footer = [0u8; SECTOR_SIZE as usize];
|
|
inner.read_exact(&mut footer)?;
|
|
if &footer[..8] != VHD_COOKIE {
|
|
return Err(VhdError::InvalidCookie);
|
|
}
|
|
|
|
let disk_type = read_be_u32(&footer, VHD_FOOTER_DISK_TYPE_OFFSET);
|
|
let (layout, virtual_size) = match disk_type {
|
|
VHD_TYPE_FIXED => (VhdLayout::Fixed, file_size - SECTOR_SIZE),
|
|
VHD_TYPE_DYNAMIC | VHD_TYPE_DIFFERENCING => {
|
|
VhdLayout::parse_sparse(&mut inner, &footer)?
|
|
}
|
|
t => return Err(VhdError::UnsupportedType(t)),
|
|
};
|
|
|
|
let ntfs_offset = find_ntfs_offset(&mut inner, &layout, virtual_size)?;
|
|
Ok(Self { inner, layout, ntfs_offset, virtual_size, pos: 0 })
|
|
}
|
|
|
|
fn ntfs_size(&self) -> u64 {
|
|
self.virtual_size - self.ntfs_offset
|
|
}
|
|
}
|
|
|
|
impl<R: Read + Seek> Read for VhdReader<R> {
|
|
fn read(&mut self, buf: &mut [u8]) -> io::Result<usize> {
|
|
let remaining = self.ntfs_size().saturating_sub(self.pos);
|
|
if remaining == 0 {
|
|
return Ok(0);
|
|
}
|
|
let cap = std::cmp::min(buf.len() as u64, remaining) as usize;
|
|
let n = self.layout.read_at(
|
|
&mut self.inner, self.ntfs_offset + self.pos, self.virtual_size, &mut buf[..cap],
|
|
)?;
|
|
self.pos += n as u64;
|
|
Ok(n)
|
|
}
|
|
}
|
|
|
|
impl<R: Read + Seek> Seek for VhdReader<R> {
|
|
fn seek(&mut self, pos: SeekFrom) -> io::Result<u64> {
|
|
let target = match pos {
|
|
SeekFrom::Start(o) => o as i64,
|
|
SeekFrom::Current(o) => self.pos as i64 + o,
|
|
SeekFrom::End(o) => self.ntfs_size() as i64 + o,
|
|
};
|
|
if target < 0 {
|
|
return Err(io::Error::new(io::ErrorKind::InvalidInput, "seek before start"));
|
|
}
|
|
self.pos = target as u64;
|
|
Ok(self.pos)
|
|
}
|
|
}
|
|
|
|
// ---------------------------------------------------------------------------
|
|
// Chained VHD reader (base + N deltas overlaid, no on-disk merge needed)
|
|
// ---------------------------------------------------------------------------
|
|
|
|
/// One layer of a VHD chain: a file handle plus its parsed layout.
|
|
struct VhdLayer<R> {
|
|
inner: R,
|
|
layout: VhdLayout,
|
|
}
|
|
|
|
/// Reads from a chain of VHDs where `layers[0]` is the base (dynamic/fixed)
|
|
/// and `layers[1..]` are differencing VHDs in parent→child order.
|
|
///
|
|
/// For each read, walks layers from top delta down to base. At each layer,
|
|
/// if the sector is present-and-modified (BAT allocated + bitmap bit set),
|
|
/// that layer's bytes win; otherwise the read falls through to the layer
|
|
/// below. The base layer's own `read_at` handles zero-fill for unallocated
|
|
/// dynamic blocks.
|
|
pub struct ChainedVhdReader<R> {
|
|
layers: Vec<VhdLayer<R>>,
|
|
ntfs_offset: u64,
|
|
virtual_size: u64,
|
|
pos: u64,
|
|
}
|
|
|
|
impl<R: Read + Seek> ChainedVhdReader<R> {
|
|
/// Build a chain reader. `readers` must be ordered base-first, top-most delta last.
|
|
pub fn new(readers: Vec<R>) -> Result<Self, VhdError> {
|
|
if readers.is_empty() {
|
|
return Err(VhdError::InvalidCookie);
|
|
}
|
|
|
|
let mut layers: Vec<VhdLayer<R>> = Vec::with_capacity(readers.len());
|
|
let mut virtual_size = 0u64;
|
|
|
|
for (idx, mut r) in readers.into_iter().enumerate() {
|
|
let file_size = r.seek(SeekFrom::End(0))?;
|
|
if file_size < SECTOR_SIZE {
|
|
return Err(VhdError::InvalidCookie);
|
|
}
|
|
r.seek(SeekFrom::Start(file_size - SECTOR_SIZE))?;
|
|
let mut footer = [0u8; SECTOR_SIZE as usize];
|
|
r.read_exact(&mut footer)?;
|
|
if &footer[..8] != VHD_COOKIE {
|
|
return Err(VhdError::InvalidCookie);
|
|
}
|
|
let disk_type = read_be_u32(&footer, VHD_FOOTER_DISK_TYPE_OFFSET);
|
|
let (layout, vsize) = match disk_type {
|
|
VHD_TYPE_FIXED => (VhdLayout::Fixed, file_size - SECTOR_SIZE),
|
|
VHD_TYPE_DYNAMIC | VHD_TYPE_DIFFERENCING => {
|
|
VhdLayout::parse_sparse(&mut r, &footer)?
|
|
}
|
|
t => return Err(VhdError::UnsupportedType(t)),
|
|
};
|
|
if idx == 0 {
|
|
virtual_size = vsize;
|
|
}
|
|
layers.push(VhdLayer { inner: r, layout });
|
|
}
|
|
|
|
let ntfs_offset = find_ntfs_offset_chain(&mut layers, virtual_size)?;
|
|
|
|
Ok(Self { layers, ntfs_offset, virtual_size, pos: 0 })
|
|
}
|
|
|
|
fn ntfs_size(&self) -> u64 {
|
|
self.virtual_size - self.ntfs_offset
|
|
}
|
|
}
|
|
|
|
/// If `layer` has the sector for `virt_off` present AND marked modified in
|
|
/// its bitmap, read from it and return `Some(bytes_read)`. Otherwise `None`
|
|
/// signals "fall through to the layer below".
|
|
///
|
|
/// Reads are capped at the current sector boundary. The VHD bitmap is
|
|
/// per-sector: a single block can have a mixed 1/0 pattern, so a larger read
|
|
/// might cross a sector that belongs to a different layer. The Read
|
|
/// implementation loops until `buf` is filled, amortising the extra calls.
|
|
fn try_read_from_layer<R: Read + Seek>(
|
|
layer: &mut VhdLayer<R>,
|
|
virt_off: u64,
|
|
buf: &mut [u8],
|
|
) -> io::Result<Option<usize>> {
|
|
match &layer.layout {
|
|
VhdLayout::Fixed => Ok(None), // Fixed deltas make no sense; fall through.
|
|
VhdLayout::Sparse { bat, block_size } => {
|
|
let bi = (virt_off / block_size) as usize;
|
|
let bo = virt_off % block_size;
|
|
if bi >= bat.len() || bat[bi] == BAT_UNUSED {
|
|
return Ok(None);
|
|
}
|
|
|
|
// Cap at the current sector to honour per-sector bitmap semantics.
|
|
let sector_remaining = (SECTOR_SIZE - (virt_off % SECTOR_SIZE)) as usize;
|
|
let n = std::cmp::min(buf.len(), sector_remaining);
|
|
let block_file_offset = bat[bi] as u64 * SECTOR_SIZE;
|
|
|
|
// Read the block's bitmap sector.
|
|
layer.inner.seek(SeekFrom::Start(block_file_offset))?;
|
|
let mut bitmap = [0u8; SECTOR_SIZE as usize];
|
|
layer.inner.read_exact(&mut bitmap)?;
|
|
|
|
let sector_in_block = (bo / SECTOR_SIZE) as usize;
|
|
let bitmap_byte = bitmap[sector_in_block / 8];
|
|
let bitmap_bit = 7 - (sector_in_block % 8); // MSB first
|
|
if (bitmap_byte >> bitmap_bit) & 1 == 0 {
|
|
return Ok(None);
|
|
}
|
|
|
|
let file_off = block_file_offset + SECTOR_SIZE + bo;
|
|
layer.inner.seek(SeekFrom::Start(file_off))?;
|
|
let got = layer.inner.read(&mut buf[..n])?;
|
|
Ok(Some(got))
|
|
}
|
|
}
|
|
}
|
|
|
|
/// Walk layers top-to-bottom; first layer that owns the sector wins.
|
|
/// The base layer (index 0) always answers (possibly with zeros for
|
|
/// unallocated dynamic blocks).
|
|
fn read_chain<R: Read + Seek>(
|
|
layers: &mut [VhdLayer<R>],
|
|
vsize: u64,
|
|
virt_off: u64,
|
|
buf: &mut [u8],
|
|
) -> io::Result<usize> {
|
|
if virt_off >= vsize {
|
|
return Ok(0);
|
|
}
|
|
let cap = std::cmp::min(buf.len() as u64, vsize - virt_off) as usize;
|
|
|
|
// Try deltas from top (last) down to just above base (index 1).
|
|
for i in (1..layers.len()).rev() {
|
|
if let Some(n) = try_read_from_layer(&mut layers[i], virt_off, &mut buf[..cap])? {
|
|
return Ok(n);
|
|
}
|
|
}
|
|
// Fall through to base.
|
|
let base = &mut layers[0];
|
|
base.layout.read_at(&mut base.inner, virt_off, vsize, &mut buf[..cap])
|
|
}
|
|
|
|
impl<R: Read + Seek> Read for ChainedVhdReader<R> {
|
|
fn read(&mut self, buf: &mut [u8]) -> io::Result<usize> {
|
|
let remaining = self.ntfs_size().saturating_sub(self.pos);
|
|
if remaining == 0 {
|
|
return Ok(0);
|
|
}
|
|
let cap = std::cmp::min(buf.len() as u64, remaining) as usize;
|
|
let virt_off = self.ntfs_offset + self.pos;
|
|
let n = read_chain(&mut self.layers, self.virtual_size, virt_off, &mut buf[..cap])?;
|
|
self.pos += n as u64;
|
|
Ok(n)
|
|
}
|
|
}
|
|
|
|
impl<R: Read + Seek> Seek for ChainedVhdReader<R> {
|
|
fn seek(&mut self, pos: SeekFrom) -> io::Result<u64> {
|
|
let target = match pos {
|
|
SeekFrom::Start(o) => o as i64,
|
|
SeekFrom::Current(o) => self.pos as i64 + o,
|
|
SeekFrom::End(o) => self.ntfs_size() as i64 + o,
|
|
};
|
|
if target < 0 {
|
|
return Err(io::Error::new(io::ErrorKind::InvalidInput, "seek before start"));
|
|
}
|
|
self.pos = target as u64;
|
|
Ok(self.pos)
|
|
}
|
|
}
|
|
|
|
// ---------------------------------------------------------------------------
|
|
// NTFS partition detection (shared between single and merged readers)
|
|
// ---------------------------------------------------------------------------
|
|
|
|
/// Find NTFS offset in a single VHD.
|
|
fn find_ntfs_offset<R: Read + Seek>(
|
|
inner: &mut R,
|
|
layout: &VhdLayout,
|
|
vsize: u64,
|
|
) -> Result<u64, VhdError> {
|
|
// Try MBR
|
|
if vsize >= SECTOR_SIZE {
|
|
let mut mbr = [0u8; SECTOR_SIZE as usize];
|
|
let _ = layout.read_at(inner, 0, vsize, &mut mbr);
|
|
|
|
if mbr[510..512] == MBR_SIGNATURE {
|
|
for i in 0..MBR_MAX_PARTITIONS {
|
|
let eo = MBR_PARTITION_TABLE_OFFSET + i * MBR_PARTITION_ENTRY_SIZE;
|
|
if mbr[eo + 4] == NTFS_PARTITION_TYPE {
|
|
let lba = u32::from_le_bytes(mbr[eo + 8..eo + 12].try_into().unwrap());
|
|
let offset = lba as u64 * SECTOR_SIZE;
|
|
if offset + 4 <= vsize && layout.read_magic(inner, offset)? == NTFS_MAGIC {
|
|
return Ok(offset);
|
|
}
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
// Probe common offsets
|
|
for offset in NTFS_PROBE_OFFSETS {
|
|
if offset + 4 <= vsize && layout.read_magic(inner, offset)? == NTFS_MAGIC {
|
|
return Ok(offset);
|
|
}
|
|
}
|
|
|
|
Err(VhdError::NoNtfsPartition)
|
|
}
|
|
|
|
/// Find NTFS offset in a chained view (base + N deltas).
|
|
fn find_ntfs_offset_chain<R: Read + Seek>(
|
|
layers: &mut [VhdLayer<R>],
|
|
vsize: u64,
|
|
) -> Result<u64, VhdError> {
|
|
// If the chain has only a base, defer to the single-VHD finder — it's simpler
|
|
// and avoids the bitmap machinery for a pure dynamic/fixed disk.
|
|
if layers.len() == 1 {
|
|
let base = &mut layers[0];
|
|
return find_ntfs_offset(&mut base.inner, &base.layout, vsize);
|
|
}
|
|
|
|
let read_magic = |layers: &mut [VhdLayer<R>], offset: u64| -> io::Result<[u8; 4]> {
|
|
let mut buf = [0u8; 4];
|
|
read_chain(layers, vsize, offset, &mut buf)?;
|
|
Ok(buf)
|
|
};
|
|
|
|
// Try MBR from merged view.
|
|
if vsize >= SECTOR_SIZE {
|
|
let mut mbr = [0u8; SECTOR_SIZE as usize];
|
|
read_chain(layers, vsize, 0, &mut mbr)?;
|
|
|
|
if mbr[510..512] == MBR_SIGNATURE {
|
|
for i in 0..MBR_MAX_PARTITIONS {
|
|
let eo = MBR_PARTITION_TABLE_OFFSET + i * MBR_PARTITION_ENTRY_SIZE;
|
|
if mbr[eo + 4] == NTFS_PARTITION_TYPE {
|
|
let lba = u32::from_le_bytes(mbr[eo + 8..eo + 12].try_into().unwrap());
|
|
let offset = lba as u64 * SECTOR_SIZE;
|
|
if offset + 4 <= vsize && read_magic(layers, offset)? == NTFS_MAGIC {
|
|
return Ok(offset);
|
|
}
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
for offset in NTFS_PROBE_OFFSETS {
|
|
if offset + 4 <= vsize && read_magic(layers, offset)? == NTFS_MAGIC {
|
|
return Ok(offset);
|
|
}
|
|
}
|
|
|
|
Err(VhdError::NoNtfsPartition)
|
|
}
|
|
|
|
// ---------------------------------------------------------------------------
|
|
// NTFS extraction (shared logic)
|
|
// ---------------------------------------------------------------------------
|
|
|
|
fn is_ntfs_system_entry(name: &str) -> bool {
|
|
name.starts_with('$') || name == "." || name == ".." || name == "System Volume Information"
|
|
}
|
|
|
|
/// Whether a directory-index entry should be skipped when extracting.
|
|
///
|
|
/// Besides NTFS system metadata, this skips DOS (8.3) short-name aliases: a file
|
|
/// that has a separate short name appears in the index *twice* — once with its
|
|
/// Win32 long name and once with the `Dos` short name. Without this, every such
|
|
/// entry is extracted a second time under its mangled `NAME~1.EXT` name (and
|
|
/// short-named directories get their whole subtree duplicated).
|
|
fn skip_index_entry(namespace: NtfsFileNamespace, name: &str) -> bool {
|
|
namespace == NtfsFileNamespace::Dos || is_ntfs_system_entry(name)
|
|
}
|
|
|
|
fn ntfs_time_to_system_time(t: NtfsTime) -> SystemTime {
|
|
let nanos = (t.nt_timestamp() - WINDOWS_EPOCH_OFFSET) * 100;
|
|
SystemTime::UNIX_EPOCH + Duration::from_nanos(nanos)
|
|
}
|
|
|
|
fn set_ntfs_timestamps<T: Read + Seek>(fs: &mut T, file: &ntfs::NtfsFile, path: &Path) {
|
|
let mut attrs = file.attributes();
|
|
while let Some(Ok(attr)) = attrs.next(fs) {
|
|
if let Ok(attr) = attr.to_attribute() {
|
|
if let Ok(NtfsAttributeType::StandardInformation) = attr.ty() {
|
|
if let Ok(info) = attr.resident_structured_value::<NtfsStandardInformation>() {
|
|
let _ = OpenOptions::new().write(true).open(path).and_then(|h| {
|
|
h.set_times(
|
|
FileTimes::new()
|
|
.set_accessed(ntfs_time_to_system_time(info.access_time()))
|
|
.set_modified(ntfs_time_to_system_time(info.modification_time())),
|
|
)
|
|
});
|
|
}
|
|
break;
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
fn extract_ntfs_dir<T: Read + Seek>(
|
|
ntfs: &Ntfs,
|
|
fs: &mut T,
|
|
dir: &ntfs::NtfsFile,
|
|
out: &Path,
|
|
pb: &ProgressBar,
|
|
) -> Result<()> {
|
|
let index = dir.directory_index(fs)?;
|
|
let mut iter = index.entries();
|
|
|
|
while let Some(entry) = iter.next(fs) {
|
|
let entry = entry?;
|
|
let key = entry.key().ok_or_else(|| anyhow!("missing key"))??;
|
|
let name = key.name().to_string_lossy();
|
|
if skip_index_entry(key.namespace(), &name) {
|
|
continue;
|
|
}
|
|
|
|
let file = entry.to_file(ntfs, fs)?;
|
|
let dest = out.join(&*name);
|
|
|
|
if key.is_directory() {
|
|
create_dir_all(&dest)?;
|
|
extract_ntfs_dir(ntfs, fs, &file, &dest, pb)?;
|
|
set_ntfs_timestamps(fs, &file, &dest);
|
|
} else if let Some(data) = file.data(fs, "") {
|
|
let data_item = data?;
|
|
let attr = data_item.to_attribute()?;
|
|
let mut reader = BufReader::with_capacity(BUF_SIZE, attr.value(fs)?.attach(fs));
|
|
let mut out_file = File::create(&dest)?;
|
|
|
|
loop {
|
|
let buf = reader.fill_buf()?;
|
|
if buf.is_empty() {
|
|
break;
|
|
}
|
|
out_file.write_all(buf)?;
|
|
let n = buf.len();
|
|
reader.consume(n);
|
|
pb.inc(n as u64);
|
|
}
|
|
out_file.flush()?;
|
|
drop(reader);
|
|
set_ntfs_timestamps(fs, &file, &dest);
|
|
}
|
|
}
|
|
Ok(())
|
|
}
|
|
|
|
fn calculate_ntfs_size<T: Read + Seek>(
|
|
ntfs: &Ntfs,
|
|
fs: &mut T,
|
|
dir: &ntfs::NtfsFile,
|
|
) -> Result<u64> {
|
|
let mut total = 0u64;
|
|
let index = dir.directory_index(fs)?;
|
|
let mut iter = index.entries();
|
|
|
|
while let Some(entry) = iter.next(fs) {
|
|
let entry = entry?;
|
|
let key = entry.key().ok_or_else(|| anyhow!("missing key"))??;
|
|
if skip_index_entry(key.namespace(), key.name().to_string_lossy().as_ref()) {
|
|
continue;
|
|
}
|
|
let file = entry.to_file(ntfs, fs)?;
|
|
if key.is_directory() {
|
|
total += calculate_ntfs_size(ntfs, fs, &file)?;
|
|
} else if let Some(data) = file.data(fs, "") {
|
|
total += data?.to_attribute()?.value_length();
|
|
}
|
|
}
|
|
Ok(total)
|
|
}
|
|
|
|
/// Shared extraction logic: given an NTFS-bearing Read+Seek, extract to output_dir.
|
|
fn extract_ntfs_to_dir<T: Read + Seek>(fs: &mut T, output_dir: &Path, prefix: &str) -> Result<()> {
|
|
let mut ntfs = Ntfs::new(fs)?;
|
|
ntfs.read_upcase_table(fs)?;
|
|
|
|
let root = ntfs.root_directory(fs)?;
|
|
let total = calculate_ntfs_size(&ntfs, fs, &root)?;
|
|
|
|
let pb = ProgressBar::new(total)
|
|
.with_style(ProgressStyle::default_bar().template(PROGRESS_STYLE)?);
|
|
pb.set_prefix(prefix.to_string());
|
|
|
|
create_dir_all(output_dir)?;
|
|
let root = ntfs.root_directory(fs)?;
|
|
extract_ntfs_dir(&ntfs, fs, &root, output_dir, &pb)?;
|
|
pb.finish();
|
|
|
|
Ok(())
|
|
}
|
|
|
|
// ---------------------------------------------------------------------------
|
|
// Public API
|
|
// ---------------------------------------------------------------------------
|
|
|
|
/// Extract all files from a single VHD's NTFS filesystem, then delete the VHD.
|
|
/// The `output_dir` is where files are extracted to.
|
|
pub fn extract_vhd(vhd_path: &Path, output_dir: &Path) -> Result<()> {
|
|
println!("Extracting VHD: {}", vhd_path.display());
|
|
|
|
let mut vhd = VhdReader::new(File::open(vhd_path)?).map_err(|e| anyhow!(e))?;
|
|
let prefix = output_dir.file_name().unwrap_or_default().to_string_lossy().to_string();
|
|
extract_ntfs_to_dir(&mut vhd, output_dir, &prefix)?;
|
|
|
|
println!("Extracted to: {}", output_dir.display());
|
|
|
|
drop(vhd);
|
|
if let Err(e) = std::fs::remove_file(vhd_path) {
|
|
println!("WARNING: Could not delete VHD: {e}");
|
|
}
|
|
Ok(())
|
|
}
|
|
|
|
/// Extract files from a chained view of a base + N differencing VHDs.
|
|
///
|
|
/// `chain` must be ordered base-first, top-most delta last. A chain of length 1
|
|
/// is equivalent to extracting just the base. Unlike [`extract_vhd`], this does
|
|
/// **not** delete the inputs — the caller is responsible, since a single VHD
|
|
/// in a chain is typically consumed by multiple extractions (one per patch
|
|
/// level) and must not be removed until all of them have completed.
|
|
pub fn extract_chained_vhd(chain: &[&Path], output_dir: &Path) -> Result<()> {
|
|
if chain.is_empty() {
|
|
return Err(anyhow!("extract_chained_vhd: empty chain"));
|
|
}
|
|
|
|
let paths_disp = chain
|
|
.iter()
|
|
.map(|p| p.display().to_string())
|
|
.collect::<Vec<_>>()
|
|
.join(" + ");
|
|
println!("Extracting chained VHD: {paths_disp}");
|
|
|
|
let readers: Vec<File> = chain
|
|
.iter()
|
|
.map(|p| File::open(p))
|
|
.collect::<io::Result<_>>()?;
|
|
let mut reader = ChainedVhdReader::new(readers).map_err(|e| anyhow!(e))?;
|
|
|
|
let prefix = output_dir.file_name().unwrap_or_default().to_string_lossy().to_string();
|
|
extract_ntfs_to_dir(&mut reader, output_dir, &prefix)?;
|
|
|
|
println!("Extracted to: {}", output_dir.display());
|
|
|
|
Ok(())
|
|
}
|
|
|
|
// ---------------------------------------------------------------------------
|
|
// Helpers
|
|
// ---------------------------------------------------------------------------
|
|
|
|
fn read_be_u32(buf: &[u8], offset: usize) -> u32 {
|
|
u32::from_be_bytes(buf[offset..offset + 4].try_into().unwrap())
|
|
}
|
|
|
|
fn read_be_u64(buf: &[u8], offset: usize) -> u64 {
|
|
u64::from_be_bytes(buf[offset..offset + 8].try_into().unwrap())
|
|
}
|
|
|
|
#[cfg(test)]
|
|
mod tests {
|
|
use super::skip_index_entry;
|
|
use ntfs::structured_values::NtfsFileNamespace;
|
|
|
|
#[test]
|
|
fn skips_dos_short_name_aliases() {
|
|
// 8.3 aliases duplicate a Win32 entry and must not be extracted again.
|
|
assert!(skip_index_entry(NtfsFileNamespace::Dos, "OXGETH~1.EXE"));
|
|
assert!(skip_index_entry(NtfsFileNamespace::Dos, "PROGRA~1"));
|
|
}
|
|
|
|
#[test]
|
|
fn keeps_long_and_native_names() {
|
|
assert!(!skip_index_entry(NtfsFileNamespace::Win32, "oxGetHwInfo.exe"));
|
|
// A name that is its own short name (no separate Dos entry) is kept.
|
|
assert!(!skip_index_entry(NtfsFileNamespace::Win32AndDos, "game.bat"));
|
|
assert!(!skip_index_entry(NtfsFileNamespace::Posix, "readme"));
|
|
}
|
|
|
|
#[test]
|
|
fn skips_system_entries_regardless_of_namespace() {
|
|
assert!(skip_index_entry(NtfsFileNamespace::Win32, "$MFT"));
|
|
assert!(skip_index_entry(NtfsFileNamespace::Win32, "System Volume Information"));
|
|
assert!(skip_index_entry(NtfsFileNamespace::Win32, "."));
|
|
assert!(skip_index_entry(NtfsFileNamespace::Win32, ".."));
|
|
}
|
|
}
|