From bad3e111b58acb4f11b66b632f2b04ccce0985df Mon Sep 17 00:00:00 2001 From: John Starks Date: Wed, 6 Apr 2016 13:44:28 -0700 Subject: [PATCH 1/6] LZX: Fix unaligned block failure This fixes the case where an unaligned block lands on a 16-bit boundary. --- wim/lzx/lzx.go | 33 ++++++++++++++++++++------------- 1 file changed, 20 insertions(+), 13 deletions(-) diff --git a/wim/lzx/lzx.go b/wim/lzx/lzx.go index 7addf45..2e9f851 100644 --- a/wim/lzx/lzx.go +++ b/wim/lzx/lzx.go @@ -21,7 +21,7 @@ const ( maxBlockSize = 32768 windowSize = 32768 - treePathLenCount = 17 + maxTreePathLen = 16 e8filesize = 12000000 maxe8offset = 0x3fffffff @@ -97,8 +97,8 @@ func (f *decompressor) feed() bool { // getBits retrieves the next n bits from the byte stream. n // must be <= 16. It sets f.err on error. func (f *decompressor) getBits(n byte) uint16 { - if f.nbits < 16 { - if !f.feed() && n > f.nbits { + if f.nbits < n { + if !f.feed() { f.err = io.ErrUnexpectedEOF } } @@ -115,12 +115,12 @@ type huffman struct { } // buildTable builds a huffman decoding table from a slice of code lengths, -// one per code, in order. Each code length must be less than treePathLenCount. +// one per code, in order. Each code length must be <= maxTreePathLen. // See https://en.wikipedia.org/wiki/Canonical_Huffman_code. func buildTable(codelens []byte) *huffman { // Determine the number of codes of each length, and the // maximum length. - var count [treePathLenCount]uint + var count [maxTreePathLen + 1]uint var max byte for _, cl := range codelens { count[cl]++ @@ -134,7 +134,7 @@ func buildTable(codelens []byte) *huffman { } // Determine the first code of each length. - var first [treePathLenCount]uint + var first [maxTreePathLen + 1]uint code := uint(0) for i := byte(1); i <= max; i++ { code <<= 1 @@ -178,7 +178,7 @@ func (f *decompressor) getCode(h *huffman) uint16 { f.err = errCorrupt return 0 } - if f.nbits < 16 { + if f.nbits < maxTreePathLen { f.feed() } // For codes with length < h.maxbits, it doesn't matter @@ -309,14 +309,21 @@ func (f *decompressor) readBlockHeader() (byte, uint16, error) { case verbatimBlock, alignedOffsetBlock: // The caller will read the huffman trees. case uncompressedBlock: - // Not sure if this can happen... - if f.nbits > 16 || f.nbits == 0 { - return 0, 0, errCorrupt + if f.nbits > 16 { + panic("impossible: more than one 16-bit word remains") } - // Drop the remaining bits in the current 16-bit word. - f.nbits = 0 - f.c = 0 + // Drop the remaining bits in the current 16-bit word + // If there are no bits left, discard a full 16-bit word. + n := f.nbits + if n == 0 { + n = 16 + } + + f.getBits(n) + if f.err != nil { + return 0, 0, f.err + } // Read the LRU values for the next block. var lru [12]byte From 3cedae2e26c9739f98b9ca713db6ef3797f6dbf1 Mon Sep 17 00:00:00 2001 From: John Starks Date: Wed, 6 Apr 2016 13:56:25 -0700 Subject: [PATCH 2/6] WIM: Fix handling of reparse points Reparse point data is not always in the first stream, sometimes it is in the file data itself. Remove ReparseStream and always put it in the file data. --- wim/wim.go | 163 ++++++++++++++++++++++++++++++----------------------- 1 file changed, 94 insertions(+), 69 deletions(-) diff --git a/wim/wim.go b/wim/wim.go index ffd2ee5..dca93ab 100644 --- a/wim/wim.go +++ b/wim/wim.go @@ -153,11 +153,15 @@ const streamentrySize = 38 // ParseError is returned when the WIM cannot be parsed. type ParseError struct { Oper string + Path string Err error } func (e *ParseError) Error() string { - return "WIM parse error at " + e.Oper + ": " + e.Err.Error() + if e.Path == "" { + return "WIM parse error at " + e.Oper + ": " + e.Err.Error() + } + return fmt.Sprintf("WIM parse error: %s %s: %s", e.Oper, e.Path, e.Err.Error()) } // Reader provides functions to read a WIM file. @@ -205,7 +209,6 @@ type FileHeader struct { LinkID int64 ReparseTag uint32 ReparseReserved uint32 - ReparseStream *Stream } // File represents a file or directory in a WIM image. @@ -227,7 +230,7 @@ func NewReader(f io.ReaderAt) (*Reader, error) { } if r.hdr.ImageTag != wimImageTag { - return nil, &ParseError{"image tag", errors.New("not a WIM file")} + return nil, &ParseError{Oper: "image tag", Err: errors.New("not a WIM file")} } if r.hdr.Flags&^supportedHdrFlags != 0 { @@ -295,12 +298,12 @@ func (r *Reader) ReadXML() (string, error) { XMLData := make([]uint16, r.hdr.XMLData.OriginalSize/2) err = binary.Read(rsrc, binary.LittleEndian, XMLData) if err != nil { - return "", &ParseError{"XML data", err} + return "", &ParseError{Oper: "XML data", Err: err} } // The BOM will always indicate little-endian UTF-16. if XMLData[0] != 0xfeff { - return "", &ParseError{"XML data", errors.New("invalid BOM")} + return "", &ParseError{Oper: "XML data", Err: errors.New("invalid BOM")} } return string(utf16.Decode(XMLData[1:])), nil } @@ -311,39 +314,39 @@ func (r *Reader) readOffsetTable(res *resourceDescriptor) (map[SHA1Hash]resource offsetTable, err := r.readResource(res) if err != nil { - return nil, nil, &ParseError{"offset table", err} + return nil, nil, &ParseError{Oper: "offset table", Err: err} } br := bytes.NewReader(offsetTable) - for { + for i := 0; ; i++ { var res streamDescriptor err := binary.Read(br, binary.LittleEndian, &res) if err == io.EOF { break } if err != nil { - return nil, nil, &ParseError{"offset table", err} + return nil, nil, &ParseError{Oper: "offset table", Err: err} } if res.Flags()&^supportedResFlags != 0 { - return nil, nil, &ParseError{"offset table", errors.New("unsupported resource flag")} + return nil, nil, &ParseError{Oper: "offset table", Err: errors.New("unsupported resource flag")} } // Validation for ad-hoc testing if validate { sec, err := r.resourceReader(&res.resourceDescriptor) if err != nil { - return nil, nil, err + panic(fmt.Sprint(i, err)) } hash := sha1.New() _, err = io.Copy(hash, sec) sec.Close() if err != nil { - return nil, nil, err + panic(fmt.Sprint(i, err)) } var cmphash SHA1Hash copy(cmphash[:], hash.Sum(nil)) if cmphash != res.Hash { - return nil, nil, errors.New("hash mismatch") + panic(fmt.Sprint(i, "hash mismatch")) } } @@ -359,7 +362,7 @@ func (r *Reader) readOffsetTable(res *resourceDescriptor) (map[SHA1Hash]resource } if len(images) != int(r.hdr.ImageCount) { - return nil, nil, &ParseError{"offset table", errors.New("mismatched image count")} + return nil, nil, &ParseError{Oper: "offset table", Err: errors.New("mismatched image count")} } return fileData, images, nil @@ -369,7 +372,7 @@ func (r *Reader) readSecurityDescriptors(rsrc io.Reader) (sds [][]byte, n int64, var secBlock securityblockDisk err = binary.Read(rsrc, binary.LittleEndian, &secBlock) if err != nil { - err = &ParseError{"security table", err} + err = &ParseError{Oper: "security table", Err: err} return } @@ -378,7 +381,7 @@ func (r *Reader) readSecurityDescriptors(rsrc io.Reader) (sds [][]byte, n int64, secSizes := make([]int64, secBlock.NumEntries) err = binary.Read(rsrc, binary.LittleEndian, &secSizes) if err != nil { - err = &ParseError{"security table sizes", err} + err = &ParseError{Oper: "security table sizes", Err: err} return } @@ -389,7 +392,7 @@ func (r *Reader) readSecurityDescriptors(rsrc io.Reader) (sds [][]byte, n int64, sd := make([]byte, size&0xffffffff) _, err = io.ReadFull(rsrc, sd) if err != nil { - err = &ParseError{"security descriptor", err} + err = &ParseError{Oper: "security descriptor", Err: err} return } n += int64(len(sd)) @@ -398,7 +401,7 @@ func (r *Reader) readSecurityDescriptors(rsrc io.Reader) (sds [][]byte, n int64, secsize := int64((secBlock.TotalLength + 7) &^ 7) if n > secsize { - err = &ParseError{"security descriptor", errors.New("security descriptor table too small")} + err = &ParseError{Oper: "security descriptor", Err: errors.New("security descriptor table too small")} return } @@ -432,7 +435,7 @@ func (img *Image) Open() (*File, error) { return nil, err } if len(f) != 1 { - return nil, &ParseError{"root directory", errors.New("expected exactly 1 root directory entry")} + return nil, &ParseError{Oper: "root directory", Err: errors.New("expected exactly 1 root directory entry")} } return f[0], err } @@ -457,7 +460,7 @@ func (img *Image) readdir(rsrc io.Reader) ([]*File, error) { func (img *Image) readNextEntry(r *bufio.Reader) (*File, error) { lengthBuf, err := r.Peek(8) if err != nil { - return nil, &ParseError{"directory length check", err} + return nil, &ParseError{Oper: "directory length check", Err: err} } left := int(binary.LittleEndian.Uint64(lengthBuf)) @@ -466,24 +469,46 @@ func (img *Image) readNextEntry(r *bufio.Reader) (*File, error) { } if left < direntrySize { - return nil, &ParseError{"directory entry", errors.New("size too short")} + return nil, &ParseError{Oper: "directory entry", Err: errors.New("size too short")} } var dentry direntry err = binary.Read(r, binary.LittleEndian, &dentry) if err != nil { - return nil, &ParseError{"directory entry", err} + return nil, &ParseError{Oper: "directory entry", Err: err} } left -= direntrySize + namesLen := int(dentry.FileNameLength + 2 + dentry.ShortNameLength) + if left < namesLen { + return nil, &ParseError{Oper: "directory entry", Err: errors.New("size too short for names")} + } + + names := make([]uint16, namesLen/2) + err = binary.Read(r, binary.LittleEndian, names) + if err != nil { + return nil, &ParseError{Oper: "file name", Err: err} + } + + left -= namesLen + + var name, shortName string + if dentry.FileNameLength > 0 { + name = string(utf16.Decode(names[:dentry.FileNameLength/2])) + } + + if dentry.ShortNameLength > 0 { + shortName = string(utf16.Decode(names[dentry.FileNameLength/2+1:])) + } + var offset resourceDescriptor zerohash := SHA1Hash{} if dentry.Hash != zerohash { var ok bool offset, ok = img.wim.fileData[dentry.Hash] if !ok { - return nil, &ParseError{"directory entry", fmt.Errorf("could not find file data matching hash %v", dentry.Hash)} + return nil, &ParseError{Oper: "directory entry", Path: name, Err: fmt.Errorf("could not find file data matching hash %#v", dentry)} } } @@ -495,6 +520,8 @@ func (img *Image) readNextEntry(r *bufio.Reader) (*File, error) { LastWriteTime: dentry.LastWriteTime, Hash: dentry.Hash, Size: offset.OriginalSize, + Name: name, + ShortName: shortName, }, offset: offset, @@ -502,38 +529,28 @@ func (img *Image) readNextEntry(r *bufio.Reader) (*File, error) { subdirOffset: dentry.SubdirOffset, } + isDir := false + if dentry.Attributes&syscall.FILE_ATTRIBUTE_REPARSE_POINT == 0 { f.LinkID = dentry.ReparseHardLink + if dentry.Attributes&syscall.FILE_ATTRIBUTE_DIRECTORY != 0 { + isDir = true + } } else { f.ReparseTag = uint32(dentry.ReparseHardLink) f.ReparseReserved = uint32(dentry.ReparseHardLink >> 32) } + if isDir && f.subdirOffset == 0 { + return nil, &ParseError{Oper: "directory entry", Path: name, Err: errors.New("no subdirectory data for directory")} + } else if !isDir && f.subdirOffset != 0 { + return nil, &ParseError{Oper: "directory entry", Path: name, Err: errors.New("unexpected subdirectory data for non-directory")} + } + if dentry.SecurityID != 0xffffffff { f.SecurityDescriptor = img.sds[dentry.SecurityID] } - namesLen := int(dentry.FileNameLength + 2 + dentry.ShortNameLength) - if left < namesLen { - return nil, &ParseError{"directory entry", errors.New("size too short for names")} - } - - names := make([]uint16, namesLen/2) - err = binary.Read(r, binary.LittleEndian, names) - if err != nil { - return nil, &ParseError{"file name", err} - } - - left -= namesLen - - if dentry.FileNameLength > 0 { - f.Name = string(utf16.Decode(names[:dentry.FileNameLength/2])) - } - - if dentry.ShortNameLength > 0 { - f.ShortName = string(utf16.Decode(names[dentry.FileNameLength/2+1:])) - } - _, err = r.Discard(left) if err != nil { return nil, err @@ -546,19 +563,20 @@ func (img *Image) readNextEntry(r *bufio.Reader) (*File, error) { if err != nil { return nil, err } - if !(s.Name == "" && s.Size == 0) { - if dentry.Attributes&syscall.FILE_ATTRIBUTE_REPARSE_POINT != 0 && s.Name == "" { - f.ReparseStream = s - } else { - streams = append(streams, s) - } + // The first unnamed stream should be treated as the file stream. + if i == 0 && s.Name == "" { + f.Hash = s.Hash + f.Size = s.Size + f.offset = s.offset + } else if s.Name != "" { + streams = append(streams, s) } } f.Streams = streams } - if dentry.Attributes&syscall.FILE_ATTRIBUTE_REPARSE_POINT != 0 && f.ReparseStream == nil { - return nil, &ParseError{"directory entry", errors.New("reparse point is missing reparse stream")} + if dentry.Attributes&syscall.FILE_ATTRIBUTE_REPARSE_POINT != 0 && f.Size == 0 { + return nil, &ParseError{Oper: "directory entry", Path: name, Err: errors.New("reparse point is missing reparse stream")} } return f, nil @@ -567,28 +585,41 @@ func (img *Image) readNextEntry(r *bufio.Reader) (*File, error) { func (img *Image) readNextStream(r *bufio.Reader) (*Stream, error) { lengthBuf, err := r.Peek(8) if err != nil { - return nil, &ParseError{"stream length check", err} + return nil, &ParseError{Oper: "stream length check", Err: err} } left := int(binary.LittleEndian.Uint64(lengthBuf)) if left < streamentrySize { - return nil, &ParseError{"stream entry", errors.New("size too short")} + return nil, &ParseError{Oper: "stream entry", Err: errors.New("size too short")} } var sentry streamentry err = binary.Read(r, binary.LittleEndian, &sentry) if err != nil { - return nil, &ParseError{"stream entry", err} + return nil, &ParseError{Oper: "stream entry", Err: err} } left -= streamentrySize + if left < int(sentry.NameLength) { + return nil, &ParseError{Oper: "stream entry", Err: errors.New("size too short for name")} + } + + names := make([]uint16, sentry.NameLength/2) + err = binary.Read(r, binary.LittleEndian, names) + if err != nil { + return nil, &ParseError{Oper: "file name", Err: err} + } + + left -= int(sentry.NameLength) + name := string(utf16.Decode(names)) + var offset resourceDescriptor if sentry.Hash != (SHA1Hash{}) { var ok bool offset, ok = img.wim.fileData[sentry.Hash] if !ok { - return nil, &ParseError{"stream entry", fmt.Errorf("could not find file data matching hash %v", sentry.Hash)} + return nil, &ParseError{Oper: "stream entry", Path: name, Err: fmt.Errorf("could not find file data matching hash %v", sentry.Hash)} } } @@ -596,24 +627,12 @@ func (img *Image) readNextStream(r *bufio.Reader) (*Stream, error) { StreamHeader: StreamHeader{ Hash: sentry.Hash, Size: offset.OriginalSize, + Name: name, }, wim: img.wim, offset: offset, } - if left < int(sentry.NameLength) { - return nil, &ParseError{"stream entry", errors.New("size too short for name")} - } - - names := make([]uint16, sentry.NameLength/2) - err = binary.Read(r, binary.LittleEndian, names) - if err != nil { - return nil, &ParseError{"file name", err} - } - - left -= int(sentry.NameLength) - s.Name = string(utf16.Decode(names)) - _, err = r.Discard(left) if err != nil { return nil, err @@ -634,7 +653,7 @@ func (f *File) Open() (io.ReadCloser, error) { // Readdir reads the directory entries. func (f *File) Readdir() ([]*File, error) { - if f.Attributes&syscall.FILE_ATTRIBUTE_DIRECTORY == 0 { + if !f.IsDir() { return nil, errors.New("not a directory") } rsrc, err := f.img.wim.resourceReaderWithOffset(&f.img.offset, f.subdirOffset) @@ -644,3 +663,9 @@ func (f *File) Readdir() ([]*File, error) { defer rsrc.Close() return f.img.readdir(rsrc) } + +// IsDir returns whether the given file is a directory. It returns false when it +// is a directory reparse point. +func (f *FileHeader) IsDir() bool { + return f.Attributes&(syscall.FILE_ATTRIBUTE_DIRECTORY|syscall.FILE_ATTRIBUTE_REPARSE_POINT) == syscall.FILE_ATTRIBUTE_DIRECTORY +} From 2748a384c002a9637c4b315097825a6d2b4bdcd7 Mon Sep 17 00:00:00 2001 From: John Starks Date: Wed, 6 Apr 2016 13:57:09 -0700 Subject: [PATCH 3/6] WIM: Don't use syscall package This allows the WIM code to build on other OSes. --- wim/wim.go | 65 ++++++++++++++++++++++++++++++++++++++++++------------ 1 file changed, 51 insertions(+), 14 deletions(-) diff --git a/wim/wim.go b/wim/wim.go index dca93ab..78828bf 100644 --- a/wim/wim.go +++ b/wim/wim.go @@ -13,10 +13,32 @@ import ( "fmt" "io" "io/ioutil" - "syscall" + "time" "unicode/utf16" ) +// File attribute constants from Windows. +const ( + FILE_ATTRIBUTE_READONLY = 0x00000001 + FILE_ATTRIBUTE_HIDDEN = 0x00000002 + FILE_ATTRIBUTE_SYSTEM = 0x00000004 + FILE_ATTRIBUTE_DIRECTORY = 0x00000010 + FILE_ATTRIBUTE_ARCHIVE = 0x00000020 + FILE_ATTRIBUTE_DEVICE = 0x00000040 + FILE_ATTRIBUTE_NORMAL = 0x00000080 + FILE_ATTRIBUTE_TEMPORARY = 0x00000100 + FILE_ATTRIBUTE_SPARSE_FILE = 0x00000200 + FILE_ATTRIBUTE_REPARSE_POINT = 0x00000400 + FILE_ATTRIBUTE_COMPRESSED = 0x00000800 + FILE_ATTRIBUTE_OFFLINE = 0x00001000 + FILE_ATTRIBUTE_NOT_CONTENT_INDEXED = 0x00002000 + FILE_ATTRIBUTE_ENCRYPTED = 0x00004000 + FILE_ATTRIBUTE_INTEGRITY_STREAM = 0x00008000 + FILE_ATTRIBUTE_VIRTUAL = 0x00010000 + FILE_ATTRIBUTE_NO_SCRUB_DATA = 0x00020000 + FILE_ATTRIBUTE_EA = 0x00040000 +) + var wimImageTag = [...]byte{'M', 'S', 'W', 'I', 'M', 0, 0, 0} type guid struct { @@ -128,9 +150,9 @@ type direntry struct { SecurityID uint32 SubdirOffset int64 Unused1, Unused2 int64 - CreationTime syscall.Filetime - LastAccessTime syscall.Filetime - LastWriteTime syscall.Filetime + CreationTime filetime + LastAccessTime filetime + LastWriteTime filetime Hash SHA1Hash Padding uint32 ReparseHardLink int64 @@ -150,6 +172,21 @@ type streamentry struct { const streamentrySize = 38 +type filetime struct { + LowDateTime uint32 + HighDateTime uint32 +} + +func (ft *filetime) Time() time.Time { + // 100-nanosecond intervals since January 1, 1601 + nsec := int64(ft.HighDateTime)<<32 + int64(ft.LowDateTime) + // change starting time to the Epoch (00:00:00 UTC, January 1, 1970) + nsec -= 116444736000000000 + // convert into nanoseconds + nsec *= 100 + return time.Unix(0, nsec) +} + // ParseError is returned when the WIM cannot be parsed. type ParseError struct { Oper string @@ -201,9 +238,9 @@ type FileHeader struct { ShortName string Attributes uint32 SecurityDescriptor []byte - CreationTime syscall.Filetime - LastAccessTime syscall.Filetime - LastWriteTime syscall.Filetime + CreationTime time.Time + LastAccessTime time.Time + LastWriteTime time.Time Hash SHA1Hash Size int64 LinkID int64 @@ -515,9 +552,9 @@ func (img *Image) readNextEntry(r *bufio.Reader) (*File, error) { f := &File{ FileHeader: FileHeader{ Attributes: dentry.Attributes, - CreationTime: dentry.CreationTime, - LastAccessTime: dentry.LastAccessTime, - LastWriteTime: dentry.LastWriteTime, + CreationTime: dentry.CreationTime.Time(), + LastAccessTime: dentry.LastAccessTime.Time(), + LastWriteTime: dentry.LastWriteTime.Time(), Hash: dentry.Hash, Size: offset.OriginalSize, Name: name, @@ -531,9 +568,9 @@ func (img *Image) readNextEntry(r *bufio.Reader) (*File, error) { isDir := false - if dentry.Attributes&syscall.FILE_ATTRIBUTE_REPARSE_POINT == 0 { + if dentry.Attributes&FILE_ATTRIBUTE_REPARSE_POINT == 0 { f.LinkID = dentry.ReparseHardLink - if dentry.Attributes&syscall.FILE_ATTRIBUTE_DIRECTORY != 0 { + if dentry.Attributes&FILE_ATTRIBUTE_DIRECTORY != 0 { isDir = true } } else { @@ -575,7 +612,7 @@ func (img *Image) readNextEntry(r *bufio.Reader) (*File, error) { f.Streams = streams } - if dentry.Attributes&syscall.FILE_ATTRIBUTE_REPARSE_POINT != 0 && f.Size == 0 { + if dentry.Attributes&FILE_ATTRIBUTE_REPARSE_POINT != 0 && f.Size == 0 { return nil, &ParseError{Oper: "directory entry", Path: name, Err: errors.New("reparse point is missing reparse stream")} } @@ -667,5 +704,5 @@ func (f *File) Readdir() ([]*File, error) { // IsDir returns whether the given file is a directory. It returns false when it // is a directory reparse point. func (f *FileHeader) IsDir() bool { - return f.Attributes&(syscall.FILE_ATTRIBUTE_DIRECTORY|syscall.FILE_ATTRIBUTE_REPARSE_POINT) == syscall.FILE_ATTRIBUTE_DIRECTORY + return f.Attributes&(FILE_ATTRIBUTE_DIRECTORY|FILE_ATTRIBUTE_REPARSE_POINT) == FILE_ATTRIBUTE_DIRECTORY } From d18f2f3f561c2d41266b875e93deeb54fbee9029 Mon Sep 17 00:00:00 2001 From: John Starks Date: Wed, 6 Apr 2016 13:58:23 -0700 Subject: [PATCH 4/6] Add validate tool to scan all directories in a WIM --- wim/validate/validate.go | 48 ++++++++++++++++++++++++++++++++++++++++ 1 file changed, 48 insertions(+) create mode 100644 wim/validate/validate.go diff --git a/wim/validate/validate.go b/wim/validate/validate.go new file mode 100644 index 0000000..e15ea70 --- /dev/null +++ b/wim/validate/validate.go @@ -0,0 +1,48 @@ +package main + +import ( + "flag" + "fmt" + "os" + + "github.com/Microsoft/go-winio/wim" +) + +func main() { + flag.Parse() + f, err := os.Open(flag.Arg(0)) + if err != nil { + panic(err) + } + + w, err := wim.NewReader(f) + if err != nil { + panic(err) + + } + dir, err := w.Image[0].Open() + if err != nil { + panic(err) + } + + err = recur(dir) + if err != nil { + panic(err) + } +} + +func recur(d *wim.File) error { + files, err := d.Readdir() + if err != nil { + return fmt.Errorf("%s: %s", d.Name, err) + } + for _, f := range files { + if f.IsDir() { + err = recur(f) + if err != nil { + return fmt.Errorf("%s: %s", f.Name, err) + } + } + } + return nil +} From 9ef8a67999ef672c1279b02e1a2fadfa96665653 Mon Sep 17 00:00:00 2001 From: John Starks Date: Wed, 6 Apr 2016 13:59:58 -0700 Subject: [PATCH 5/6] Add .gitignore --- .gitignore | 1 + 1 file changed, 1 insertion(+) create mode 100644 .gitignore diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..b883f1f --- /dev/null +++ b/.gitignore @@ -0,0 +1 @@ +*.exe From a7636ef88a652d6e4233bf03c63af4ad78634710 Mon Sep 17 00:00:00 2001 From: John Starks Date: Wed, 6 Apr 2016 15:16:44 -0700 Subject: [PATCH 6/6] WIM: Parse XML data This parses the XML data so that callers do not have to. --- wim/validate/validate.go | 3 + wim/wim.go | 134 +++++++++++++++++++++++++++++++++++---- 2 files changed, 123 insertions(+), 14 deletions(-) diff --git a/wim/validate/validate.go b/wim/validate/validate.go index e15ea70..ba03fc9 100644 --- a/wim/validate/validate.go +++ b/wim/validate/validate.go @@ -20,6 +20,9 @@ func main() { panic(err) } + + fmt.Printf("%#v\n%#v\n", w.Image[0], w.Image[0].Windows) + dir, err := w.Image[0].Open() if err != nil { panic(err) diff --git a/wim/wim.go b/wim/wim.go index 78828bf..1b7b566 100644 --- a/wim/wim.go +++ b/wim/wim.go @@ -9,10 +9,12 @@ import ( "bytes" "crypto/sha1" "encoding/binary" + "encoding/xml" "errors" "fmt" "io" "io/ioutil" + "strconv" "time" "unicode/utf16" ) @@ -39,6 +41,23 @@ const ( FILE_ATTRIBUTE_EA = 0x00040000 ) +// Windows processor architectures. +const ( + PROCESSOR_ARCHITECTURE_INTEL = 0 + PROCESSOR_ARCHITECTURE_MIPS = 1 + PROCESSOR_ARCHITECTURE_ALPHA = 2 + PROCESSOR_ARCHITECTURE_PPC = 3 + PROCESSOR_ARCHITECTURE_SHX = 4 + PROCESSOR_ARCHITECTURE_ARM = 5 + PROCESSOR_ARCHITECTURE_IA64 = 6 + PROCESSOR_ARCHITECTURE_ALPHA64 = 7 + PROCESSOR_ARCHITECTURE_MSIL = 8 + PROCESSOR_ARCHITECTURE_AMD64 = 9 + PROCESSOR_ARCHITECTURE_IA32_ON_WIN64 = 10 + PROCESSOR_ARCHITECTURE_NEUTRAL = 11 + PROCESSOR_ARCHITECTURE_ARM64 = 12 +) + var wimImageTag = [...]byte{'M', 'S', 'W', 'I', 'M', 0, 0, 0} type guid struct { @@ -150,9 +169,9 @@ type direntry struct { SecurityID uint32 SubdirOffset int64 Unused1, Unused2 int64 - CreationTime filetime - LastAccessTime filetime - LastWriteTime filetime + CreationTime Filetime + LastAccessTime Filetime + LastWriteTime Filetime Hash SHA1Hash Padding uint32 ReparseHardLink int64 @@ -172,12 +191,14 @@ type streamentry struct { const streamentrySize = 38 -type filetime struct { +// Filetime represents a Windows time. +type Filetime struct { LowDateTime uint32 HighDateTime uint32 } -func (ft *filetime) Time() time.Time { +// Time returns the time as time.Time. +func (ft *Filetime) Time() time.Time { // 100-nanosecond intervals since January 1, 1601 nsec := int64(ft.HighDateTime)<<32 + int64(ft.LowDateTime) // change starting time to the Epoch (00:00:00 UTC, January 1, 1970) @@ -187,6 +208,67 @@ func (ft *filetime) Time() time.Time { return time.Unix(0, nsec) } +// UnmarshalXML unmarshals the time from a WIM XML blob. +func (ft *Filetime) UnmarshalXML(d *xml.Decoder, start xml.StartElement) error { + type time struct { + Low string `xml:"LOWPART"` + High string `xml:"HIGHPART"` + } + var t time + err := d.DecodeElement(&t, &start) + if err != nil { + return err + } + + low, err := strconv.ParseUint(t.Low, 0, 32) + if err != nil { + return err + } + high, err := strconv.ParseUint(t.High, 0, 32) + if err != nil { + return err + } + + ft.LowDateTime = uint32(low) + ft.HighDateTime = uint32(high) + return nil +} + +type info struct { + Image []ImageInfo `xml:"IMAGE"` +} + +// ImageInfo contains information about the image. +type ImageInfo struct { + Name string `xml:"NAME"` + Index int `xml:"INDEX,attr"` + CreationTime Filetime `xml:"CREATIONTIME"` + ModTime Filetime `xml:"LASTMODIFICATIONTIME"` + Windows *WindowsInfo `xml:"WINDOWS"` +} + +// WindowsInfo contains information about the Windows installation in the image. +type WindowsInfo struct { + Arch byte `xml:"ARCH"` + ProductName string `xml:"PRODUCTNAME"` + EditionID string `xml:"EDITIONID"` + InstallationType string `xml:"INSTALLATIONTYPE"` + ProductType string `xml:"PRODUCTTYPE"` + Languages []string `xml:"LANGUAGES>LANGUAGE"` + DefaultLanguage string `xml:"LANGUAGES>DEFAULT"` + Version Version `xml:"VERSION"` + SystemRoot string `xml:"SYSTEMROOT"` +} + +// Version represents a Windows build version. +type Version struct { + Major int `xml:"MAJOR"` + Minor int `xml:"MINOR"` + Build int `xml:"BUILD"` + SPBuild int `xml:"SPBUILD"` + SPLevel int `xml:"SPLEVEL"` +} + // ParseError is returned when the WIM cannot be parsed. type ParseError struct { Oper string @@ -207,7 +289,8 @@ type Reader struct { r io.ReaderAt fileData map[SHA1Hash]resourceDescriptor - Image []*Image // The WIM's images. + XMLInfo string // The XML information about the WIM. + Image []*Image // The WIM's images. } // Image represents an image within a WIM file. @@ -216,6 +299,8 @@ type Image struct { offset resourceDescriptor sds [][]byte rootOffset int64 + + ImageInfo } // StreamHeader contains alternate data stream metadata. @@ -238,9 +323,9 @@ type FileHeader struct { ShortName string Attributes uint32 SecurityDescriptor []byte - CreationTime time.Time - LastAccessTime time.Time - LastWriteTime time.Time + CreationTime Filetime + LastAccessTime Filetime + LastWriteTime Filetime Hash SHA1Hash Size int64 LinkID int64 @@ -286,8 +371,30 @@ func NewReader(f io.ReaderAt) (*Reader, error) { if err != nil { return nil, err } + + xmlinfo, err := r.readXML() + if err != nil { + return nil, err + } + + var info info + err = xml.Unmarshal([]byte(xmlinfo), &info) + if err != nil { + return nil, &ParseError{Oper: "XML info", Err: err} + } + + for i, img := range images { + for _, imgInfo := range info.Image { + if imgInfo.Index == i+1 { + img.ImageInfo = imgInfo + break + } + } + } + r.fileData = fileData r.Image = images + r.XMLInfo = xmlinfo return r, nil } @@ -321,8 +428,7 @@ func (r *Reader) readResource(hdr *resourceDescriptor) ([]byte, error) { return ioutil.ReadAll(rsrc) } -// ReadXML reads the XML metadata from a WIM. -func (r *Reader) ReadXML() (string, error) { +func (r *Reader) readXML() (string, error) { if r.hdr.XMLData.CompressedSize() == 0 { return "", nil } @@ -552,9 +658,9 @@ func (img *Image) readNextEntry(r *bufio.Reader) (*File, error) { f := &File{ FileHeader: FileHeader{ Attributes: dentry.Attributes, - CreationTime: dentry.CreationTime.Time(), - LastAccessTime: dentry.LastAccessTime.Time(), - LastWriteTime: dentry.LastWriteTime.Time(), + CreationTime: dentry.CreationTime, + LastAccessTime: dentry.LastAccessTime, + LastWriteTime: dentry.LastWriteTime, Hash: dentry.Hash, Size: offset.OriginalSize, Name: name,