diff --git a/src/libraries/System.IO.Packaging/src/Resources/Strings.resx b/src/libraries/System.IO.Packaging/src/Resources/Strings.resx index 536ebbe57483ae..bec9f0ceb4215c 100644 --- a/src/libraries/System.IO.Packaging/src/Resources/Strings.resx +++ b/src/libraries/System.IO.Packaging/src/Resources/Strings.resx @@ -72,6 +72,9 @@ ContentType string cannot have leading/trailing Linear White Spaces [LWS - RFC 2616]. + + The '[Content_Types].xml' part exceeds the maximum allowed size of {0} bytes. + Unrecognized root element in Core Properties part. diff --git a/src/libraries/System.IO.Packaging/src/System/IO/Packaging/ZipPackage.cs b/src/libraries/System.IO.Packaging/src/System/IO/Packaging/ZipPackage.cs index f3e1a416e360b6..9f77bee64dd6e2 100644 --- a/src/libraries/System.IO.Packaging/src/System/IO/Packaging/ZipPackage.cs +++ b/src/libraries/System.IO.Packaging/src/System/IO/Packaging/ZipPackage.cs @@ -1141,12 +1141,34 @@ private void ParseContentTypesFile(System.Collections.ObjectModel.ReadOnlyCollec throw new FormatException(SR.BadPackageFormat); } + if (_contentTypeZipArchiveEntry.Length > MaxContentTypesXmlSize) + { + throw new FileFormatException(SR.Format(SR.ContentTypeStreamTooLarge, MaxContentTypesXmlSize)); + } + _contentTypeStreamExists = true; - return _zipStreamManager.Open(_contentTypeZipArchiveEntry, FileAccess.ReadWrite); + return _zipStreamManager.Open(_contentTypeZipArchiveEntry, FileAccess.Read); } // If the content type stream is interleaved, validate the piece numbering. else if (partPieces != null) { + // Sum the piece lengths without risking overflow: each ZipArchiveEntry.Length is a + // non-negative long that (with Zip64) can be as large as long.MaxValue, so a malicious + // archive could otherwise make an unchecked running total wrap around and defeat the + // size check below. Comparing against the remaining budget instead of adding first keeps + // totalLength bounded by MaxContentTypesXmlSize at all times. + long totalLength = 0; + foreach (ZipPackagePartPiece piece in partPieces) + { + long pieceLength = piece.ZipArchiveEntry.Length; + if (pieceLength > MaxContentTypesXmlSize - totalLength) + { + throw new FileFormatException(SR.Format(SR.ContentTypeStreamTooLarge, MaxContentTypesXmlSize)); + } + + totalLength += pieceLength; + } + _contentTypeStreamExists = true; _contentTypeStreamPieces = partPieces; @@ -1304,6 +1326,17 @@ private static void ThrowIfXmlAttributeMissing(string attributeName, string? att private CompressionLevel _cachedCompressionLevel; private const string ContentTypesFile = "[Content_Types].xml"; private const string ContentTypesFileUpperInvariant = "[CONTENT_TYPES].XML"; + + // Maximum allowed (uncompressed) size for the "[Content_Types].xml" part. This part is package + // metadata rather than user content, so - similarly to the metadata block size limit applied to + // TAR archives (see https://github.com/dotnet/runtime/pull/127602) - its size can be bounded to a + // value that comfortably covers legitimate packages while preventing a small, malformed, or + // malicious archive from declaring an implausibly large entry that gets eagerly buffered in memory + // when the package is opened for ReadWrite access. 4 MB covers packages with 100,000+ parts even in + // the worst case where every part has a distinct content type (forcing an element each), + // which is far beyond what real-world OPC packages (e.g. Office documents) contain. + private const long MaxContentTypesXmlSize = 4 * 1024 * 1024; + private const int DefaultDictionaryInitialSize = 16; private const int OverrideDictionaryInitialSize = 8; diff --git a/src/libraries/System.IO.Packaging/tests/PartPieceTests.cs b/src/libraries/System.IO.Packaging/tests/PartPieceTests.cs index 37736f4d6d7f6f..e3cd9725f1ae00 100644 --- a/src/libraries/System.IO.Packaging/tests/PartPieceTests.cs +++ b/src/libraries/System.IO.Packaging/tests/PartPieceTests.cs @@ -271,6 +271,23 @@ public void CanParseInterleavedContentTypesFile() Assert.NotEmpty(zipPackage.GetParts()); } + // Regression test: an interleaved "[Content_Types].xml" (i.e. one split into pieces) must be + // bounded by the same maximum size as an atomic "[Content_Types].xml", since its pieces are + // recombined and parsed by the same XmlReader. Otherwise a malicious package could bypass the + // atomic-entry size guard simply by splitting the content types part into pieces. + [Fact] + public void InterleavedContentTypesExceedingMaxSizeThrows() + { + // Two highly-compressible (all-zero) pieces whose combined declared uncompressed size + // exceeds the 4 MB cap, even though neither piece alone does. + byte[] package = CreatePackage( + new PartConstructionParameters("AtomicPartEntry.bin", true, false, false, false, [200], GenerateRandomBytes), + new PartConstructionParameters("[Content_Types].xml", false, true, false, false, [2_500_000, 2_500_001], (_, totalLength) => new byte[totalLength])); + + using var ms = new MemoryStream(package); + Assert.Throws(() => Package.Open(ms)); + } + // Verify that the IComparable implementation on ZipPackagePartPiece works properly. // If it is, we should see the list reordered by piece number [Theory] diff --git a/src/libraries/System.IO.Packaging/tests/Tests.cs b/src/libraries/System.IO.Packaging/tests/Tests.cs index 7988907382fbdd..d7be1f72be2e1a 100644 --- a/src/libraries/System.IO.Packaging/tests/Tests.cs +++ b/src/libraries/System.IO.Packaging/tests/Tests.cs @@ -1,6 +1,7 @@ // Licensed to the .NET Foundation under one or more agreements. // The .NET Foundation licenses this file to you under the MIT license. +using System.Buffers.Binary; using System.Linq; using System.Runtime.CompilerServices; using System.Text; @@ -101,6 +102,79 @@ public void GetStreamCreate_OverwritesExistingPartContentWithoutLeftoverBytes(Fi } } + [Fact] + public void Open_ContentTypesEntryDeclaredSizeExceedsMaximum_ThrowsFileFormatException() + { + // Regression test: Package.Open's default ReadWrite access automatically parses the mandatory + // [Content_Types].xml part during Open(). This part is package metadata, not user content, so + // its declared (but untrusted) uncompressed size must be bounded to a sane maximum. Otherwise a + // small, corrupt, or maliciously crafted archive could declare an implausibly large entry that + // gets eagerly buffered in memory (as a MemoryStream sized to the declared value) when the + // package is opened for ReadWrite access. + FileInfo file = GetTempFileInfoWithExtension(".zip"); + + using (Package package = Package.Open(file.FullName, FileMode.Create, FileAccess.ReadWrite)) + { + PackagePart part = package.CreatePart( + PackUriHelper.CreatePartUri(new Uri("MyFile.xml", UriKind.Relative)), + Mime_MediaTypeNames_Text_Xml, + CompressionOption.Normal); + using Stream s = part.GetStream(FileMode.Create, FileAccess.Write); + byte[] content = Encoding.UTF8.GetBytes(s_DocumentXml); + s.Write(content, 0, content.Length); + } + + byte[] archiveBytes = File.ReadAllBytes(file.FullName); + // Comfortably above the 4 MB cap, but small enough that a regression in the guard would not + // risk a large allocation while running this test. + PatchContentTypesUncompressedSize(archiveBytes, oversizedUncompressedSize: 5_000_000); + File.WriteAllBytes(file.FullName, archiveBytes); + + Assert.Throws(() => Package.Open(file.FullName, FileMode.Open, FileAccess.ReadWrite)); + } + + // Patches the declared uncompressed size field (in both the local file header and the central + // directory record) for the "[Content_Types].xml" entry within a raw, in-memory zip byte array. + private static void PatchContentTypesUncompressedSize(byte[] archiveBytes, uint oversizedUncompressedSize) + { + const string EntryName = "[Content_Types].xml"; + byte[] nameBytes = Encoding.ASCII.GetBytes(EntryName); + ReadOnlySpan localHeaderSignature = [0x50, 0x4B, 0x03, 0x04]; + ReadOnlySpan centralDirectorySignature = [0x50, 0x4B, 0x01, 0x02]; + + int patchedCount = 0; + int searchStart = 0; + Span archiveSpan = archiveBytes; + while (true) + { + int nameIndex = archiveSpan.Slice(searchStart).IndexOf((ReadOnlySpan)nameBytes); + if (nameIndex < 0) + { + break; + } + nameIndex += searchStart; + searchStart = nameIndex + 1; + + // Local file header: fixed 30-byte header immediately precedes the file name; the + // uncompressed size field is the 4 bytes located 8 bytes before the file name starts. + if (nameIndex >= 30 && archiveSpan.Slice(nameIndex - 30, 4).SequenceEqual(localHeaderSignature)) + { + BinaryPrimitives.WriteUInt32LittleEndian(archiveSpan.Slice(nameIndex - 8, 4), oversizedUncompressedSize); + patchedCount++; + } + // Central directory file header: fixed 46-byte header immediately precedes the file name; + // the uncompressed size field is the 4 bytes located 22 bytes before the file name starts. + else if (nameIndex >= 46 && archiveSpan.Slice(nameIndex - 46, 4).SequenceEqual(centralDirectorySignature)) + { + BinaryPrimitives.WriteUInt32LittleEndian(archiveSpan.Slice(nameIndex - 22, 4), oversizedUncompressedSize); + patchedCount++; + } + } + + // Sanity check: both the local header and central directory copies must have been found and patched. + Assert.Equal(2, patchedCount); + } + [Fact] public void T201_FileFormatException() {