diff --git a/src/SharpCompress/Common/CompressionType.cs b/src/SharpCompress/Common/CompressionType.cs
index b26e5a95..20f166c4 100644
--- a/src/SharpCompress/Common/CompressionType.cs
+++ b/src/SharpCompress/Common/CompressionType.cs
@@ -15,5 +15,6 @@ public enum CompressionType
Xz,
Unknown,
Deflate64,
- Shrink
+ Shrink,
+ Lzw
}
diff --git a/src/SharpCompress/Compressors/Lzw/LzwConstants.cs b/src/SharpCompress/Compressors/Lzw/LzwConstants.cs
new file mode 100644
index 00000000..0325adbb
--- /dev/null
+++ b/src/SharpCompress/Compressors/Lzw/LzwConstants.cs
@@ -0,0 +1,65 @@
+namespace SharpCompress.Compressors.Lzw
+{
+ ///
+ /// This class contains constants used for LZW
+ ///
+ [System.Diagnostics.CodeAnalysis.SuppressMessage(
+ "Naming",
+ "CA1707:Identifiers should not contain underscores",
+ Justification = "kept for backwards compatibility"
+ )]
+ public sealed class LzwConstants
+ {
+ ///
+ /// Magic number found at start of LZW header: 0x1f 0x9d
+ ///
+ public const int MAGIC = 0x1f9d;
+
+ ///
+ /// Maximum number of bits per code
+ ///
+ public const int MAX_BITS = 16;
+
+ /* 3rd header byte:
+ * bit 0..4 Number of compression bits
+ * bit 5 Extended header
+ * bit 6 Free
+ * bit 7 Block mode
+ */
+
+ ///
+ /// Mask for 'number of compression bits'
+ ///
+ public const int BIT_MASK = 0x1f;
+
+ ///
+ /// Indicates the presence of a fourth header byte
+ ///
+ public const int EXTENDED_MASK = 0x20;
+
+ //public const int FREE_MASK = 0x40;
+
+ ///
+ /// Reserved bits
+ ///
+ public const int RESERVED_MASK = 0x60;
+
+ ///
+ /// Block compression: if table is full and compression rate is dropping,
+ /// clear the dictionary.
+ ///
+ public const int BLOCK_MODE_MASK = 0x80;
+
+ ///
+ /// LZW file header size (in bytes)
+ ///
+ public const int HDR_SIZE = 3;
+
+ ///
+ /// Initial number of bits per code
+ ///
+ public const int INIT_BITS = 9;
+
+ private LzwConstants() { }
+ }
+}
diff --git a/src/SharpCompress/Compressors/Lzw/LzwStream.cs b/src/SharpCompress/Compressors/Lzw/LzwStream.cs
new file mode 100644
index 00000000..488ba1aa
--- /dev/null
+++ b/src/SharpCompress/Compressors/Lzw/LzwStream.cs
@@ -0,0 +1,597 @@
+using System;
+using System.IO;
+using SharpCompress.Common;
+
+namespace SharpCompress.Compressors.Lzw
+{
+ ///
+ /// This filter stream is used to decompress a LZW format stream.
+ /// Specifically, a stream that uses the LZC compression method.
+ /// This file format is usually associated with the .Z file extension.
+ ///
+ /// See http://en.wikipedia.org/wiki/Compress
+ /// See http://wiki.wxwidgets.org/Development:_Z_File_Format
+ ///
+ /// The file header consists of 3 (or optionally 4) bytes. The first two bytes
+ /// contain the magic marker "0x1f 0x9d", followed by a byte of flags.
+ ///
+ /// Based on Java code by Ronald Tschalar, which in turn was based on the unlzw.c
+ /// code in the gzip package.
+ ///
+ /// This sample shows how to unzip a compressed file
+ ///
+ /// using System;
+ /// using System.IO;
+ ///
+ /// using ICSharpCode.SharpZipLib.Core;
+ /// using ICSharpCode.SharpZipLib.LZW;
+ ///
+ /// class MainClass
+ /// {
+ /// public static void Main(string[] args)
+ /// {
+ /// using (Stream inStream = new LzwInputStream(File.OpenRead(args[0])))
+ /// using (FileStream outStream = File.Create(Path.GetFileNameWithoutExtension(args[0]))) {
+ /// byte[] buffer = new byte[4096];
+ /// StreamUtils.Copy(inStream, outStream, buffer);
+ /// // OR
+ /// inStream.Read(buffer, 0, buffer.Length);
+ /// // now do something with the buffer
+ /// }
+ /// }
+ /// }
+ ///
+ ///
+ public class LzwStream : Stream
+ {
+ public static bool IsLzwStream(Stream stream)
+ {
+ try
+ {
+ byte[] hdr = new byte[LzwConstants.HDR_SIZE];
+
+ int result = stream.Read(hdr, 0, hdr.Length);
+
+ // Check the magic marker
+ if (result < 0)
+ throw new IncompleteArchiveException("Failed to read LZW header");
+
+ if (hdr[0] != (LzwConstants.MAGIC >> 8) || hdr[1] != (LzwConstants.MAGIC & 0xff))
+ {
+ throw new IncompleteArchiveException(
+ String.Format(
+ "Wrong LZW header. Magic bytes don't match. 0x{0:x2} 0x{1:x2}",
+ hdr[0],
+ hdr[1]
+ )
+ );
+ }
+ }
+ catch (Exception)
+ {
+ return false;
+ }
+ return true;
+ }
+
+ ///
+ /// Gets or sets a flag indicating ownership of underlying stream.
+ /// When the flag is true will close the underlying stream also.
+ ///
+ /// The default value is true.
+ public bool IsStreamOwner { get; set; } = false;
+
+ ///
+ /// Creates a LzwInputStream
+ ///
+ ///
+ /// The stream to read compressed data from (baseInputStream LZW format)
+ ///
+ public LzwStream(Stream baseInputStream)
+ {
+ this.baseInputStream = baseInputStream;
+ }
+
+ ///
+ /// See
+ ///
+ ///
+ public override int ReadByte()
+ {
+ int b = Read(one, 0, 1);
+ if (b == 1)
+ return (one[0] & 0xff);
+ return -1;
+ }
+
+ ///
+ /// Reads decompressed data into the provided buffer byte array
+ ///
+ ///
+ /// The array to read and decompress data into
+ ///
+ ///
+ /// The offset indicating where the data should be placed
+ ///
+ ///
+ /// The number of bytes to decompress
+ ///
+ /// The number of bytes read. Zero signals the end of stream
+ public override int Read(byte[] buffer, int offset, int count)
+ {
+ if (!headerParsed)
+ ParseHeader();
+
+ if (eof)
+ return 0;
+
+ int start = offset;
+
+ /* Using local copies of various variables speeds things up by as
+ * much as 30% in Java! Performance not tested in C#.
+ */
+ int[] lTabPrefix = tabPrefix;
+ byte[] lTabSuffix = tabSuffix;
+ byte[] lStack = stack;
+ int lNBits = nBits;
+ int lMaxCode = maxCode;
+ int lMaxMaxCode = maxMaxCode;
+ int lBitMask = bitMask;
+ int lOldCode = oldCode;
+ byte lFinChar = finChar;
+ int lStackP = stackP;
+ int lFreeEnt = freeEnt;
+ byte[] lData = data;
+ int lBitPos = bitPos;
+
+ // empty stack if stuff still left
+ int sSize = lStack.Length - lStackP;
+ if (sSize > 0)
+ {
+ int num = (sSize >= count) ? count : sSize;
+ Array.Copy(lStack, lStackP, buffer, offset, num);
+ offset += num;
+ count -= num;
+ lStackP += num;
+ }
+
+ if (count == 0)
+ {
+ stackP = lStackP;
+ return offset - start;
+ }
+
+ // loop, filling local buffer until enough data has been decompressed
+ MainLoop:
+ do
+ {
+ if (end < EXTRA)
+ {
+ Fill();
+ }
+
+ int bitIn = (got > 0) ? (end - end % lNBits) << 3 : (end << 3) - (lNBits - 1);
+
+ while (lBitPos < bitIn)
+ {
+ #region A
+
+ // handle 1-byte reads correctly
+ if (count == 0)
+ {
+ nBits = lNBits;
+ maxCode = lMaxCode;
+ maxMaxCode = lMaxMaxCode;
+ bitMask = lBitMask;
+ oldCode = lOldCode;
+ finChar = lFinChar;
+ stackP = lStackP;
+ freeEnt = lFreeEnt;
+ bitPos = lBitPos;
+
+ return offset - start;
+ }
+
+ // check for code-width expansion
+ if (lFreeEnt > lMaxCode)
+ {
+ int nBytes = lNBits << 3;
+ lBitPos = (lBitPos - 1) + nBytes - (lBitPos - 1 + nBytes) % nBytes;
+
+ lNBits++;
+ lMaxCode = (lNBits == maxBits) ? lMaxMaxCode : (1 << lNBits) - 1;
+
+ lBitMask = (1 << lNBits) - 1;
+ lBitPos = ResetBuf(lBitPos);
+ goto MainLoop;
+ }
+
+ #endregion A
+
+ #region B
+
+ // read next code
+ int pos = lBitPos >> 3;
+ int code =
+ (
+ (
+ (lData[pos] & 0xFF)
+ | ((lData[pos + 1] & 0xFF) << 8)
+ | ((lData[pos + 2] & 0xFF) << 16)
+ ) >> (lBitPos & 0x7)
+ ) & lBitMask;
+
+ lBitPos += lNBits;
+
+ // handle first iteration
+ if (lOldCode == -1)
+ {
+ if (code >= 256)
+ throw new IncompleteArchiveException(
+ "corrupt input: " + code + " > 255"
+ );
+
+ lFinChar = (byte)(lOldCode = code);
+ buffer[offset++] = lFinChar;
+ count--;
+ continue;
+ }
+
+ // handle CLEAR code
+ if (code == TBL_CLEAR && blockMode)
+ {
+ Array.Copy(zeros, 0, lTabPrefix, 0, zeros.Length);
+ lFreeEnt = TBL_FIRST - 1;
+
+ int nBytes = lNBits << 3;
+ lBitPos = (lBitPos - 1) + nBytes - (lBitPos - 1 + nBytes) % nBytes;
+ lNBits = LzwConstants.INIT_BITS;
+ lMaxCode = (1 << lNBits) - 1;
+ lBitMask = lMaxCode;
+
+ // Code tables reset
+
+ lBitPos = ResetBuf(lBitPos);
+ goto MainLoop;
+ }
+
+ #endregion B
+
+ #region C
+
+ // setup
+ int inCode = code;
+ lStackP = lStack.Length;
+
+ // Handle KwK case
+ if (code >= lFreeEnt)
+ {
+ if (code > lFreeEnt)
+ {
+ throw new IncompleteArchiveException(
+ "corrupt input: code=" + code + ", freeEnt=" + lFreeEnt
+ );
+ }
+
+ lStack[--lStackP] = lFinChar;
+ code = lOldCode;
+ }
+
+ // Generate output characters in reverse order
+ while (code >= 256)
+ {
+ lStack[--lStackP] = lTabSuffix[code];
+ code = lTabPrefix[code];
+ }
+
+ lFinChar = lTabSuffix[code];
+ buffer[offset++] = lFinChar;
+ count--;
+
+ // And put them out in forward order
+ sSize = lStack.Length - lStackP;
+ int num = (sSize >= count) ? count : sSize;
+ Array.Copy(lStack, lStackP, buffer, offset, num);
+ offset += num;
+ count -= num;
+ lStackP += num;
+
+ #endregion C
+
+ #region D
+
+ // generate new entry in table
+ if (lFreeEnt < lMaxMaxCode)
+ {
+ lTabPrefix[lFreeEnt] = lOldCode;
+ lTabSuffix[lFreeEnt] = lFinChar;
+ lFreeEnt++;
+ }
+
+ // Remember previous code
+ lOldCode = inCode;
+
+ // if output buffer full, then return
+ if (count == 0)
+ {
+ nBits = lNBits;
+ maxCode = lMaxCode;
+ bitMask = lBitMask;
+ oldCode = lOldCode;
+ finChar = lFinChar;
+ stackP = lStackP;
+ freeEnt = lFreeEnt;
+ bitPos = lBitPos;
+
+ return offset - start;
+ }
+
+ #endregion D
+ } // while
+
+ lBitPos = ResetBuf(lBitPos);
+ } while (got > 0); // do..while
+
+ nBits = lNBits;
+ maxCode = lMaxCode;
+ bitMask = lBitMask;
+ oldCode = lOldCode;
+ finChar = lFinChar;
+ stackP = lStackP;
+ freeEnt = lFreeEnt;
+ bitPos = lBitPos;
+
+ eof = true;
+ return offset - start;
+ }
+
+ ///
+ /// Moves the unread data in the buffer to the beginning and resets
+ /// the pointers.
+ ///
+ ///
+ ///
+ private int ResetBuf(int bitPosition)
+ {
+ int pos = bitPosition >> 3;
+ Array.Copy(data, pos, data, 0, end - pos);
+ end -= pos;
+ return 0;
+ }
+
+ private void Fill()
+ {
+ got = baseInputStream.Read(data, end, data.Length - 1 - end);
+ if (got > 0)
+ {
+ end += got;
+ }
+ }
+
+ private void ParseHeader()
+ {
+ headerParsed = true;
+
+ byte[] hdr = new byte[LzwConstants.HDR_SIZE];
+
+ int result = baseInputStream.Read(hdr, 0, hdr.Length);
+
+ // Check the magic marker
+ if (result < 0)
+ throw new IncompleteArchiveException("Failed to read LZW header");
+
+ if (hdr[0] != (LzwConstants.MAGIC >> 8) || hdr[1] != (LzwConstants.MAGIC & 0xff))
+ {
+ throw new IncompleteArchiveException(
+ String.Format(
+ "Wrong LZW header. Magic bytes don't match. 0x{0:x2} 0x{1:x2}",
+ hdr[0],
+ hdr[1]
+ )
+ );
+ }
+
+ // Check the 3rd header byte
+ blockMode = (hdr[2] & LzwConstants.BLOCK_MODE_MASK) > 0;
+ maxBits = hdr[2] & LzwConstants.BIT_MASK;
+
+ if (maxBits > LzwConstants.MAX_BITS)
+ {
+ throw new ArchiveException(
+ "Stream compressed with "
+ + maxBits
+ + " bits, but decompression can only handle "
+ + LzwConstants.MAX_BITS
+ + " bits."
+ );
+ }
+
+ if ((hdr[2] & LzwConstants.RESERVED_MASK) > 0)
+ {
+ throw new ArchiveException("Unsupported bits set in the header.");
+ }
+
+ // Initialize variables
+ maxMaxCode = 1 << maxBits;
+ nBits = LzwConstants.INIT_BITS;
+ maxCode = (1 << nBits) - 1;
+ bitMask = maxCode;
+ oldCode = -1;
+ finChar = 0;
+ freeEnt = blockMode ? TBL_FIRST : 256;
+
+ tabPrefix = new int[1 << maxBits];
+ tabSuffix = new byte[1 << maxBits];
+ stack = new byte[1 << maxBits];
+ stackP = stack.Length;
+
+ for (int idx = 255; idx >= 0; idx--)
+ tabSuffix[idx] = (byte)idx;
+ }
+
+ #region Stream Overrides
+
+ ///
+ /// Gets a value indicating whether the current stream supports reading
+ ///
+ public override bool CanRead
+ {
+ get { return baseInputStream.CanRead; }
+ }
+
+ ///
+ /// Gets a value of false indicating seeking is not supported for this stream.
+ ///
+ public override bool CanSeek
+ {
+ get { return false; }
+ }
+
+ ///
+ /// Gets a value of false indicating that this stream is not writeable.
+ ///
+ public override bool CanWrite
+ {
+ get { return false; }
+ }
+
+ ///
+ /// A value representing the length of the stream in bytes.
+ ///
+ public override long Length
+ {
+ get { return got; }
+ }
+
+ ///
+ /// The current position within the stream.
+ /// Throws a NotSupportedException when attempting to set the position
+ ///
+ /// Attempting to set the position
+ public override long Position
+ {
+ get { return baseInputStream.Position; }
+ set { throw new NotSupportedException("InflaterInputStream Position not supported"); }
+ }
+
+ ///
+ /// Flushes the baseInputStream
+ ///
+ public override void Flush()
+ {
+ baseInputStream.Flush();
+ }
+
+ ///
+ /// Sets the position within the current stream
+ /// Always throws a NotSupportedException
+ ///
+ /// The relative offset to seek to.
+ /// The defining where to seek from.
+ /// The new position in the stream.
+ /// Any access
+ public override long Seek(long offset, SeekOrigin origin)
+ {
+ throw new NotSupportedException("Seek not supported");
+ }
+
+ ///
+ /// Set the length of the current stream
+ /// Always throws a NotSupportedException
+ ///
+ /// The new length value for the stream.
+ /// Any access
+ public override void SetLength(long value)
+ {
+ throw new NotSupportedException("InflaterInputStream SetLength not supported");
+ }
+
+ ///
+ /// Writes a sequence of bytes to stream and advances the current position
+ /// This method always throws a NotSupportedException
+ ///
+ /// The buffer containing data to write.
+ /// The offset of the first byte to write.
+ /// The number of bytes to write.
+ /// Any access
+ public override void Write(byte[] buffer, int offset, int count)
+ {
+ throw new NotSupportedException("InflaterInputStream Write not supported");
+ }
+
+ ///
+ /// Writes one byte to the current stream and advances the current position
+ /// Always throws a NotSupportedException
+ ///
+ /// The byte to write.
+ /// Any access
+ public override void WriteByte(byte value)
+ {
+ throw new NotSupportedException("InflaterInputStream WriteByte not supported");
+ }
+
+ ///
+ /// Closes the input stream. When
+ /// is true the underlying stream is also closed.
+ ///
+ protected override void Dispose(bool disposing)
+ {
+ if (!isClosed)
+ {
+ isClosed = true;
+ if (IsStreamOwner)
+ {
+ baseInputStream.Dispose();
+ }
+ }
+ }
+
+ #endregion Stream Overrides
+
+ #region Instance Fields
+
+ private Stream baseInputStream;
+
+ ///
+ /// Flag indicating wether this instance has been closed or not.
+ ///
+ private bool isClosed;
+
+ private readonly byte[] one = new byte[1];
+ private bool headerParsed;
+
+ // string table stuff
+ private const int TBL_CLEAR = 0x100;
+
+ private const int TBL_FIRST = TBL_CLEAR + 1;
+
+ private int[] tabPrefix = new int[0]; //
+ private byte[] tabSuffix = new byte[0]; //
+ private readonly int[] zeros = new int[256];
+ private byte[] stack = new byte[0]; //
+
+ // various state
+ private bool blockMode;
+
+ private int nBits;
+ private int maxBits;
+ private int maxMaxCode;
+ private int maxCode;
+ private int bitMask;
+ private int oldCode;
+ private byte finChar;
+ private int stackP;
+ private int freeEnt;
+
+ // input buffer
+ private readonly byte[] data = new byte[1024 * 8];
+
+ private int bitPos;
+ private int end;
+ private int got;
+ private bool eof;
+ private const int EXTRA = 64;
+
+ #endregion Instance Fields
+ }
+}
diff --git a/src/SharpCompress/Factories/TarFactory.cs b/src/SharpCompress/Factories/TarFactory.cs
index 54c24c02..715a05d2 100644
--- a/src/SharpCompress/Factories/TarFactory.cs
+++ b/src/SharpCompress/Factories/TarFactory.cs
@@ -6,6 +6,7 @@ using SharpCompress.Common;
using SharpCompress.Compressors;
using SharpCompress.Compressors.BZip2;
using SharpCompress.Compressors.LZMA;
+using SharpCompress.Compressors.Lzw;
using SharpCompress.Compressors.Xz;
using SharpCompress.IO;
using SharpCompress.Readers;
@@ -160,6 +161,19 @@ public class TarFactory
}
}
+ rewindableStream.Rewind(false);
+ if (LzwStream.IsLzwStream(rewindableStream))
+ {
+ rewindableStream.Rewind(false);
+ var testStream = new LzwStream(rewindableStream);
+ if (TarArchive.IsTarFile(testStream))
+ {
+ rewindableStream.Rewind(true);
+ reader = new TarReader(rewindableStream, options, CompressionType.Lzw);
+ return true;
+ }
+ }
+
return false;
}
diff --git a/src/SharpCompress/Readers/Tar/TarReader.cs b/src/SharpCompress/Readers/Tar/TarReader.cs
index 55ef132f..528e9cf4 100644
--- a/src/SharpCompress/Readers/Tar/TarReader.cs
+++ b/src/SharpCompress/Readers/Tar/TarReader.cs
@@ -1,4 +1,4 @@
-using System;
+using System;
using System.Collections.Generic;
using System.IO;
using SharpCompress.Archives.GZip;
@@ -10,6 +10,7 @@ using SharpCompress.Compressors.BZip2;
using SharpCompress.Compressors.Deflate;
using SharpCompress.Compressors.LZMA;
using SharpCompress.Compressors.Xz;
+using SharpCompress.Compressors.Lzw;
using SharpCompress.IO;
namespace SharpCompress.Readers.Tar;
@@ -36,6 +37,7 @@ public class TarReader : AbstractReader
CompressionType.GZip => new GZipStream(stream, CompressionMode.Decompress),
CompressionType.LZip => new LZipStream(stream, CompressionMode.Decompress),
CompressionType.Xz => new XZStream(stream),
+ CompressionType.Lzw => new LzwStream(stream),
CompressionType.None => stream,
_ => throw new NotSupportedException("Invalid compression type: " + compressionType)
};
diff --git a/src/SharpCompress/SharpCompress.csproj b/src/SharpCompress/SharpCompress.csproj
index 5a9fa3e5..dec8a386 100644
--- a/src/SharpCompress/SharpCompress.csproj
+++ b/src/SharpCompress/SharpCompress.csproj
@@ -26,6 +26,9 @@
true
README.md
+
+
+
@@ -43,6 +46,6 @@
-
+
diff --git a/tests/SharpCompress.Test/Tar/TarReaderTests.cs b/tests/SharpCompress.Test/Tar/TarReaderTests.cs
index 56ea72ed..1ee4ae8d 100644
--- a/tests/SharpCompress.Test/Tar/TarReaderTests.cs
+++ b/tests/SharpCompress.Test/Tar/TarReaderTests.cs
@@ -40,6 +40,9 @@ public class TarReaderTests : ReaderTests
}
}
+ [Fact]
+ public void Tar_Z_Reader() => Read("Tar.tar.Z", CompressionType.Lzw);
+
[Fact]
public void Tar_BZip2_Reader() => Read("Tar.tar.bz2", CompressionType.BZip2);
diff --git a/tests/TestArchives/Archives/Tar.tar.Z b/tests/TestArchives/Archives/Tar.tar.Z
new file mode 100644
index 00000000..0eb90fb1
Binary files /dev/null and b/tests/TestArchives/Archives/Tar.tar.Z differ