HTTPRepository: improving handling of archives
Avoid hard-coding the archive extension, and ensure the extracted archive directory is not orphaned on update. Finally, use the literal filename in the .dirindex when computing the hash, rather than adding a .tgz extension.
This commit is contained in:
committed by
Automatic Release Builder
parent
a0d7f0e172
commit
6f9f694eff
@@ -286,6 +286,7 @@ else()
|
|||||||
find_package(ZLIB 1.2.4 REQUIRED)
|
find_package(ZLIB 1.2.4 REQUIRED)
|
||||||
endif()
|
endif()
|
||||||
|
|
||||||
|
find_package(LibLZMA REQUIRED)
|
||||||
find_package(CURL REQUIRED)
|
find_package(CURL REQUIRED)
|
||||||
|
|
||||||
if (SYSTEM_EXPAT)
|
if (SYSTEM_EXPAT)
|
||||||
|
|||||||
@@ -0,0 +1,124 @@
|
|||||||
|
# Distributed under the OSI-approved BSD 3-Clause License. See accompanying
|
||||||
|
# file Copyright.txt or https://cmake.org/licensing for details.
|
||||||
|
|
||||||
|
#[=======================================================================[.rst:
|
||||||
|
FindLibLZMA
|
||||||
|
-----------
|
||||||
|
|
||||||
|
Find LZMA compression algorithm headers and library.
|
||||||
|
|
||||||
|
|
||||||
|
Imported Targets
|
||||||
|
^^^^^^^^^^^^^^^^
|
||||||
|
|
||||||
|
This module defines :prop_tgt:`IMPORTED` target ``LibLZMA::LibLZMA``, if
|
||||||
|
liblzma has been found.
|
||||||
|
|
||||||
|
Result variables
|
||||||
|
^^^^^^^^^^^^^^^^
|
||||||
|
|
||||||
|
This module will set the following variables in your project:
|
||||||
|
|
||||||
|
``LIBLZMA_FOUND``
|
||||||
|
True if liblzma headers and library were found.
|
||||||
|
``LIBLZMA_INCLUDE_DIRS``
|
||||||
|
Directory where liblzma headers are located.
|
||||||
|
``LIBLZMA_LIBRARIES``
|
||||||
|
Lzma libraries to link against.
|
||||||
|
``LIBLZMA_HAS_AUTO_DECODER``
|
||||||
|
True if lzma_auto_decoder() is found (required).
|
||||||
|
``LIBLZMA_HAS_EASY_ENCODER``
|
||||||
|
True if lzma_easy_encoder() is found (required).
|
||||||
|
``LIBLZMA_HAS_LZMA_PRESET``
|
||||||
|
True if lzma_lzma_preset() is found (required).
|
||||||
|
``LIBLZMA_VERSION_MAJOR``
|
||||||
|
The major version of lzma
|
||||||
|
``LIBLZMA_VERSION_MINOR``
|
||||||
|
The minor version of lzma
|
||||||
|
``LIBLZMA_VERSION_PATCH``
|
||||||
|
The patch version of lzma
|
||||||
|
``LIBLZMA_VERSION_STRING``
|
||||||
|
version number as a string (ex: "5.0.3")
|
||||||
|
#]=======================================================================]
|
||||||
|
|
||||||
|
find_path(LIBLZMA_INCLUDE_DIR lzma.h )
|
||||||
|
if(NOT LIBLZMA_LIBRARY)
|
||||||
|
find_library(LIBLZMA_LIBRARY_RELEASE NAMES lzma liblzma NAMES_PER_DIR PATH_SUFFIXES lib)
|
||||||
|
find_library(LIBLZMA_LIBRARY_DEBUG NAMES lzmad liblzmad NAMES_PER_DIR PATH_SUFFIXES lib)
|
||||||
|
include(SelectLibraryConfigurations)
|
||||||
|
select_library_configurations(LIBLZMA)
|
||||||
|
else()
|
||||||
|
file(TO_CMAKE_PATH "${LIBLZMA_LIBRARY}" LIBLZMA_LIBRARY)
|
||||||
|
endif()
|
||||||
|
|
||||||
|
if(LIBLZMA_INCLUDE_DIR AND EXISTS "${LIBLZMA_INCLUDE_DIR}/lzma/version.h")
|
||||||
|
file(STRINGS "${LIBLZMA_INCLUDE_DIR}/lzma/version.h" LIBLZMA_HEADER_CONTENTS REGEX "#define LZMA_VERSION_[A-Z]+ [0-9]+")
|
||||||
|
|
||||||
|
string(REGEX REPLACE ".*#define LZMA_VERSION_MAJOR ([0-9]+).*" "\\1" LIBLZMA_VERSION_MAJOR "${LIBLZMA_HEADER_CONTENTS}")
|
||||||
|
string(REGEX REPLACE ".*#define LZMA_VERSION_MINOR ([0-9]+).*" "\\1" LIBLZMA_VERSION_MINOR "${LIBLZMA_HEADER_CONTENTS}")
|
||||||
|
string(REGEX REPLACE ".*#define LZMA_VERSION_PATCH ([0-9]+).*" "\\1" LIBLZMA_VERSION_PATCH "${LIBLZMA_HEADER_CONTENTS}")
|
||||||
|
|
||||||
|
set(LIBLZMA_VERSION_STRING "${LIBLZMA_VERSION_MAJOR}.${LIBLZMA_VERSION_MINOR}.${LIBLZMA_VERSION_PATCH}")
|
||||||
|
unset(LIBLZMA_HEADER_CONTENTS)
|
||||||
|
endif()
|
||||||
|
|
||||||
|
# We're using new code known now as XZ, even library still been called LZMA
|
||||||
|
# it can be found in http://tukaani.org/xz/
|
||||||
|
# Avoid using old codebase
|
||||||
|
if (LIBLZMA_LIBRARY)
|
||||||
|
include(CheckLibraryExists)
|
||||||
|
set(CMAKE_REQUIRED_QUIET_SAVE ${CMAKE_REQUIRED_QUIET})
|
||||||
|
set(CMAKE_REQUIRED_QUIET ${LibLZMA_FIND_QUIETLY})
|
||||||
|
if(NOT LIBLZMA_LIBRARY_RELEASE AND NOT LIBLZMA_LIBRARY_DEBUG)
|
||||||
|
set(LIBLZMA_LIBRARY_check ${LIBLZMA_LIBRARY})
|
||||||
|
elseif(LIBLZMA_LIBRARY_RELEASE)
|
||||||
|
set(LIBLZMA_LIBRARY_check ${LIBLZMA_LIBRARY_RELEASE})
|
||||||
|
elseif(LIBLZMA_LIBRARY_DEBUG)
|
||||||
|
set(LIBLZMA_LIBRARY_check ${LIBLZMA_LIBRARY_DEBUG})
|
||||||
|
endif()
|
||||||
|
CHECK_LIBRARY_EXISTS(${LIBLZMA_LIBRARY_check} lzma_auto_decoder "" LIBLZMA_HAS_AUTO_DECODER)
|
||||||
|
CHECK_LIBRARY_EXISTS(${LIBLZMA_LIBRARY_check} lzma_easy_encoder "" LIBLZMA_HAS_EASY_ENCODER)
|
||||||
|
CHECK_LIBRARY_EXISTS(${LIBLZMA_LIBRARY_check} lzma_lzma_preset "" LIBLZMA_HAS_LZMA_PRESET)
|
||||||
|
unset(LIBLZMA_LIBRARY_check)
|
||||||
|
set(CMAKE_REQUIRED_QUIET ${CMAKE_REQUIRED_QUIET_SAVE})
|
||||||
|
endif ()
|
||||||
|
|
||||||
|
include(FindPackageHandleStandardArgs)
|
||||||
|
find_package_handle_standard_args(LibLZMA REQUIRED_VARS LIBLZMA_LIBRARY
|
||||||
|
LIBLZMA_INCLUDE_DIR
|
||||||
|
LIBLZMA_HAS_AUTO_DECODER
|
||||||
|
LIBLZMA_HAS_EASY_ENCODER
|
||||||
|
LIBLZMA_HAS_LZMA_PRESET
|
||||||
|
VERSION_VAR LIBLZMA_VERSION_STRING
|
||||||
|
)
|
||||||
|
mark_as_advanced( LIBLZMA_INCLUDE_DIR LIBLZMA_LIBRARY )
|
||||||
|
|
||||||
|
if (LIBLZMA_FOUND)
|
||||||
|
set(LIBLZMA_LIBRARIES ${LIBLZMA_LIBRARY})
|
||||||
|
set(LIBLZMA_INCLUDE_DIRS ${LIBLZMA_INCLUDE_DIR})
|
||||||
|
if(NOT TARGET LibLZMA::LibLZMA)
|
||||||
|
add_library(LibLZMA::LibLZMA UNKNOWN IMPORTED)
|
||||||
|
set_target_properties(LibLZMA::LibLZMA PROPERTIES
|
||||||
|
INTERFACE_INCLUDE_DIRECTORIES ${LIBLZMA_INCLUDE_DIR}
|
||||||
|
IMPORTED_LINK_INTERFACE_LANGUAGES C)
|
||||||
|
|
||||||
|
if(LIBLZMA_LIBRARY_RELEASE)
|
||||||
|
set_property(TARGET LibLZMA::LibLZMA APPEND PROPERTY
|
||||||
|
IMPORTED_CONFIGURATIONS RELEASE)
|
||||||
|
set_target_properties(LibLZMA::LibLZMA PROPERTIES
|
||||||
|
IMPORTED_LOCATION_RELEASE "${LIBLZMA_LIBRARY_RELEASE}")
|
||||||
|
endif()
|
||||||
|
|
||||||
|
if(LIBLZMA_LIBRARY_DEBUG)
|
||||||
|
set_property(TARGET LibLZMA::LibLZMA APPEND PROPERTY
|
||||||
|
IMPORTED_CONFIGURATIONS DEBUG)
|
||||||
|
set_target_properties(LibLZMA::LibLZMA PROPERTIES
|
||||||
|
IMPORTED_LOCATION_DEBUG "${LIBLZMA_LIBRARY_DEBUG}")
|
||||||
|
endif()
|
||||||
|
|
||||||
|
if(NOT LIBLZMA_LIBRARY_RELEASE AND NOT LIBLZMA_LIBRARY_DEBUG)
|
||||||
|
set_target_properties(LibLZMA::LibLZMA PROPERTIES
|
||||||
|
IMPORTED_LOCATION "${LIBLZMA_LIBRARY}")
|
||||||
|
endif()
|
||||||
|
endif()
|
||||||
|
endif ()
|
||||||
@@ -1,6 +1,7 @@
|
|||||||
include(CMakeFindDependencyMacro)
|
include(CMakeFindDependencyMacro)
|
||||||
|
|
||||||
find_dependency(ZLIB)
|
find_dependency(ZLIB)
|
||||||
|
find_dependency(LibLZMA)
|
||||||
find_dependency(Threads)
|
find_dependency(Threads)
|
||||||
|
|
||||||
# OSG
|
# OSG
|
||||||
|
|||||||
@@ -168,7 +168,9 @@ target_link_libraries(SimGearCore PRIVATE
|
|||||||
${COCOA_LIBRARY}
|
${COCOA_LIBRARY}
|
||||||
${CURL_LIBRARIES}
|
${CURL_LIBRARIES}
|
||||||
${WINSOCK_LIBRARY}
|
${WINSOCK_LIBRARY}
|
||||||
${SHLWAPI_LIBRARY})
|
${SHLWAPI_LIBRARY}
|
||||||
|
LibLZMA::LibLZMA
|
||||||
|
)
|
||||||
|
|
||||||
if(SYSTEM_EXPAT)
|
if(SYSTEM_EXPAT)
|
||||||
target_link_libraries(SimGearCore PRIVATE ${EXPAT_LIBRARIES})
|
target_link_libraries(SimGearCore PRIVATE ${EXPAT_LIBRARIES})
|
||||||
|
|||||||
@@ -0,0 +1,90 @@
|
|||||||
|
// Copyright (C) 2021 James Turner - <james@flightgear.org>
|
||||||
|
//
|
||||||
|
// This library is free software; you can redistribute it and/or
|
||||||
|
// modify it under the terms of the GNU Library General Public
|
||||||
|
// License as published by the Free Software Foundation; either
|
||||||
|
// version 2 of the License, or (at your option) any later version.
|
||||||
|
//
|
||||||
|
// This library is distributed in the hope that it will be useful,
|
||||||
|
// but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||||
|
// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
|
||||||
|
// Library General Public License for more details.
|
||||||
|
//
|
||||||
|
// You should have received a copy of the GNU General Public License
|
||||||
|
// along with this program; if not, write to the Free Software
|
||||||
|
// Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301, USA.
|
||||||
|
//
|
||||||
|
|
||||||
|
|
||||||
|
#pragma once
|
||||||
|
|
||||||
|
#include "untar.hxx"
|
||||||
|
|
||||||
|
namespace simgear {
|
||||||
|
|
||||||
|
class ArchiveExtractorPrivate
|
||||||
|
{
|
||||||
|
public:
|
||||||
|
ArchiveExtractorPrivate(ArchiveExtractor* o) : outer(o)
|
||||||
|
{
|
||||||
|
assert(outer);
|
||||||
|
}
|
||||||
|
|
||||||
|
virtual ~ArchiveExtractorPrivate() = default;
|
||||||
|
|
||||||
|
typedef enum {
|
||||||
|
INVALID = 0,
|
||||||
|
READING_HEADER,
|
||||||
|
READING_FILE,
|
||||||
|
READING_PADDING,
|
||||||
|
READING_PAX_GLOBAL_ATTRIBUTES,
|
||||||
|
READING_PAX_FILE_ATTRIBUTES,
|
||||||
|
PRE_END_OF_ARCHVE,
|
||||||
|
END_OF_ARCHIVE,
|
||||||
|
ERROR_STATE, ///< states above this are error conditions
|
||||||
|
BAD_ARCHIVE,
|
||||||
|
BAD_DATA,
|
||||||
|
FILTER_STOPPED
|
||||||
|
} State;
|
||||||
|
|
||||||
|
State state = INVALID;
|
||||||
|
ArchiveExtractor* outer = nullptr;
|
||||||
|
|
||||||
|
virtual void extractBytes(const uint8_t* bytes, size_t count) = 0;
|
||||||
|
|
||||||
|
virtual void flush() = 0;
|
||||||
|
|
||||||
|
SGPath extractRootPath()
|
||||||
|
{
|
||||||
|
return outer->_rootPath;
|
||||||
|
}
|
||||||
|
|
||||||
|
ArchiveExtractor::PathResult filterPath(std::string& pathToExtract)
|
||||||
|
{
|
||||||
|
return outer->filterPath(pathToExtract);
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
bool isSafePath(const std::string& p) const
|
||||||
|
{
|
||||||
|
if (p.empty()) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
// reject absolute paths
|
||||||
|
if (p.at(0) == '/') {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
// reject paths containing '..'
|
||||||
|
size_t doubleDot = p.find("..");
|
||||||
|
if (doubleDot != std::string::npos) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
// on POSIX could use realpath to sanity check
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
} // namespace simgear
|
||||||
@@ -265,6 +265,19 @@ public:
|
|||||||
free(buf);
|
free(buf);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// helper to check and erase 'fooBar' from paths, if passed fooBar.zip, fooBar.tgz, etc.
|
||||||
|
void removeExtractedDirectoryFromList(PathList& paths, const std::string& tarballName)
|
||||||
|
{
|
||||||
|
const auto directoryName = SGPath::fromUtf8(tarballName).file_base();
|
||||||
|
auto it = std::find_if(paths.begin(), paths.end(), [directoryName](const SGPath& p) {
|
||||||
|
return p.isDir() && (p.file() == directoryName);
|
||||||
|
});
|
||||||
|
|
||||||
|
if (it != paths.end()) {
|
||||||
|
paths.erase(it);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
void updateChildrenBasedOnHash()
|
void updateChildrenBasedOnHash()
|
||||||
{
|
{
|
||||||
using SAct = HTTPRepository::SyncAction;
|
using SAct = HTTPRepository::SyncAction;
|
||||||
@@ -298,6 +311,11 @@ public:
|
|||||||
orphans.end());
|
orphans.end());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ensure the extracted directory corresponding to a tarball, is *not* considered an orphan
|
||||||
|
if (c.type == HTTPRepository::TarballType) {
|
||||||
|
removeExtractedDirectoryFromList(orphans, c.name);
|
||||||
|
}
|
||||||
|
|
||||||
if (_repository->syncPredicate) {
|
if (_repository->syncPredicate) {
|
||||||
const auto pathOnDisk = isNew ? absolutePath() / c.name : *p;
|
const auto pathOnDisk = isNew ? absolutePath() / c.name : *p;
|
||||||
// never handle deletes here, do them at the end
|
// never handle deletes here, do them at the end
|
||||||
@@ -320,7 +338,6 @@ public:
|
|||||||
toBeUpdated.push_back(c);
|
toBeUpdated.push_back(c);
|
||||||
} else {
|
} else {
|
||||||
// File/Directory exists and hash is valid.
|
// File/Directory exists and hash is valid.
|
||||||
|
|
||||||
if (c.type == HTTPRepository::DirectoryType) {
|
if (c.type == HTTPRepository::DirectoryType) {
|
||||||
// If it's a directory,perform a recursive check.
|
// If it's a directory,perform a recursive check.
|
||||||
HTTPDirectory *childDir = childDirectory(c.name);
|
HTTPDirectory *childDir = childDirectory(c.name);
|
||||||
@@ -509,42 +526,42 @@ public:
|
|||||||
_repository->totalDownloaded += sz;
|
_repository->totalDownloaded += sz;
|
||||||
SGPath p = SGPath(absolutePath(), file);
|
SGPath p = SGPath(absolutePath(), file);
|
||||||
|
|
||||||
if ((p.extension() == "tgz") || (p.extension() == "zip")) {
|
if (it->type == HTTPRepository::TarballType) {
|
||||||
// We require that any compressed files have the same filename as the file or directory
|
// We require that any compressed files have the same filename as the file or directory
|
||||||
// they expand to, so we can remove the old file/directory before extracting the new
|
// they expand to, so we can remove the old file/directory before extracting the new
|
||||||
// data.
|
// data.
|
||||||
SGPath removePath = SGPath(p.base());
|
SGPath removePath = SGPath(p.base());
|
||||||
bool pathAvailable = true;
|
bool pathAvailable = true;
|
||||||
if (removePath.exists()) {
|
if (removePath.exists()) {
|
||||||
if (removePath.isDir()) {
|
if (removePath.isDir()) {
|
||||||
simgear::Dir pd(removePath);
|
simgear::Dir pd(removePath);
|
||||||
pathAvailable = pd.removeChildren();
|
pathAvailable = pd.removeChildren();
|
||||||
} else {
|
} else {
|
||||||
pathAvailable = removePath.remove();
|
pathAvailable = removePath.remove();
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
|
||||||
|
|
||||||
if (pathAvailable) {
|
if (pathAvailable) {
|
||||||
// we use a Task helper to extract tarballs incrementally.
|
// we use a Task helper to extract tarballs incrementally.
|
||||||
// without this, archive extraction blocks here, which
|
// without this, archive extraction blocks here, which
|
||||||
// prevents other repositories downloading / updating.
|
// prevents other repositories downloading / updating.
|
||||||
// Unfortunately due Windows AV (Defender, etc) we cna block
|
// Unfortunately due Windows AV (Defender, etc) we cna block
|
||||||
// here for many minutes.
|
// here for many minutes.
|
||||||
|
|
||||||
// use a lambda to own this shared_ptr; this means when the
|
// use a lambda to own this shared_ptr; this means when the
|
||||||
// lambda is destroyed, the ArchiveExtraTask will get
|
// lambda is destroyed, the ArchiveExtraTask will get
|
||||||
// cleaned up.
|
// cleaned up.
|
||||||
ArchiveExtractTaskPtr t =
|
ArchiveExtractTaskPtr t =
|
||||||
std::make_shared<ArchiveExtractTask>(p, _relativePath);
|
std::make_shared<ArchiveExtractTask>(p, _relativePath);
|
||||||
auto cb = [t](HTTPRepoPrivate *repo) {
|
auto cb = [t](HTTPRepoPrivate* repo) {
|
||||||
return t->run(repo);
|
return t->run(repo);
|
||||||
};
|
};
|
||||||
|
|
||||||
_repository->addTask(cb);
|
_repository->addTask(cb);
|
||||||
} else {
|
} else {
|
||||||
SG_LOG(SG_TERRASYNC, SG_ALERT, "Unable to remove old file/directory " << removePath);
|
SG_LOG(SG_TERRASYNC, SG_ALERT, "Unable to remove old file/directory " << removePath);
|
||||||
} // of pathAvailable
|
} // of pathAvailable
|
||||||
} // of handling tgz files
|
} // of handling archive files
|
||||||
} // of hash matches
|
} // of hash matches
|
||||||
} // of found in child list
|
} // of found in child list
|
||||||
}
|
}
|
||||||
@@ -736,11 +753,9 @@ private:
|
|||||||
std::string hashForChild(const ChildInfo& child) const
|
std::string hashForChild(const ChildInfo& child) const
|
||||||
{
|
{
|
||||||
SGPath p(child.path);
|
SGPath p(child.path);
|
||||||
if (child.type == HTTPRepository::DirectoryType)
|
if (child.type == HTTPRepository::DirectoryType) {
|
||||||
p.append(".dirindex");
|
p.append(".dirindex");
|
||||||
if (child.type == HTTPRepository::TarballType)
|
}
|
||||||
p.concat(
|
|
||||||
".tgz"); // For tarballs the hash is against the tarball file itself
|
|
||||||
return hashForPath(p);
|
return hashForPath(p);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
Binary file not shown.
@@ -32,7 +32,7 @@ void testTarGz()
|
|||||||
uint8_t* buf = (uint8_t*) alloca(8192);
|
uint8_t* buf = (uint8_t*) alloca(8192);
|
||||||
size_t bufSize = f.read((char*) buf, 8192);
|
size_t bufSize = f.read((char*) buf, 8192);
|
||||||
|
|
||||||
SG_VERIFY(ArchiveExtractor::determineType(buf, bufSize) == ArchiveExtractor::TarData);
|
SG_VERIFY(ArchiveExtractor::determineType(buf, bufSize) == ArchiveExtractor::GZData);
|
||||||
|
|
||||||
f.close();
|
f.close();
|
||||||
}
|
}
|
||||||
@@ -174,6 +174,34 @@ void testPAXAttributes()
|
|||||||
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
void testExtractXZ()
|
||||||
|
{
|
||||||
|
SGPath p = SGPath(SRC_DIR);
|
||||||
|
p.append("test.tar.xz");
|
||||||
|
|
||||||
|
SGBinaryFile f(p);
|
||||||
|
f.open(SG_IO_IN);
|
||||||
|
|
||||||
|
SGPath extractDir = simgear::Dir::current().path() / "test_extract_xz";
|
||||||
|
simgear::Dir pd(extractDir);
|
||||||
|
pd.removeChildren();
|
||||||
|
|
||||||
|
ArchiveExtractor ex(extractDir);
|
||||||
|
|
||||||
|
uint8_t* buf = (uint8_t*)alloca(128);
|
||||||
|
while (!f.eof()) {
|
||||||
|
size_t bufSize = f.read((char*)buf, 128);
|
||||||
|
ex.extractBytes(buf, bufSize);
|
||||||
|
}
|
||||||
|
|
||||||
|
ex.flush();
|
||||||
|
SG_VERIFY(ex.isAtEndOfArchive());
|
||||||
|
SG_VERIFY(ex.hasError() == false);
|
||||||
|
|
||||||
|
SG_VERIFY((extractDir / "testDir/hello.c").exists());
|
||||||
|
SG_VERIFY((extractDir / "testDir/foo.txt").exists());
|
||||||
|
}
|
||||||
|
|
||||||
int main(int ac, char ** av)
|
int main(int ac, char ** av)
|
||||||
{
|
{
|
||||||
testTarGz();
|
testTarGz();
|
||||||
@@ -181,7 +209,8 @@ int main(int ac, char ** av)
|
|||||||
testFilterTar();
|
testFilterTar();
|
||||||
testExtractStreamed();
|
testExtractStreamed();
|
||||||
testExtractZip();
|
testExtractZip();
|
||||||
|
testExtractXZ();
|
||||||
|
|
||||||
// disabled to avoiding checking in large PAX archive
|
// disabled to avoiding checking in large PAX archive
|
||||||
// testPAXAttributes();
|
// testPAXAttributes();
|
||||||
|
|
||||||
|
|||||||
+406
-372
@@ -38,82 +38,11 @@
|
|||||||
#include <simgear/package/unzip.h>
|
#include <simgear/package/unzip.h>
|
||||||
#include <simgear/structure/exception.hxx>
|
#include <simgear/structure/exception.hxx>
|
||||||
|
|
||||||
|
#include "ArchiveExtractor_private.hxx"
|
||||||
|
|
||||||
namespace simgear
|
namespace simgear
|
||||||
{
|
{
|
||||||
|
|
||||||
class ArchiveExtractorPrivate
|
|
||||||
{
|
|
||||||
public:
|
|
||||||
ArchiveExtractorPrivate(ArchiveExtractor* o) :
|
|
||||||
outer(o)
|
|
||||||
{
|
|
||||||
assert(outer);
|
|
||||||
}
|
|
||||||
|
|
||||||
virtual ~ArchiveExtractorPrivate() = default;
|
|
||||||
|
|
||||||
typedef enum {
|
|
||||||
INVALID = 0,
|
|
||||||
READING_HEADER,
|
|
||||||
READING_FILE,
|
|
||||||
READING_PADDING,
|
|
||||||
READING_PAX_GLOBAL_ATTRIBUTES,
|
|
||||||
READING_PAX_FILE_ATTRIBUTES,
|
|
||||||
PRE_END_OF_ARCHVE,
|
|
||||||
END_OF_ARCHIVE,
|
|
||||||
ERROR_STATE, ///< states above this are error conditions
|
|
||||||
BAD_ARCHIVE,
|
|
||||||
BAD_DATA,
|
|
||||||
FILTER_STOPPED
|
|
||||||
} State;
|
|
||||||
|
|
||||||
State state = INVALID;
|
|
||||||
ArchiveExtractor* outer = nullptr;
|
|
||||||
|
|
||||||
virtual void extractBytes(const uint8_t* bytes, size_t count) = 0;
|
|
||||||
|
|
||||||
virtual void flush() = 0;
|
|
||||||
|
|
||||||
SGPath extractRootPath()
|
|
||||||
{
|
|
||||||
return outer->_rootPath;
|
|
||||||
}
|
|
||||||
|
|
||||||
ArchiveExtractor::PathResult filterPath(std::string& pathToExtract)
|
|
||||||
{
|
|
||||||
return outer->filterPath(pathToExtract);
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
bool isSafePath(const std::string& p) const
|
|
||||||
{
|
|
||||||
if (p.empty()) {
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
// reject absolute paths
|
|
||||||
if (p.at(0) == '/') {
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
// reject paths containing '..'
|
|
||||||
size_t doubleDot = p.find("..");
|
|
||||||
if (doubleDot != std::string::npos) {
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
// on POSIX could use realpath to sanity check
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
};
|
|
||||||
|
|
||||||
///////////////////////////////////////////////////////////////////////////////////////////////////
|
|
||||||
|
|
||||||
const int ZLIB_DECOMPRESS_BUFFER_SIZE = 32 * 1024;
|
|
||||||
const int ZLIB_INFLATE_WINDOW_BITS = MAX_WBITS;
|
|
||||||
const int ZLIB_DECODE_GZIP_HEADER = 16;
|
|
||||||
|
|
||||||
/* tar Header Block, from POSIX 1003.1-1990. */
|
|
||||||
|
|
||||||
typedef struct
|
typedef struct
|
||||||
{
|
{
|
||||||
@@ -153,320 +82,410 @@ typedef struct
|
|||||||
#define FIFOTYPE '6' /* FIFO special */
|
#define FIFOTYPE '6' /* FIFO special */
|
||||||
#define CONTTYPE '7' /* reserved */
|
#define CONTTYPE '7' /* reserved */
|
||||||
|
|
||||||
const char PAX_GLOBAL_HEADER = 'g';
|
const char PAX_GLOBAL_HEADER = 'g';
|
||||||
const char PAX_FILE_ATTRIBUTES = 'x';
|
const char PAX_FILE_ATTRIBUTES = 'x';
|
||||||
|
|
||||||
|
|
||||||
class TarExtractorPrivate : public ArchiveExtractorPrivate
|
///////////////////////////////////////////////////////////////////////////////////////////////////
|
||||||
|
|
||||||
|
const int ZLIB_DECOMPRESS_BUFFER_SIZE = 32 * 1024;
|
||||||
|
const int ZLIB_INFLATE_WINDOW_BITS = MAX_WBITS;
|
||||||
|
const int ZLIB_DECODE_GZIP_HEADER = 16;
|
||||||
|
|
||||||
|
/* tar Header Block, from POSIX 1003.1-1990. */
|
||||||
|
|
||||||
|
|
||||||
|
class TarExtractorPrivate : public ArchiveExtractorPrivate
|
||||||
|
{
|
||||||
|
public:
|
||||||
|
union {
|
||||||
|
UstarHeaderBlock header;
|
||||||
|
uint8_t headerBytes[TAR_HEADER_BLOCK_SIZE];
|
||||||
|
};
|
||||||
|
|
||||||
|
size_t bytesRemaining;
|
||||||
|
std::unique_ptr<SGFile> currentFile;
|
||||||
|
size_t currentFileSize;
|
||||||
|
|
||||||
|
uint8_t* headerPtr;
|
||||||
|
bool skipCurrentEntry = false;
|
||||||
|
std::string paxAttributes;
|
||||||
|
std::string paxPathName;
|
||||||
|
|
||||||
|
TarExtractorPrivate(ArchiveExtractor* o) : ArchiveExtractorPrivate(o)
|
||||||
|
{
|
||||||
|
setState(TarExtractorPrivate::READING_HEADER);
|
||||||
|
}
|
||||||
|
|
||||||
|
~TarExtractorPrivate() = default;
|
||||||
|
|
||||||
|
void readPaddingIfRequired()
|
||||||
|
{
|
||||||
|
size_t pad = currentFileSize % TAR_HEADER_BLOCK_SIZE;
|
||||||
|
if (pad) {
|
||||||
|
bytesRemaining = TAR_HEADER_BLOCK_SIZE - pad;
|
||||||
|
setState(READING_PADDING);
|
||||||
|
} else {
|
||||||
|
setState(READING_HEADER);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
void checkEndOfState()
|
||||||
|
{
|
||||||
|
if (bytesRemaining > 0) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (state == READING_FILE) {
|
||||||
|
if (currentFile) {
|
||||||
|
currentFile->close();
|
||||||
|
currentFile.reset();
|
||||||
|
}
|
||||||
|
readPaddingIfRequired();
|
||||||
|
} else if (state == READING_HEADER) {
|
||||||
|
processHeader();
|
||||||
|
} else if (state == PRE_END_OF_ARCHVE) {
|
||||||
|
if (headerIsAllZeros()) {
|
||||||
|
setState(END_OF_ARCHIVE);
|
||||||
|
} else {
|
||||||
|
// what does the spec say here?
|
||||||
|
}
|
||||||
|
} else if (state == READING_PAX_GLOBAL_ATTRIBUTES) {
|
||||||
|
parsePAXAttributes(true);
|
||||||
|
readPaddingIfRequired();
|
||||||
|
} else if (state == READING_PAX_FILE_ATTRIBUTES) {
|
||||||
|
parsePAXAttributes(false);
|
||||||
|
readPaddingIfRequired();
|
||||||
|
} else if (state == READING_PADDING) {
|
||||||
|
setState(READING_HEADER);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
void setState(State newState)
|
||||||
|
{
|
||||||
|
if ((newState == READING_HEADER) || (newState == PRE_END_OF_ARCHVE)) {
|
||||||
|
bytesRemaining = TAR_HEADER_BLOCK_SIZE;
|
||||||
|
headerPtr = headerBytes;
|
||||||
|
}
|
||||||
|
|
||||||
|
state = newState;
|
||||||
|
}
|
||||||
|
|
||||||
|
void extractBytes(const uint8_t* bytes, size_t count) override
|
||||||
|
{
|
||||||
|
// uncompressed, just pass through directly
|
||||||
|
processBytes((const char*)bytes, count);
|
||||||
|
}
|
||||||
|
|
||||||
|
void flush() override
|
||||||
|
{
|
||||||
|
// no-op for tar files, we process everything greedily
|
||||||
|
}
|
||||||
|
|
||||||
|
void processHeader()
|
||||||
|
{
|
||||||
|
if (headerIsAllZeros()) {
|
||||||
|
if (state == PRE_END_OF_ARCHVE) {
|
||||||
|
setState(END_OF_ARCHIVE);
|
||||||
|
} else {
|
||||||
|
setState(PRE_END_OF_ARCHVE);
|
||||||
|
}
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (strncmp(header.magic, TMAGIC, TMAGLEN) != 0) {
|
||||||
|
SG_LOG(SG_IO, SG_WARN, "Untar: magic is wrong");
|
||||||
|
state = BAD_ARCHIVE;
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
skipCurrentEntry = false;
|
||||||
|
std::string tarPath = std::string(header.prefix) + std::string(header.fileName);
|
||||||
|
if (!paxPathName.empty()) {
|
||||||
|
tarPath = paxPathName;
|
||||||
|
paxPathName.clear(); // clear for next file
|
||||||
|
}
|
||||||
|
|
||||||
|
if (!isSafePath(tarPath)) {
|
||||||
|
SG_LOG(SG_IO, SG_WARN, "unsafe tar path, skipping::" << tarPath);
|
||||||
|
skipCurrentEntry = true;
|
||||||
|
}
|
||||||
|
|
||||||
|
auto result = filterPath(tarPath);
|
||||||
|
if (result == ArchiveExtractor::Stop) {
|
||||||
|
setState(FILTER_STOPPED);
|
||||||
|
return;
|
||||||
|
} else if (result == ArchiveExtractor::Skipped) {
|
||||||
|
skipCurrentEntry = true;
|
||||||
|
}
|
||||||
|
|
||||||
|
SGPath p = extractRootPath() / tarPath;
|
||||||
|
if (header.typeflag == DIRTYPE) {
|
||||||
|
if (!skipCurrentEntry) {
|
||||||
|
Dir dir(p);
|
||||||
|
dir.create(0755);
|
||||||
|
}
|
||||||
|
setState(READING_HEADER);
|
||||||
|
} else if ((header.typeflag == REGTYPE) || (header.typeflag == AREGTYPE)) {
|
||||||
|
currentFileSize = ::strtol(header.size, NULL, 8);
|
||||||
|
bytesRemaining = currentFileSize;
|
||||||
|
if (!skipCurrentEntry) {
|
||||||
|
currentFile.reset(new SGBinaryFile(p));
|
||||||
|
currentFile->open(SG_IO_OUT);
|
||||||
|
}
|
||||||
|
setState(READING_FILE);
|
||||||
|
} else if (header.typeflag == PAX_GLOBAL_HEADER) {
|
||||||
|
setState(READING_PAX_GLOBAL_ATTRIBUTES);
|
||||||
|
currentFileSize = ::strtol(header.size, NULL, 8);
|
||||||
|
bytesRemaining = currentFileSize;
|
||||||
|
paxAttributes.clear();
|
||||||
|
} else if (header.typeflag == PAX_FILE_ATTRIBUTES) {
|
||||||
|
setState(READING_PAX_FILE_ATTRIBUTES);
|
||||||
|
currentFileSize = ::strtol(header.size, NULL, 8);
|
||||||
|
bytesRemaining = currentFileSize;
|
||||||
|
paxAttributes.clear();
|
||||||
|
} else if ((header.typeflag == SYMTYPE) || (header.typeflag == LNKTYPE)) {
|
||||||
|
SG_LOG(SG_IO, SG_WARN, "Tarball contains a link or symlink, will be skipped:" << tarPath);
|
||||||
|
skipCurrentEntry = true;
|
||||||
|
setState(READING_HEADER);
|
||||||
|
} else {
|
||||||
|
SG_LOG(SG_IO, SG_WARN, "Unsupported tar file type:" << header.typeflag);
|
||||||
|
state = BAD_ARCHIVE;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
void processBytes(const char* bytes, size_t count)
|
||||||
|
{
|
||||||
|
if ((state >= ERROR_STATE) || (state == END_OF_ARCHIVE)) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
size_t curBytes = std::min(bytesRemaining, count);
|
||||||
|
if (state == READING_FILE) {
|
||||||
|
if (currentFile) {
|
||||||
|
currentFile->write(bytes, curBytes);
|
||||||
|
}
|
||||||
|
bytesRemaining -= curBytes;
|
||||||
|
} else if ((state == READING_HEADER) || (state == PRE_END_OF_ARCHVE) || (state == END_OF_ARCHIVE)) {
|
||||||
|
memcpy(headerPtr, bytes, curBytes);
|
||||||
|
bytesRemaining -= curBytes;
|
||||||
|
headerPtr += curBytes;
|
||||||
|
} else if (state == READING_PADDING) {
|
||||||
|
bytesRemaining -= curBytes;
|
||||||
|
} else if ((state == READING_PAX_FILE_ATTRIBUTES) || (state == READING_PAX_GLOBAL_ATTRIBUTES)) {
|
||||||
|
bytesRemaining -= curBytes;
|
||||||
|
paxAttributes.append(bytes, curBytes);
|
||||||
|
}
|
||||||
|
|
||||||
|
checkEndOfState();
|
||||||
|
if (count > curBytes) {
|
||||||
|
// recurse with unprocessed bytes
|
||||||
|
processBytes(bytes + curBytes, count - curBytes);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
bool headerIsAllZeros() const
|
||||||
|
{
|
||||||
|
char* headerAsChar = (char*)&header;
|
||||||
|
for (size_t i = 0; i < offsetof(UstarHeaderBlock, magic); ++i) {
|
||||||
|
if (*headerAsChar++ != 0) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
// https://www.ibm.com/support/knowledgecenter/en/SSLTBW_2.3.0/com.ibm.zos.v2r3.bpxa500/paxex.htm#paxex
|
||||||
|
void parsePAXAttributes(bool areGlobal)
|
||||||
|
{
|
||||||
|
auto lineStart = 0;
|
||||||
|
for (;;) {
|
||||||
|
auto firstSpace = paxAttributes.find(' ', lineStart);
|
||||||
|
auto firstEq = paxAttributes.find('=', lineStart);
|
||||||
|
if ((firstEq == std::string::npos) || (firstSpace == std::string::npos)) {
|
||||||
|
SG_LOG(SG_IO, SG_WARN, "Malfroemd PAX attributes in tarfile");
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
|
||||||
|
uint32_t lengthBytes = std::stoul(paxAttributes.substr(lineStart, firstSpace));
|
||||||
|
uint32_t dataBytes = lengthBytes - (firstEq + 1);
|
||||||
|
std::string name = paxAttributes.substr(firstSpace + 1, firstEq - (firstSpace + 1));
|
||||||
|
|
||||||
|
// dataBytes - 1 here to trim off the trailing newline
|
||||||
|
std::string data = paxAttributes.substr(firstEq + 1, dataBytes - 1);
|
||||||
|
|
||||||
|
processPAXAttribute(areGlobal, name, data);
|
||||||
|
|
||||||
|
lineStart += lengthBytes;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
void processPAXAttribute(bool isGlobalAttr, const std::string& attrName, const std::string& data)
|
||||||
|
{
|
||||||
|
if (!isGlobalAttr && (attrName == "path")) {
|
||||||
|
// data is UTF-8 encoded path name
|
||||||
|
paxPathName = data;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
///////////////////////////////////////////////////////////////////////////////
|
||||||
|
|
||||||
|
|
||||||
|
class GZTarExtractor : public TarExtractorPrivate
|
||||||
{
|
{
|
||||||
public:
|
public:
|
||||||
|
GZTarExtractor(ArchiveExtractor* outer) : TarExtractorPrivate(outer)
|
||||||
union {
|
|
||||||
UstarHeaderBlock header;
|
|
||||||
uint8_t headerBytes[TAR_HEADER_BLOCK_SIZE];
|
|
||||||
};
|
|
||||||
|
|
||||||
size_t bytesRemaining;
|
|
||||||
std::unique_ptr<SGFile> currentFile;
|
|
||||||
size_t currentFileSize;
|
|
||||||
z_stream zlibStream;
|
|
||||||
uint8_t* zlibOutput;
|
|
||||||
bool haveInitedZLib = false;
|
|
||||||
bool uncompressedData = false; // set if reading a plain .tar (not tar.gz)
|
|
||||||
uint8_t* headerPtr;
|
|
||||||
bool skipCurrentEntry = false;
|
|
||||||
std::string paxAttributes;
|
|
||||||
std::string paxPathName;
|
|
||||||
|
|
||||||
TarExtractorPrivate(ArchiveExtractor* o) :
|
|
||||||
ArchiveExtractorPrivate(o)
|
|
||||||
{
|
{
|
||||||
memset(&zlibStream, 0, sizeof(z_stream));
|
memset(&zlibStream, 0, sizeof(z_stream));
|
||||||
zlibOutput = (unsigned char*)malloc(ZLIB_DECOMPRESS_BUFFER_SIZE);
|
zlibOutput = (unsigned char*)malloc(ZLIB_DECOMPRESS_BUFFER_SIZE);
|
||||||
zlibStream.zalloc = Z_NULL;
|
zlibStream.zalloc = Z_NULL;
|
||||||
zlibStream.zfree = Z_NULL;
|
zlibStream.zfree = Z_NULL;
|
||||||
zlibStream.avail_out = ZLIB_DECOMPRESS_BUFFER_SIZE;
|
zlibStream.avail_out = ZLIB_DECOMPRESS_BUFFER_SIZE;
|
||||||
zlibStream.next_out = zlibOutput;
|
zlibStream.next_out = zlibOutput;
|
||||||
}
|
}
|
||||||
|
|
||||||
~TarExtractorPrivate()
|
~GZTarExtractor()
|
||||||
{
|
{
|
||||||
free(zlibOutput);
|
free(zlibOutput);
|
||||||
}
|
}
|
||||||
|
|
||||||
void readPaddingIfRequired()
|
void extractBytes(const uint8_t* bytes, size_t count) override
|
||||||
{
|
{
|
||||||
size_t pad = currentFileSize % TAR_HEADER_BLOCK_SIZE;
|
zlibStream.next_in = (uint8_t*)bytes;
|
||||||
if (pad) {
|
zlibStream.avail_in = count;
|
||||||
bytesRemaining = TAR_HEADER_BLOCK_SIZE - pad;
|
|
||||||
setState(READING_PADDING);
|
|
||||||
} else {
|
|
||||||
setState(READING_HEADER);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
void checkEndOfState()
|
|
||||||
{
|
|
||||||
if (bytesRemaining > 0) {
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
if (state == READING_FILE) {
|
if (!haveInitedZLib) {
|
||||||
if (currentFile) {
|
// now we have data, see if we're dealing with GZ-compressed data or not
|
||||||
currentFile->close();
|
if ((bytes[0] == 0x1f) && (bytes[1] == 0x8b)) {
|
||||||
currentFile.reset();
|
// GZIP identification bytes
|
||||||
}
|
if (inflateInit2(&zlibStream, ZLIB_INFLATE_WINDOW_BITS | ZLIB_DECODE_GZIP_HEADER) != Z_OK) {
|
||||||
readPaddingIfRequired();
|
SG_LOG(SG_IO, SG_WARN, "inflateInit2 failed");
|
||||||
} else if (state == READING_HEADER) {
|
state = TarExtractorPrivate::BAD_DATA;
|
||||||
processHeader();
|
return;
|
||||||
} else if (state == PRE_END_OF_ARCHVE) {
|
}
|
||||||
if (headerIsAllZeros()) {
|
|
||||||
setState(END_OF_ARCHIVE);
|
|
||||||
} else {
|
} else {
|
||||||
// what does the spec say here?
|
// set error state
|
||||||
|
state = TarExtractorPrivate::BAD_DATA;
|
||||||
|
return;
|
||||||
}
|
}
|
||||||
} else if (state == READING_PAX_GLOBAL_ATTRIBUTES) {
|
|
||||||
parsePAXAttributes(true);
|
|
||||||
readPaddingIfRequired();
|
|
||||||
} else if (state == READING_PAX_FILE_ATTRIBUTES) {
|
|
||||||
parsePAXAttributes(false);
|
|
||||||
readPaddingIfRequired();
|
|
||||||
} else if (state == READING_PADDING) {
|
|
||||||
setState(READING_HEADER);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
void setState(State newState)
|
haveInitedZLib = true;
|
||||||
{
|
setState(TarExtractorPrivate::READING_HEADER);
|
||||||
if ((newState == READING_HEADER) || (newState == PRE_END_OF_ARCHVE)) {
|
} // of init on first-bytes case
|
||||||
bytesRemaining = TAR_HEADER_BLOCK_SIZE;
|
|
||||||
headerPtr = headerBytes;
|
|
||||||
}
|
|
||||||
|
|
||||||
state = newState;
|
size_t writtenSize;
|
||||||
}
|
// loop, running zlib() inflate and sending output bytes to
|
||||||
|
// our request body handler. Keep calling inflate until no bytes are
|
||||||
|
// written, and ZLIB has consumed all available input
|
||||||
|
do {
|
||||||
|
zlibStream.next_out = zlibOutput;
|
||||||
|
zlibStream.avail_out = ZLIB_DECOMPRESS_BUFFER_SIZE;
|
||||||
|
int result = inflate(&zlibStream, Z_NO_FLUSH);
|
||||||
|
if (result == Z_OK || result == Z_STREAM_END) {
|
||||||
|
// nothing to do
|
||||||
|
|
||||||
void extractBytes(const uint8_t* bytes, size_t count) override
|
} else if (result == Z_BUF_ERROR) {
|
||||||
{
|
// transient error, fall through
|
||||||
zlibStream.next_in = (uint8_t*) bytes;
|
|
||||||
zlibStream.avail_in = count;
|
|
||||||
|
|
||||||
if (!haveInitedZLib) {
|
|
||||||
// now we have data, see if we're dealing with GZ-compressed data or not
|
|
||||||
if ((bytes[0] == 0x1f) && (bytes[1] == 0x8b)) {
|
|
||||||
// GZIP identification bytes
|
|
||||||
if (inflateInit2(&zlibStream, ZLIB_INFLATE_WINDOW_BITS | ZLIB_DECODE_GZIP_HEADER) != Z_OK) {
|
|
||||||
SG_LOG(SG_IO, SG_WARN, "inflateInit2 failed");
|
|
||||||
state = TarExtractorPrivate::BAD_DATA;
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
} else {
|
|
||||||
UstarHeaderBlock* header = (UstarHeaderBlock*)bytes;
|
|
||||||
if (strncmp(header->magic, TMAGIC, TMAGLEN) != 0) {
|
|
||||||
SG_LOG(SG_IO, SG_WARN, "didn't find tar magic in header");
|
|
||||||
state = TarExtractorPrivate::BAD_DATA;
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
uncompressedData = true;
|
|
||||||
}
|
|
||||||
|
|
||||||
haveInitedZLib = true;
|
|
||||||
setState(TarExtractorPrivate::READING_HEADER);
|
|
||||||
} // of init on first-bytes case
|
|
||||||
|
|
||||||
if (uncompressedData) {
|
|
||||||
processBytes((const char*) bytes, count);
|
|
||||||
} else {
|
|
||||||
size_t writtenSize;
|
|
||||||
// loop, running zlib() inflate and sending output bytes to
|
|
||||||
// our request body handler. Keep calling inflate until no bytes are
|
|
||||||
// written, and ZLIB has consumed all available input
|
|
||||||
do {
|
|
||||||
zlibStream.next_out = zlibOutput;
|
|
||||||
zlibStream.avail_out = ZLIB_DECOMPRESS_BUFFER_SIZE;
|
|
||||||
int result = inflate(&zlibStream, Z_NO_FLUSH);
|
|
||||||
if (result == Z_OK || result == Z_STREAM_END) {
|
|
||||||
// nothing to do
|
|
||||||
|
|
||||||
}
|
|
||||||
else if (result == Z_BUF_ERROR) {
|
|
||||||
// transient error, fall through
|
|
||||||
}
|
|
||||||
else {
|
|
||||||
// _error = result;
|
|
||||||
SG_LOG(SG_IO, SG_WARN, "Permanent ZLib error:" << zlibStream.msg);
|
|
||||||
state = TarExtractorPrivate::BAD_DATA;
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
writtenSize = ZLIB_DECOMPRESS_BUFFER_SIZE - zlibStream.avail_out;
|
|
||||||
if (writtenSize > 0) {
|
|
||||||
processBytes((const char*) zlibOutput, writtenSize);
|
|
||||||
}
|
|
||||||
|
|
||||||
if (result == Z_STREAM_END) {
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
} while ((zlibStream.avail_in > 0) || (writtenSize > 0));
|
|
||||||
} // of Zlib-compressed data
|
|
||||||
}
|
|
||||||
|
|
||||||
void flush() override
|
|
||||||
{
|
|
||||||
// no-op for tar files, we process everything greedily
|
|
||||||
}
|
|
||||||
|
|
||||||
void processHeader()
|
|
||||||
{
|
|
||||||
if (headerIsAllZeros()) {
|
|
||||||
if (state == PRE_END_OF_ARCHVE) {
|
|
||||||
setState(END_OF_ARCHIVE);
|
|
||||||
} else {
|
} else {
|
||||||
setState(PRE_END_OF_ARCHVE);
|
// _error = result;
|
||||||
|
SG_LOG(SG_IO, SG_WARN, "Permanent ZLib error:" << zlibStream.msg);
|
||||||
|
state = TarExtractorPrivate::BAD_DATA;
|
||||||
|
return;
|
||||||
}
|
}
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
if (strncmp(header.magic, TMAGIC, TMAGLEN) != 0) {
|
writtenSize = ZLIB_DECOMPRESS_BUFFER_SIZE - zlibStream.avail_out;
|
||||||
SG_LOG(SG_IO, SG_WARN, "Untar: magic is wrong");
|
if (writtenSize > 0) {
|
||||||
state = BAD_ARCHIVE;
|
processBytes((const char*)zlibOutput, writtenSize);
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
skipCurrentEntry = false;
|
|
||||||
std::string tarPath = std::string(header.prefix) + std::string(header.fileName);
|
|
||||||
if (!paxPathName.empty()) {
|
|
||||||
tarPath = paxPathName;
|
|
||||||
paxPathName.clear(); // clear for next file
|
|
||||||
}
|
|
||||||
|
|
||||||
if (!isSafePath(tarPath)) {
|
|
||||||
SG_LOG(SG_IO, SG_WARN, "unsafe tar path, skipping::" << tarPath);
|
|
||||||
skipCurrentEntry = true;
|
|
||||||
}
|
|
||||||
|
|
||||||
auto result = filterPath(tarPath);
|
|
||||||
if (result == ArchiveExtractor::Stop) {
|
|
||||||
setState(FILTER_STOPPED);
|
|
||||||
return;
|
|
||||||
} else if (result == ArchiveExtractor::Skipped) {
|
|
||||||
skipCurrentEntry = true;
|
|
||||||
}
|
|
||||||
|
|
||||||
SGPath p = extractRootPath() / tarPath;
|
|
||||||
if (header.typeflag == DIRTYPE) {
|
|
||||||
if (!skipCurrentEntry) {
|
|
||||||
Dir dir(p);
|
|
||||||
dir.create(0755);
|
|
||||||
}
|
}
|
||||||
setState(READING_HEADER);
|
|
||||||
} else if ((header.typeflag == REGTYPE) || (header.typeflag == AREGTYPE)) {
|
|
||||||
currentFileSize = ::strtol(header.size, NULL, 8);
|
|
||||||
bytesRemaining = currentFileSize;
|
|
||||||
if (!skipCurrentEntry) {
|
|
||||||
currentFile.reset(new SGBinaryFile(p));
|
|
||||||
currentFile->open(SG_IO_OUT);
|
|
||||||
}
|
|
||||||
setState(READING_FILE);
|
|
||||||
} else if (header.typeflag == PAX_GLOBAL_HEADER) {
|
|
||||||
setState(READING_PAX_GLOBAL_ATTRIBUTES);
|
|
||||||
currentFileSize = ::strtol(header.size, NULL, 8);
|
|
||||||
bytesRemaining = currentFileSize;
|
|
||||||
paxAttributes.clear();
|
|
||||||
} else if (header.typeflag == PAX_FILE_ATTRIBUTES) {
|
|
||||||
setState(READING_PAX_FILE_ATTRIBUTES);
|
|
||||||
currentFileSize = ::strtol(header.size, NULL, 8);
|
|
||||||
bytesRemaining = currentFileSize;
|
|
||||||
paxAttributes.clear();
|
|
||||||
} else if ((header.typeflag == SYMTYPE) || (header.typeflag == LNKTYPE)) {
|
|
||||||
SG_LOG(SG_IO, SG_WARN, "Tarball contains a link or symlink, will be skipped:" << tarPath);
|
|
||||||
skipCurrentEntry = true;
|
|
||||||
setState(READING_HEADER);
|
|
||||||
} else {
|
|
||||||
SG_LOG(SG_IO, SG_WARN, "Unsupported tar file type:" << header.typeflag);
|
|
||||||
state = BAD_ARCHIVE;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
void processBytes(const char* bytes, size_t count)
|
if (result == Z_STREAM_END) {
|
||||||
{
|
|
||||||
if ((state >= ERROR_STATE) || (state == END_OF_ARCHIVE)) {
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
size_t curBytes = std::min(bytesRemaining, count);
|
|
||||||
if (state == READING_FILE) {
|
|
||||||
if (currentFile) {
|
|
||||||
currentFile->write(bytes, curBytes);
|
|
||||||
}
|
|
||||||
bytesRemaining -= curBytes;
|
|
||||||
} else if ((state == READING_HEADER) || (state == PRE_END_OF_ARCHVE) || (state == END_OF_ARCHIVE)) {
|
|
||||||
memcpy(headerPtr, bytes, curBytes);
|
|
||||||
bytesRemaining -= curBytes;
|
|
||||||
headerPtr += curBytes;
|
|
||||||
} else if (state == READING_PADDING) {
|
|
||||||
bytesRemaining -= curBytes;
|
|
||||||
} else if ((state == READING_PAX_FILE_ATTRIBUTES) || (state == READING_PAX_GLOBAL_ATTRIBUTES)) {
|
|
||||||
bytesRemaining -= curBytes;
|
|
||||||
paxAttributes.append(bytes, curBytes);
|
|
||||||
}
|
|
||||||
|
|
||||||
checkEndOfState();
|
|
||||||
if (count > curBytes) {
|
|
||||||
// recurse with unprocessed bytes
|
|
||||||
processBytes(bytes + curBytes, count - curBytes);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
bool headerIsAllZeros() const
|
|
||||||
{
|
|
||||||
char* headerAsChar = (char*) &header;
|
|
||||||
for (size_t i=0; i < offsetof(UstarHeaderBlock, magic); ++i) {
|
|
||||||
if (*headerAsChar++ != 0) {
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
|
|
||||||
// https://www.ibm.com/support/knowledgecenter/en/SSLTBW_2.3.0/com.ibm.zos.v2r3.bpxa500/paxex.htm#paxex
|
|
||||||
void parsePAXAttributes(bool areGlobal)
|
|
||||||
{
|
|
||||||
auto lineStart = 0;
|
|
||||||
for (;;) {
|
|
||||||
auto firstSpace = paxAttributes.find(' ', lineStart);
|
|
||||||
auto firstEq = paxAttributes.find('=', lineStart);
|
|
||||||
if ((firstEq == std::string::npos) || (firstSpace == std::string::npos)) {
|
|
||||||
SG_LOG(SG_IO, SG_WARN, "Malfroemd PAX attributes in tarfile");
|
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
|
} while ((zlibStream.avail_in > 0) || (writtenSize > 0));
|
||||||
uint32_t lengthBytes = std::stoul(paxAttributes.substr(lineStart, firstSpace));
|
|
||||||
uint32_t dataBytes = lengthBytes - (firstEq + 1);
|
|
||||||
std::string name = paxAttributes.substr(firstSpace+1, firstEq - (firstSpace + 1));
|
|
||||||
|
|
||||||
// dataBytes - 1 here to trim off the trailing newline
|
|
||||||
std::string data = paxAttributes.substr(firstEq+1, dataBytes - 1);
|
|
||||||
|
|
||||||
processPAXAttribute(areGlobal, name, data);
|
|
||||||
|
|
||||||
lineStart += lengthBytes;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
void processPAXAttribute(bool isGlobalAttr, const std::string& attrName, const std::string& data)
|
|
||||||
{
|
|
||||||
if (!isGlobalAttr && (attrName == "path")) {
|
|
||||||
// data is UTF-8 encoded path name
|
|
||||||
paxPathName = data;
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
private:
|
||||||
|
z_stream zlibStream;
|
||||||
|
uint8_t* zlibOutput;
|
||||||
|
bool haveInitedZLib = false;
|
||||||
};
|
};
|
||||||
|
|
||||||
///////////////////////////////////////////////////////////////////////////////
|
///////////////////////////////////////////////////////////////////////////////
|
||||||
|
|
||||||
|
#if 1 || defined(HAVE_XZ)
|
||||||
|
|
||||||
|
#include <lzma.h>
|
||||||
|
|
||||||
|
class XZTarExtractor : public TarExtractorPrivate
|
||||||
|
{
|
||||||
|
public:
|
||||||
|
XZTarExtractor(ArchiveExtractor* outer) : TarExtractorPrivate(outer)
|
||||||
|
{
|
||||||
|
_xzStream = LZMA_STREAM_INIT;
|
||||||
|
_outputBuffer = (uint8_t*)malloc(ZLIB_DECOMPRESS_BUFFER_SIZE);
|
||||||
|
|
||||||
|
|
||||||
|
auto ret = lzma_stream_decoder(&_xzStream, UINT64_MAX, LZMA_TELL_ANY_CHECK);
|
||||||
|
if (ret != LZMA_OK) {
|
||||||
|
setState(BAD_ARCHIVE);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
~XZTarExtractor()
|
||||||
|
{
|
||||||
|
free(_outputBuffer);
|
||||||
|
}
|
||||||
|
|
||||||
|
void extractBytes(const uint8_t* bytes, size_t count) override
|
||||||
|
{
|
||||||
|
lzma_action action = LZMA_RUN;
|
||||||
|
_xzStream.next_in = bytes;
|
||||||
|
_xzStream.avail_in = count;
|
||||||
|
|
||||||
|
size_t writtenSize;
|
||||||
|
do {
|
||||||
|
_xzStream.avail_out = ZLIB_DECOMPRESS_BUFFER_SIZE;
|
||||||
|
_xzStream.next_out = _outputBuffer;
|
||||||
|
|
||||||
|
const auto ret = lzma_code(&_xzStream, action);
|
||||||
|
|
||||||
|
writtenSize = ZLIB_DECOMPRESS_BUFFER_SIZE - _xzStream.avail_out;
|
||||||
|
if (writtenSize > 0) {
|
||||||
|
processBytes((const char*)_outputBuffer, writtenSize);
|
||||||
|
}
|
||||||
|
|
||||||
|
if (ret == LZMA_GET_CHECK) {
|
||||||
|
//
|
||||||
|
} else if (ret == LZMA_STREAM_END) {
|
||||||
|
setState(END_OF_ARCHIVE);
|
||||||
|
break;
|
||||||
|
} else if (ret != LZMA_OK) {
|
||||||
|
setState(BAD_ARCHIVE);
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
} while ((_xzStream.avail_in > 0) || (writtenSize > 0));
|
||||||
|
}
|
||||||
|
|
||||||
|
void flush() override
|
||||||
|
{
|
||||||
|
const auto ret = lzma_code(&_xzStream, LZMA_FINISH);
|
||||||
|
if (ret != LZMA_STREAM_END) {
|
||||||
|
setState(BAD_ARCHIVE);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private:
|
||||||
|
lzma_stream _xzStream;
|
||||||
|
uint8_t* _outputBuffer = nullptr;
|
||||||
|
};
|
||||||
|
|
||||||
|
#endif
|
||||||
|
|
||||||
|
///////////////////////////////////////////////////////////////////////////////
|
||||||
|
|
||||||
extern "C" {
|
extern "C" {
|
||||||
void fill_memory_filefunc(zlib_filefunc_def*);
|
void fill_memory_filefunc(zlib_filefunc_def*);
|
||||||
}
|
}
|
||||||
@@ -637,17 +656,19 @@ void ArchiveExtractor::extractBytes(const uint8_t* bytes, size_t count)
|
|||||||
|
|
||||||
if (r == TarData) {
|
if (r == TarData) {
|
||||||
d.reset(new TarExtractorPrivate(this));
|
d.reset(new TarExtractorPrivate(this));
|
||||||
}
|
} else if (r == GZData) {
|
||||||
else if (r == ZipData) {
|
d.reset(new GZTarExtractor(this));
|
||||||
d.reset(new ZipExtractorPrivate(this));
|
} else if (r == XZData) {
|
||||||
}
|
d.reset(new XZTarExtractor(this));
|
||||||
else {
|
} else if (r == ZipData) {
|
||||||
SG_LOG(SG_IO, SG_WARN, "Invalid archive type");
|
d.reset(new ZipExtractorPrivate(this));
|
||||||
|
} else {
|
||||||
|
SG_LOG(SG_IO, SG_WARN, "Invalid archive type");
|
||||||
_invalidDataType = true;
|
_invalidDataType = true;
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
|
|
||||||
// if hit here, we created the extractor. Feed the prefbuffer
|
// if hit here, we created the extractor. Feed the prefbuffer
|
||||||
// bytes through it
|
// bytes through it
|
||||||
d->extractBytes((uint8_t*) _prebuffer.data(), _prebuffer.size());
|
d->extractBytes((uint8_t*) _prebuffer.data(), _prebuffer.size());
|
||||||
_prebuffer.clear();
|
_prebuffer.clear();
|
||||||
@@ -699,9 +720,18 @@ ArchiveExtractor::DetermineResult ArchiveExtractor::determineType(const uint8_t*
|
|||||||
return ZipData;
|
return ZipData;
|
||||||
}
|
}
|
||||||
|
|
||||||
auto r = isTarData(bytes, count);
|
if (count < 6) {
|
||||||
if ((r == TarData) || (r == InsufficientData))
|
return InsufficientData;
|
||||||
return r;
|
}
|
||||||
|
|
||||||
|
const uint8_t XZ_HEADER[6] = {0xFD, '7', 'z', 'X', 'Z', 0x00};
|
||||||
|
if (memcmp(bytes, XZ_HEADER, 6) == 0) {
|
||||||
|
return XZData;
|
||||||
|
}
|
||||||
|
|
||||||
|
auto r = isTarData(bytes, count);
|
||||||
|
if ((r == TarData) || (r == InsufficientData) || (r == GZData))
|
||||||
|
return r;
|
||||||
|
|
||||||
return Invalid;
|
return Invalid;
|
||||||
}
|
}
|
||||||
@@ -714,6 +744,8 @@ ArchiveExtractor::DetermineResult ArchiveExtractor::isTarData(const uint8_t* byt
|
|||||||
}
|
}
|
||||||
|
|
||||||
UstarHeaderBlock* header = 0;
|
UstarHeaderBlock* header = 0;
|
||||||
|
DetermineResult result = InsufficientData;
|
||||||
|
|
||||||
if ((bytes[0] == 0x1f) && (bytes[1] == 0x8b)) {
|
if ((bytes[0] == 0x1f) && (bytes[1] == 0x8b)) {
|
||||||
// GZIP identification bytes
|
// GZIP identification bytes
|
||||||
z_stream z;
|
z_stream z;
|
||||||
@@ -731,11 +763,11 @@ ArchiveExtractor::DetermineResult ArchiveExtractor::isTarData(const uint8_t* byt
|
|||||||
return Invalid;
|
return Invalid;
|
||||||
}
|
}
|
||||||
|
|
||||||
int result = inflate(&z, Z_SYNC_FLUSH);
|
int zResult = inflate(&z, Z_SYNC_FLUSH);
|
||||||
if ((result == Z_OK) || (result == Z_STREAM_END)) {
|
if ((zResult == Z_OK) || (zResult == Z_STREAM_END)) {
|
||||||
// all good
|
// all good
|
||||||
} else {
|
} else {
|
||||||
SG_LOG(SG_IO, SG_WARN, "isTarData: Zlib inflate failed:" << result);
|
SG_LOG(SG_IO, SG_WARN, "isTarData: Zlib inflate failed:" << zResult);
|
||||||
inflateEnd(&z);
|
inflateEnd(&z);
|
||||||
return Invalid; // not tar data
|
return Invalid; // not tar data
|
||||||
}
|
}
|
||||||
@@ -748,6 +780,7 @@ ArchiveExtractor::DetermineResult ArchiveExtractor::isTarData(const uint8_t* byt
|
|||||||
|
|
||||||
header = reinterpret_cast<UstarHeaderBlock*>(zlibOutput);
|
header = reinterpret_cast<UstarHeaderBlock*>(zlibOutput);
|
||||||
inflateEnd(&z);
|
inflateEnd(&z);
|
||||||
|
result = GZData;
|
||||||
} else {
|
} else {
|
||||||
// uncompressed tar
|
// uncompressed tar
|
||||||
if (count < TAR_HEADER_BLOCK_SIZE) {
|
if (count < TAR_HEADER_BLOCK_SIZE) {
|
||||||
@@ -755,13 +788,14 @@ ArchiveExtractor::DetermineResult ArchiveExtractor::isTarData(const uint8_t* byt
|
|||||||
}
|
}
|
||||||
|
|
||||||
header = (UstarHeaderBlock*) bytes;
|
header = (UstarHeaderBlock*) bytes;
|
||||||
|
result = TarData;
|
||||||
}
|
}
|
||||||
|
|
||||||
if (strncmp(header->magic, TMAGIC, TMAGLEN) != 0) {
|
if (strncmp(header->magic, TMAGIC, TMAGLEN) != 0) {
|
||||||
return Invalid;
|
return Invalid;
|
||||||
}
|
}
|
||||||
|
|
||||||
return TarData;
|
return result;
|
||||||
}
|
}
|
||||||
|
|
||||||
void ArchiveExtractor::extractLocalFile(const SGPath& archiveFile)
|
void ArchiveExtractor::extractLocalFile(const SGPath& archiveFile)
|
||||||
|
|||||||
@@ -36,15 +36,16 @@ public:
|
|||||||
ArchiveExtractor(const SGPath& rootPath);
|
ArchiveExtractor(const SGPath& rootPath);
|
||||||
virtual ~ArchiveExtractor();
|
virtual ~ArchiveExtractor();
|
||||||
|
|
||||||
enum DetermineResult
|
enum DetermineResult {
|
||||||
{
|
Invalid,
|
||||||
Invalid,
|
InsufficientData,
|
||||||
InsufficientData,
|
TarData,
|
||||||
TarData,
|
ZipData,
|
||||||
ZipData
|
GZData, // Gzipped-tar
|
||||||
};
|
XZData // XZ (aka LZMA) tar
|
||||||
|
};
|
||||||
|
|
||||||
static DetermineResult determineType(const uint8_t* bytes, size_t count);
|
static DetermineResult determineType(const uint8_t* bytes, size_t count);
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* @brief API to extract a local zip or tar.gz
|
* @brief API to extract a local zip or tar.gz
|
||||||
|
|||||||
Reference in New Issue
Block a user