diff --git a/src/sevenz_extract.c b/src/sevenz_extract.c new file mode 100644 index 0000000..a387a32 --- /dev/null +++ b/src/sevenz_extract.c @@ -0,0 +1,1738 @@ +/* Safe 7z extraction engine used by the /api/extract task. + + Input arrives through src/sevenz_volstream.c, so a single `name.7z` and a + byte-split set (`name.7z.001`, `.002`, ...) take the same path: the SDK's + header reader sees one continuous stream either way. + + Folders are decoded by src/sevenz_chain.c, which parses the folder + descriptor itself and drives the coder chain in a pull pipeline. That is + what makes solid 7z archives usable on a console: the SDK's own + SzArEx_Extract() wants the whole folder in memory, and a solid block is + routinely gigabytes. + + The publish / staging / rollback / name-validation machinery is mirrored + from rar_extract.c, which in turn mirrors zip_extract.c, so that any + consumer of zipx_result_t gets the same error and progress contract + regardless of archive format. */ + +#include "sevenz_extract.h" + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "7z.h" +#include "7zAlloc.h" +#include "7zCrc.h" +#include "7zFile.h" + +#include "sevenz_chain.h" +#include "sevenz_volstream.h" + +#ifndef O_CLOEXEC +#define O_CLOEXEC 0 +#endif + +#define SZX_STAGING_PREFIX ".wfm-extract-" +#define SZX_TMP_ATTEMPTS 64 +#define SZX_PUBLISH_MAX_DEPTH 128 +#define SZX_SPACE_SLACK_PER_ENTRY 512 +/* 256 KiB: big enough that a solid folder is not refilled constantly, small + enough that a hostile dictionary size cannot turn into an allocation storm. */ +#define SZX_INPUT_BUF_SIZE (1u << 18) +/* Room for a name plus room to notice that it did not fit. */ +#define SZX_NAME_MAX (ZIPX_PATH_MAX + 16) + +/* 7-Zip's attribute word: bit 15 says "the high half is a Unix mode", bit 4 is + the DOS directory flag. Both come from PropID::kWinAttrib. */ +#define SZX_ATTRIB_UNIX_EXTENSION 0x8000u +#define SZX_ATTRIB_DIRECTORY 0x0010u + +/* Windows FILETIME counts 100 ns ticks from 1601-01-01; Unix time starts at + 1970. The offset is the standard 11644473600 seconds. */ +#define SZX_FILETIME_UNIX_DELTA 11644473600ULL + +static ISzAlloc g_sz_alloc = { SzAlloc, SzFree }; + +/************************************************************************** + * context + small helpers (mirrors of rarx_*; names prefixed szx_) + **************************************************************************/ + +typedef struct { + zipx_conflict_t conflict; + zipx_limits_t limits; + zipx_cancel_fn cancel; + zipx_progress_fn progress; + void *userdata; + zipx_result_t *result; + const char *password; + uint64_t entries_total; + uint64_t entries_done; + uint64_t bytes_total; + uint64_t bytes_done; + uint64_t files_created; + uint64_t dirs_created; + uint64_t progress_floor; + long long last_report_ms; + char staging[ZIPX_PATH_MAX]; + char **created; + size_t created_count; + size_t created_cap; + unsigned char *created_is_dir; + int staging_created; +} szx_ctx_t; + +static void +szx_set_detail(szx_ctx_t *c, const char *path) { + if(path) { + snprintf(c->result->detail, sizeof(c->result->detail), "%s", path); + } +} + +static int +szx_fail(szx_ctx_t *c, zipx_status_t status, const char *detail, + const char *fmt, ...) { + va_list ap; + + c->result->status = status; + c->result->sys_errno = errno; + szx_set_detail(c, detail); + va_start(ap, fmt); + if(fmt) { + vsnprintf(c->result->message, sizeof(c->result->message), fmt, ap); + } + va_end(ap); + if(!c->result->message[0]) { + snprintf(c->result->message, sizeof(c->result->message), "%s", + zipx_status_string(status)); + } + return (int)status; +} + +static long long +szx_mono_ms(void) { + struct timespec ts; + + clock_gettime(CLOCK_MONOTONIC, &ts); + return (long long)ts.tv_sec * 1000 + ts.tv_nsec / 1000000; +} + +/* The decoder calls this from its innermost loop, so a report goes out only + every 200 ms, or every 1 MiB when the clock has not moved. */ +static void +szx_report(szx_ctx_t *c, int phase, const char *current, int force) { + zipx_progress_t p; + long long now; + + if(!c->progress) { + return; + } + now = szx_mono_ms(); + if(!force && c->last_report_ms && now - c->last_report_ms < 200 && + c->bytes_done - c->progress_floor < (1024 * 1024)) { + return; + } + c->last_report_ms = now; + c->progress_floor = c->bytes_done; + + p.phase = phase; + p.entries_total = c->entries_total; + p.entries_done = c->entries_done; + p.bytes_total = c->bytes_total; + p.bytes_done = c->bytes_done; + p.current = current; + c->progress(c->userdata, &p); +} + +static int +szx_canceled(szx_ctx_t *c) { + return c->cancel && c->cancel(c->userdata); +} + +/* Remembers what the publish phase put in dst_dir, so a later failure can undo + it. Paths are recorded in publish order; rollback walks them backwards. */ +static int +szx_remember_published(szx_ctx_t *c, const char *path, int is_dir) { + if(c->created_count == c->created_cap) { + size_t cap = c->created_cap ? c->created_cap * 2 : 64; + char **paths = realloc(c->created, cap * sizeof(*paths)); + unsigned char *flags = realloc(c->created_is_dir, cap * sizeof(*flags)); + + if(!paths || !flags) { + free(paths); + free(flags); + return -1; + } + c->created = paths; + c->created_is_dir = flags; + c->created_cap = cap; + } + if(!(c->created[c->created_count] = strdup(path))) { + return -1; + } + c->created_is_dir[c->created_count] = is_dir ? 1 : 0; + c->created_count++; + return 0; +} + +static void +szx_free_published(szx_ctx_t *c) { + size_t i; + + for(i = 0; i < c->created_count; i++) { + free(c->created[i]); + } + free(c->created); + free(c->created_is_dir); + c->created = NULL; + c->created_is_dir = NULL; + c->created_count = c->created_cap = 0; +} + +/************************************************************************** + * duplicate detection (FNV-1a hash set; same trick as zip_extract.c) + **************************************************************************/ + +typedef struct { + uint64_t *hash; + unsigned char *is_dir; + size_t cap; + size_t count; +} szx_nameset_t; + +static uint64_t +szx_fnv1a(const char *s, size_t len) { + uint64_t h = 1469598103934665603ULL; + size_t i; + + for(i = 0; i < len; i++) { + h ^= (unsigned char)s[i]; + h *= 1099511628211ULL; + } + return h ? h : 1; +} + +static int +szx_nameset_init(szx_nameset_t *set, size_t hint) { + size_t cap = 1024; + + while(cap < hint * 2) { + cap *= 2; + } + set->hash = calloc(cap, sizeof(*set->hash)); + set->is_dir = calloc(cap, sizeof(*set->is_dir)); + if(!set->hash || !set->is_dir) { + free(set->hash); + free(set->is_dir); + set->hash = NULL; + set->is_dir = NULL; + return -1; + } + set->cap = cap; + set->count = 0; + return 0; +} + +static void +szx_nameset_free(szx_nameset_t *set) { + free(set->hash); + free(set->is_dir); + set->hash = NULL; + set->is_dir = NULL; + set->cap = set->count = 0; +} + +static size_t +szx_nameset_slot(szx_nameset_t *set, uint64_t h) { + size_t mask = set->cap - 1; + size_t i = (size_t)(h & mask); + + while(set->hash[i] && set->hash[i] != h) { + i = (i + 1) & mask; + } + return i; +} + +static int +szx_nameset_grow(szx_nameset_t *set) { + szx_nameset_t bigger; + size_t i; + + if(szx_nameset_init(&bigger, set->cap)) { + return -1; + } + for(i = 0; i < set->cap; i++) { + size_t slot; + + if(!set->hash[i]) { + continue; + } + slot = szx_nameset_slot(&bigger, set->hash[i]); + bigger.hash[slot] = set->hash[i]; + bigger.is_dir[slot] = set->is_dir[i]; + bigger.count++; + } + szx_nameset_free(set); + *set = bigger; + return 0; +} + +static int +szx_nameset_get(szx_nameset_t *set, uint64_t h) { + size_t i = szx_nameset_slot(set, h); + + if(!set->hash[i]) { + return 0; + } + return set->is_dir[i] ? 1 : 2; +} + +static int +szx_nameset_put(szx_nameset_t *set, uint64_t h, int is_dir) { + size_t i; + + if(set->count * 4 >= set->cap * 3 && szx_nameset_grow(set)) { + return -1; + } + i = szx_nameset_slot(set, h); + if(!set->hash[i]) { + set->hash[i] = h; + set->count++; + } + set->is_dir[i] = is_dir ? 1 : 0; + return 0; +} + +/* Registers `name` plus every directory on the way to it, and refuses a name + that reuses an existing file as a directory. */ +static int +szx_nameset_add_path(szx_nameset_t *set, const char *name, int is_dir, + szx_ctx_t *c) { + char buf[ZIPX_PATH_MAX]; + size_t len = strlen(name); + size_t i; + + if(len >= sizeof(buf)) { + return szx_fail(c, ZIPX_ERR_LIMIT_NAME, name, "entry name is too long"); + } + memcpy(buf, name, len + 1); + + for(i = 0; i < len; i++) { + if(buf[i] != '/') { + continue; + } + buf[i] = 0; + if(szx_nameset_get(set, szx_fnv1a(buf, i)) == 2) { + return szx_fail(c, ZIPX_ERR_DUPLICATE, name, + "entry uses a file as a directory"); + } + if(szx_nameset_put(set, szx_fnv1a(buf, i), 1)) { + return szx_fail(c, ZIPX_ERR_INTERNAL, name, "out of memory"); + } + buf[i] = '/'; + } + + if(szx_nameset_put(set, szx_fnv1a(name, len), is_dir)) { + return szx_fail(c, ZIPX_ERR_INTERNAL, name, "out of memory"); + } + return 0; +} + +static int +szx_nameset_check_duplicate(szx_nameset_t *set, const char *name, int is_dir, + szx_ctx_t *c) { + int existing = szx_nameset_get(set, szx_fnv1a(name, strlen(name))); + + if(existing == 2 || (existing == 1 && !is_dir)) { + return szx_fail(c, ZIPX_ERR_DUPLICATE, name, "duplicate entry name"); + } + return 0; +} + +/************************************************************************** + * entry name validation (mirrors zip_extract.c::normalize_name; 7z stores + * names as UTF-16, so the caller converts first) + **************************************************************************/ + +/* Rewrites an archive name into a safe relative path. A name that would + escape the destination, hide a control character, or nest past the depth + limit is refused with a message that names it. */ +static int +szx_normalize_name(const char *in, char *out, size_t out_size, uint32_t *depth, + szx_ctx_t *c) { + size_t in_len = strlen(in); + size_t out_len = 0; + size_t i = 0; + size_t seg_len = 0; + uint32_t levels = 0; + + *depth = 0; + if(!in_len) { + return szx_fail(c, ZIPX_ERR_UNSAFE_NAME, in, "empty entry name"); + } + if(in_len > c->limits.max_path_len) { + return szx_fail(c, ZIPX_ERR_LIMIT_NAME, in, "entry name is too long"); + } + if(in[0] == '/' || in[0] == '\\') { + return szx_fail(c, ZIPX_ERR_UNSAFE_NAME, in, "absolute entry name"); + } + if(in_len >= 2 && ((in[0] >= 'A' && in[0] <= 'Z') || + (in[0] >= 'a' && in[0] <= 'z')) && in[1] == ':') { + return szx_fail(c, ZIPX_ERR_UNSAFE_NAME, in, "drive letter in entry name"); + } + + while(i < in_len) { + char ch = in[i]; + + if((unsigned char)ch < 0x20 || (unsigned char)ch == 0x7f) { + return szx_fail(c, ZIPX_ERR_UNSAFE_NAME, in, + "control character in entry name"); + } + /* 7-Zip writes '/', but a name added from a Windows path can still carry + a backslash. Normalize so the staging tree is consistent. */ + if(ch == '\\') { + ch = '/'; + } + if(ch == '/') { + if(!seg_len) { + return szx_fail(c, ZIPX_ERR_UNSAFE_NAME, in, "empty path segment"); + } + if(seg_len == 1 && out[out_len - 1] == '.') { + return szx_fail(c, ZIPX_ERR_UNSAFE_NAME, in, "'.' path segment"); + } + if(seg_len == 2 && !memcmp(out + out_len - 2, "..", 2)) { + return szx_fail(c, ZIPX_ERR_UNSAFE_NAME, in, "'..' path segment"); + } + if(seg_len > c->limits.max_name_len) { + return szx_fail(c, ZIPX_ERR_LIMIT_NAME, in, + "path component is too long"); + } + out[out_len++] = '/'; + levels++; + seg_len = 0; + i++; + continue; + } + if(out_len + 2 >= out_size) { + return szx_fail(c, ZIPX_ERR_LIMIT_NAME, in, "entry name is too long"); + } + out[out_len++] = ch; + seg_len++; + i++; + } + + if(!seg_len) { + return szx_fail(c, ZIPX_ERR_UNSAFE_NAME, in, "empty entry name"); + } + if(seg_len == 1 && out[out_len - 1] == '.') { + return szx_fail(c, ZIPX_ERR_UNSAFE_NAME, in, "'.' path segment"); + } + if(seg_len == 2 && !memcmp(out + out_len - 2, "..", 2)) { + return szx_fail(c, ZIPX_ERR_UNSAFE_NAME, in, "'..' path segment"); + } + if(seg_len > c->limits.max_name_len) { + return szx_fail(c, ZIPX_ERR_LIMIT_NAME, in, "path component is too long"); + } + out[out_len] = 0; + levels++; + + if(levels > c->limits.max_depth) { + return szx_fail(c, ZIPX_ERR_LIMIT_DEPTH, in, "entry path is too deep"); + } + *depth = levels; + return 0; +} + +/************************************************************************** + * staging helpers (same as zip_extract.c / rar_extract.c) + **************************************************************************/ + +static int +szx_path_parent(const char *path, char *out, size_t out_size) { + size_t len = strlen(path); + size_t i; + + if(!len) { + return -1; + } + i = len; + while(i && path[i - 1] == '/') { + i--; + } + while(i && path[i - 1] != '/') { + i--; + } + while(i > 1 && path[i - 1] == '/') { + i--; + } + if(!i || i >= out_size) { + return -1; + } + memcpy(out, path, i); + out[i] = 0; + return 0; +} + +static int +szx_make_staging(szx_ctx_t *c, const char *parent) { + unsigned int attempt; + + for(attempt = 0; attempt < SZX_TMP_ATTEMPTS; attempt++) { + int n = snprintf(c->staging, sizeof(c->staging), "%s/%s%ld-%lld-%u", + parent, SZX_STAGING_PREFIX, (long)getpid(), + (long long)time(NULL), attempt); + + if(n < 0 || (size_t)n >= sizeof(c->staging)) { + return szx_fail(c, ZIPX_ERR_INTERNAL, NULL, "staging path is too long"); + } + if(!mkdir(c->staging, 0777)) { + c->staging_created = 1; + return 0; + } + if(errno != EEXIST) { + return szx_fail(c, ZIPX_ERR_IO, c->staging, "cannot create staging: %s", + strerror(errno)); + } + } + return szx_fail(c, ZIPX_ERR_IO, parent, "cannot create staging directory"); +} + +static void +szx_remove_tree(const char *path) { + DIR *dir = opendir(path); + struct dirent *ent; + + if(!dir) { + rmdir(path); + return; + } + while((ent = readdir(dir))) { + char child[ZIPX_PATH_MAX]; + struct stat st; + + if(!strcmp(ent->d_name, ".") || !strcmp(ent->d_name, "..")) { + continue; + } + if(snprintf(child, sizeof(child), "%s/%s", path, ent->d_name) >= + (int)sizeof(child)) { + continue; + } + if(!lstat(child, &st) && S_ISDIR(st.st_mode)) { + szx_remove_tree(child); + } else { + unlink(child); + } + } + closedir(dir); + rmdir(path); +} + +static void +szx_cleanup_staging(szx_ctx_t *c) { + if(c->staging_created) { + szx_remove_tree(c->staging); + c->staging_created = 0; + } +} + +static void +szx_rollback_published(szx_ctx_t *c) { + size_t i = c->created_count; + + while(i) { + i--; + if(c->created_is_dir[i]) { + rmdir(c->created[i]); + } else { + unlink(c->created[i]); + } + } +} + +/* Creates every directory between the staging root and `rel`, but not rel + itself. Names were validated during scan, so no segment can escape. */ +static int +szx_make_dirs(szx_ctx_t *c, const char *rel) { + char buf[ZIPX_PATH_MAX]; + char full[ZIPX_PATH_MAX]; + size_t i; + + if(snprintf(buf, sizeof(buf), "%s", rel) >= (int)sizeof(buf)) { + return szx_fail(c, ZIPX_ERR_LIMIT_NAME, rel, "path is too long"); + } + for(i = 0; buf[i]; i++) { + char saved; + + if(buf[i] != '/') { + continue; + } + saved = buf[i]; + buf[i] = 0; + if(snprintf(full, sizeof(full), "%s/%s", c->staging, buf) < + (int)sizeof(full) && + mkdir(full, 0777) && errno != EEXIST) { + return szx_fail(c, ZIPX_ERR_IO, full, "cannot create directory: %s", + strerror(errno)); + } + buf[i] = saved; + } + return 0; +} + +static int +szx_check_space(szx_ctx_t *c, const char *target) { + struct statvfs vfs; + unsigned long long available; + unsigned long long block; + unsigned long long required = c->bytes_total + + c->entries_total * SZX_SPACE_SLACK_PER_ENTRY; + + if(statvfs(target, &vfs)) { + return szx_fail(c, ZIPX_ERR_SPACE, target, "cannot read free space: %s", + strerror(errno)); + } + block = vfs.f_frsize ? vfs.f_frsize : vfs.f_bsize; + available = (unsigned long long)vfs.f_bavail * block; + if(available < required) { + return szx_fail(c, ZIPX_ERR_SPACE, target, + "not enough space, required %llu bytes, available %llu bytes", + required, available); + } + return 0; +} + +/************************************************************************** + * 7z entry inspection + **************************************************************************/ + +/* UTF-16LE (the 7z name table encoding) to UTF-8. Returns -1 when the result + would not fit, so the caller can refuse the entry instead of writing a name + that does not match the archive. Unpaired surrogates become U+FFFD: a + mangled name is still a name, and the entry may be perfectly good. */ +static int +szx_utf16_to_utf8(const UInt16 *src, char *dst, size_t dst_size) { + size_t out = 0; + + while(*src) { + UInt32 c = *src++; + + if(c >= 0xD800 && c <= 0xDBFF) { + if(*src >= 0xDC00 && *src <= 0xDFFF) { + c = 0x10000 + ((c - 0xD800) << 10) + (*src++ - 0xDC00); + } else { + c = 0xFFFD; + } + } else if(c >= 0xDC00 && c <= 0xDFFF) { + c = 0xFFFD; + } + if(c < 0x80) { + if(out + 1 >= dst_size) { + return -1; + } + dst[out++] = (char)c; + } else if(c < 0x800) { + if(out + 2 >= dst_size) { + return -1; + } + dst[out++] = (char)(0xC0 | (c >> 6)); + dst[out++] = (char)(0x80 | (c & 0x3F)); + } else if(c < 0x10000) { + if(out + 3 >= dst_size) { + return -1; + } + dst[out++] = (char)(0xE0 | (c >> 12)); + dst[out++] = (char)(0x80 | ((c >> 6) & 0x3F)); + dst[out++] = (char)(0x80 | (c & 0x3F)); + } else { + if(out + 4 >= dst_size) { + return -1; + } + dst[out++] = (char)(0xF0 | (c >> 18)); + dst[out++] = (char)(0x80 | ((c >> 12) & 0x3F)); + dst[out++] = (char)(0x80 | ((c >> 6) & 0x3F)); + dst[out++] = (char)(0x80 | (c & 0x3F)); + } + } + dst[out] = 0; + return 0; +} + +/* Converts entry `index`'s name into `out`. Returns non-zero with the message + already set when the name is unusable. */ +static int +szx_entry_name(const CSzArEx *db, UInt32 index, char *out, size_t out_size, + szx_ctx_t *c) { + size_t units = SzArEx_GetFileNameUtf16(db, index, NULL); + UInt16 *wide; + + if(units == 0) { + return szx_fail(c, ZIPX_ERR_FORMAT, NULL, "entry %u has an empty name", + (unsigned)index); + } + wide = (UInt16 *)malloc((units + 1) * sizeof(UInt16)); + if(!wide) { + return szx_fail(c, ZIPX_ERR_INTERNAL, NULL, "out of memory"); + } + SzArEx_GetFileNameUtf16(db, index, wide); + if(szx_utf16_to_utf8(wide, out, out_size) == 0 && out[0]) { + free(wide); + return 0; + } + free(wide); + return szx_fail(c, ZIPX_ERR_LIMIT_NAME, NULL, + "entry %u has a name that is empty or too long", + (unsigned)index); +} + +/* The 7z attribute word: DOS bits in the low half, Unix mode in the high half + when bit 15 is set. Returns 0 when the archive stores no attributes. */ +static UInt32 +szx_entry_attrib(const CSzArEx *db, UInt32 index) { + if(!SzBitWithVals_Check(&db->Attribs, index)) { + return 0; + } + return db->Attribs.Vals[index]; +} + +static UInt32 +szx_unix_mode(UInt32 attrib) { + return (attrib & SZX_ATTRIB_UNIX_EXTENSION) ? (attrib >> 16) : 0; +} + +static int +szx_entry_is_dir(const CSzArEx *db, UInt32 index, UInt32 attrib) { + UInt32 mode; + + if(SzArEx_IsDir(db, index)) { + return 1; + } + if(attrib & SZX_ATTRIB_DIRECTORY) { + return 1; + } + mode = szx_unix_mode(attrib); + return (mode && (mode & S_IFMT) == S_IFDIR) ? 1 : 0; +} + +static uint64_t +szx_entry_size(const CSzArEx *db, UInt32 index) { + return (uint64_t)(db->UnpackPositions[(size_t)index + 1] - + db->UnpackPositions[index]); +} + +/* -1 when the archive stores no time for this entry. */ +static time_t +szx_entry_mtime(const CSzArEx *db, UInt32 index) { + UInt64 ft; + + if(!SzBitWithVals_Check(&db->MTime, index)) { + return (time_t)-1; + } + ft = ((UInt64)db->MTime.Vals[index].High << 32) | db->MTime.Vals[index].Low; + if(ft < SZX_FILETIME_UNIX_DELTA * 10000000ULL) { + return (time_t)-1; + } + return (time_t)(ft / 10000000ULL - SZX_FILETIME_UNIX_DELTA); +} + +/* Applies the entry's Unix mode and modification time to a staged object. + Both are best effort: a filesystem that refuses them must not fail an + otherwise good extraction. */ +static void +szx_apply_metadata(szx_ctx_t *c, const char *rel, UInt32 attrib, time_t mtime) { + char full[ZIPX_PATH_MAX]; + UInt32 mode = szx_unix_mode(attrib); + + if(snprintf(full, sizeof(full), "%s/%s", c->staging, rel) >= + (int)sizeof(full)) { + return; + } + if(mode & 07777) { + (void)chmod(full, (mode_t)(mode & 07777)); + } + if(mtime != (time_t)-1) { + struct timeval tv[2]; + + tv[0].tv_sec = mtime; + tv[0].tv_usec = 0; + tv[1] = tv[0]; + (void)utimes(full, tv); + } +} + +/************************************************************************** + * scan phase + **************************************************************************/ + +/* A first pass over the whole archive that writes nothing: it validates every + name, rejects the entry types and limits we cannot honour, and totals the + bytes so the free-space check happens before a single file is created. + Anything that is going to fail should fail here, not three gigabytes in. */ +static int +scan_entries(const CSzArEx *db, szx_ctx_t *c) { + szx_nameset_t set; + UInt32 i; + int ret = 0; + + if(szx_nameset_init(&set, db->NumFiles ? db->NumFiles : 16)) { + return szx_fail(c, ZIPX_ERR_INTERNAL, NULL, "out of memory"); + } + + for(i = 0; i < db->NumFiles; i++) { + char raw[SZX_NAME_MAX]; + char name[ZIPX_PATH_MAX]; + UInt32 attrib; + UInt32 mode; + uint64_t size; + uint32_t depth = 0; + int is_dir; + int rc; + + if(szx_entry_name(db, i, raw, sizeof(raw), c)) { + ret = (int)c->result->status; + break; + } + attrib = szx_entry_attrib(db, i); + mode = szx_unix_mode(attrib); + is_dir = szx_entry_is_dir(db, i, attrib); + size = szx_entry_size(db, i); + + rc = szx_normalize_name(raw, name, sizeof(name), &depth, c); + (void)depth; + if(rc) { + ret = rc; + break; + } + + /* 7-Zip records a symbolic link as a file whose bytes are the target + path, with S_IFLNK in the attribute word. Honouring one would need + link(), which is not usable for a plain unzip target on the PS5 + filesystem; dropping the type would silently turn the link into a + regular file holding a path. Refuse it by name instead. */ + if((mode & S_IFMT) == S_IFLNK) { + ret = szx_fail(c, ZIPX_ERR_SPECIAL, name, "symbolic link entry"); + break; + } + if((mode & S_IFMT) && (mode & S_IFMT) != S_IFREG && + (mode & S_IFMT) != S_IFDIR) { + ret = szx_fail(c, ZIPX_ERR_SPECIAL, name, "special file entry"); + break; + } + + if(szx_nameset_check_duplicate(&set, name, is_dir, c) || + szx_nameset_add_path(&set, name, is_dir, c)) { + ret = (int)c->result->status; + break; + } + + if(!is_dir && size) { + if(size > c->limits.max_file_bytes) { + ret = szx_fail(c, ZIPX_ERR_LIMIT_FILE, name, + "entry is larger than %llu bytes", + (unsigned long long)c->limits.max_file_bytes); + break; + } + if(size >= c->limits.max_total_bytes || + c->bytes_total > c->limits.max_total_bytes - size) { + ret = szx_fail(c, ZIPX_ERR_LIMIT_TOTAL, name, + "archive contents are larger than %llu bytes", + (unsigned long long)c->limits.max_total_bytes); + break; + } + c->bytes_total += size; + } + + c->entries_total++; + if(c->entries_total > c->limits.max_entries) { + ret = szx_fail(c, ZIPX_ERR_LIMIT_ENTRIES, name, + "archive has more than %llu entries", + (unsigned long long)c->limits.max_entries); + break; + } + if(c->entries_total % 4096 == 0) { + if(szx_canceled(c)) { + ret = szx_fail(c, ZIPX_ERR_CANCELED, name, NULL); + break; + } + szx_report(c, ZIPX_PHASE_SCAN, name, 0); + } + } + + szx_nameset_free(&set); + return ret; +} + +/* Fills `pack_positions` for a folder, after checking that the folder does not + use more packed streams than the chain builder can hold. */ +static int +szx_folder_packs(const CSzArEx *db, UInt32 folder, uint64_t *pack_positions, + UInt32 *pack_count, szx_ctx_t *c) { + const UInt32 first = db->db.FoStartPackStreamIndex[folder]; + const UInt32 count = db->db.FoStartPackStreamIndex[(size_t)folder + 1] - first; + UInt32 k; + + if(count > SZ_CHAIN_MAX_STREAMS) { + return szx_fail(c, ZIPX_ERR_UNSUPPORTED, NULL, + "folder %u uses %u packed streams, at most %d are " + "supported", + (unsigned)folder, (unsigned)count, SZ_CHAIN_MAX_STREAMS); + } + for(k = 0; k <= count; k++) { + pack_positions[k] = db->db.PackPositions[first + k]; + } + *pack_count = count; + return 0; +} + +/* Walk every folder and parse its coder chain once, without decoding: + * a coder graph we cannot drive is reported by folder and by chain name, + before any output exists; + * a folder that needs a password is reported the same way, so a protected + archive does not first waste a whole extraction pass. + The chains are freed again; the extract phase re-parses them per folder, so + an archive with a million folders never holds more than one at a time. */ +static int +precheck_folders(const CSzArEx *db, szx_ctx_t *c) { + UInt32 folder; + + for(folder = 0; folder < db->db.NumFolders; folder++) { + uint64_t pack_positions[SZ_CHAIN_MAX_STREAMS + 1]; + UInt32 pack_count = 0; + const uint8_t *blob = db->db.CodersData + db->db.FoCodersOffsets[folder]; + size_t blob_size = db->db.FoCodersOffsets[(size_t)folder + 1] - + db->db.FoCodersOffsets[folder]; + const uint64_t *cus = + &db->db.CoderUnpackSizes[db->db.FoToCoderUnpackSizes[folder]]; + uint64_t unpack_size = SzAr_GetFolderUnpackSize(&db->db, folder); + sz_chain *chain = NULL; + sz_chain_err_t cerr; + char desc[256]; + + if(szx_folder_packs(db, folder, pack_positions, &pack_count, c)) { + return (int)c->result->status; + } + if(sz_chain_parse(&chain, blob, blob_size, pack_positions, pack_count, cus, + unpack_size, sz_chain_default_limits(), &cerr) != 0) { + return szx_fail(c, + (cerr.status == SZ_CHAIN_ERR_METHOD || + cerr.status == SZ_CHAIN_ERR_LAYOUT) + ? ZIPX_ERR_UNSUPPORTED + : ZIPX_ERR_FORMAT, + NULL, "folder %u: %s", (unsigned)folder, cerr.message); + } + sz_chain_describe(chain, desc, sizeof(desc)); + + if(sz_chain_needs_password(chain) && (!c->password || !c->password[0])) { + sz_chain_free(chain); + return szx_fail(c, ZIPX_ERR_PASSWORD, NULL, + "the archive is encrypted: folder %u holds %s and needs " + "a password", + (unsigned)folder, desc); + } + sz_chain_free(chain); + } + return 0; +} + +/* A solid folder's ratio cannot be judged entry by entry: the packed stream + belongs to all of them. Comparing the folder's declared output against the + bytes it actually occupies is the honest screen, and it is the one that + catches a decompression bomb before the disk fills. */ +static int +check_folder_ratio(const CSzArEx *db, UInt32 folder, szx_ctx_t *c) { + uint64_t unpacked = SzAr_GetFolderUnpackSize(&db->db, folder); + const UInt32 first = db->db.FoStartPackStreamIndex[folder]; + const UInt32 last = db->db.FoStartPackStreamIndex[(size_t)folder + 1]; + uint64_t packed; + + if(!c->limits.max_ratio || !unpacked || + unpacked < c->limits.ratio_min_bytes) { + return 0; + } + /* The vendored decoder subset does not include SzArEx_GetFolderFullPackSize, + and the folder's packed extent is a plain difference of PackPositions + anyway. */ + packed = (uint64_t)(db->db.PackPositions[last] - db->db.PackPositions[first]); + if(!packed) { + return 0; + } + if(unpacked > packed * (uint64_t)c->limits.max_ratio) { + return szx_fail(c, ZIPX_ERR_LIMIT_RATIO, NULL, + "folder %u expands %llu bytes into %llu, above the %u:1 " + "ratio limit", + (unsigned)folder, (unsigned long long)packed, + (unsigned long long)unpacked, c->limits.max_ratio); + } + return 0; +} + +/************************************************************************** + * extract phase + **************************************************************************/ + +typedef struct { + UInt32 file_index; + uint64_t size; +} szx_plan_slot_t; + +typedef struct { + szx_ctx_t *c; + const CSzArEx *db; + + /* The ordered entries of the folder being decoded, plus how far into them + the byte stream has reached. Exactly one folder is live at a time. */ + szx_plan_slot_t *plan; + size_t plan_len; + size_t plan_pos; + + uint64_t written; + int fd; + char cur_name[ZIPX_PATH_MAX]; + uint32_t entry_crc; +} szx_sink_t; + +/* Reads `size` bytes of packed data at `offset`, which is counted from the + first packed byte of the archive (db.dataPos). */ +typedef struct { + ISeekInStream *stream; + UInt64 base; +} szx_reader_t; + +static int +szx_read_at(void *ctx, uint64_t offset, void *dst, size_t size) { + szx_reader_t *r = (szx_reader_t *)ctx; + Int64 pos = (Int64)(r->base + offset); + size_t done = 0; + + if(r->stream->Seek(r->stream, &pos, SZ_SEEK_SET) != SZ_OK) { + return -1; + } + while(done < size) { + size_t want = size - done; + + if(r->stream->Read(r->stream, (Byte *)dst + done, &want) != SZ_OK) { + return -1; + } + if(want == 0) { + return -1; + } + done += want; + } + return 0; +} + +static void +szx_plan_free(szx_sink_t *s) { + free(s->plan); + s->plan = NULL; + s->plan_len = s->plan_pos = 0; +} + +/* Collects the non-empty entries of `folder` in stream order. */ +static int +szx_plan_build(szx_sink_t *s, UInt32 folder) { + const CSzArEx *db = s->db; + UInt32 first = db->FolderToFile[folder]; + UInt32 last = db->FolderToFile[(size_t)folder + 1]; + UInt32 i; + + szx_plan_free(s); + /* A folder abandoned mid-entry leaves `written` pointing into a file it + never finished; carrying that into the next folder would make every later + entry look partly written. */ + s->written = 0; + if(last <= first) { + return 0; + } + s->plan = (szx_plan_slot_t *)malloc(sizeof(szx_plan_slot_t) * + (size_t)(last - first)); + if(!s->plan) { + return szx_fail(s->c, ZIPX_ERR_INTERNAL, NULL, "out of memory"); + } + + for(i = first; i < last; i++) { + uint64_t size = szx_entry_size(db, i); + + if(db->FileToFolder[i] != folder || size == 0) { + continue; + } + s->plan[s->plan_len].file_index = i; + s->plan[s->plan_len].size = size; + s->plan_len++; + } + return 0; +} + +static void +szx_sink_close(szx_sink_t *s) { + if(s->fd >= 0) { + close(s->fd); + s->fd = -1; + } +} + +/* Opens the next entry of the plan inside the staging tree. */ +static int +szx_sink_open(szx_sink_t *s, UInt32 file_index) { + char raw[SZX_NAME_MAX]; + char rel[ZIPX_PATH_MAX]; + char full[ZIPX_PATH_MAX]; + uint32_t depth = 0; + + if(szx_entry_name(s->db, file_index, raw, sizeof(raw), s->c)) { + return -1; + } + /* The name was validated during scan, so this can only fail if the archive + changed underneath us; still, never write a path we have not checked. */ + if(szx_normalize_name(raw, rel, sizeof(rel), &depth, s->c)) { + return -1; + } + if(szx_make_dirs(s->c, rel)) { + return -1; + } + if(snprintf(full, sizeof(full), "%s/%s", s->c->staging, rel) >= + (int)sizeof(full)) { + return szx_fail(s->c, ZIPX_ERR_LIMIT_NAME, rel, "path is too long"); + } + + s->fd = open(full, O_WRONLY | O_CREAT | O_TRUNC | O_CLOEXEC, 0666); + if(s->fd < 0) { + int saved = errno; + + return szx_fail(s->c, ZIPX_ERR_IO, rel, "cannot create: %s (errno=%d)", + strerror(saved), saved); + } + snprintf(s->cur_name, sizeof(s->cur_name), "%s", rel); + s->entry_crc = CRC_INIT_VAL; + return 0; +} + +/* The sink handed to sz_chain_decode(). A folder's decoded bytes are a single + stream that has to be split across the entries living in it, which is + exactly what makes a solid block solid. */ +static int +szx_sink_write(void *ctx, const void *data, size_t size) { + szx_sink_t *s = (szx_sink_t *)ctx; + const uint8_t *p = (const uint8_t *)data; + + while(size > 0) { + szx_plan_slot_t *e; + uint64_t remain; + size_t take; + + if(s->plan_pos >= s->plan_len) { + return szx_fail(s->c, ZIPX_ERR_FORMAT, NULL, + "folder produced %llu bytes more than its entries hold", + (unsigned long long)size); + } + e = &s->plan[s->plan_pos]; + remain = e->size - s->written; + take = (size_t)((uint64_t)size < remain ? (uint64_t)size : remain); + + if(s->fd < 0 && szx_sink_open(s, e->file_index)) { + return -1; + } + if(take) { + size_t done = 0; + + while(done < take) { + ssize_t n = write(s->fd, p + done, take - done); + + if(n <= 0) { + if(n < 0 && errno == EINTR) { + continue; + } + return szx_fail(s->c, ZIPX_ERR_IO, s->cur_name, "cannot write: %s", + strerror(errno)); + } + done += (size_t)n; + } + s->entry_crc = CrcUpdate(s->entry_crc, p, take); + } + s->written += take; + s->c->bytes_done += take; + p += take; + size -= take; + szx_report(s->c, ZIPX_PHASE_EXTRACT, s->cur_name, 0); + + if(s->written == e->size) { + UInt32 index = e->file_index; + UInt32 attrib = szx_entry_attrib(s->db, index); + + szx_sink_close(s); + if(SzBitWithVals_Check(&s->db->CRCs, index) && + CRC_GET_DIGEST(s->entry_crc) != s->db->CRCs.Vals[index]) { + return szx_fail(s->c, ZIPX_ERR_CRC, s->cur_name, + "checksum mismatch in entry data"); + } + szx_apply_metadata(s->c, s->cur_name, attrib, + szx_entry_mtime(s->db, index)); + if(szx_entry_is_dir(s->db, index, attrib)) { + s->c->dirs_created++; + } else { + s->c->files_created++; + } + s->c->entries_done++; + s->plan_pos++; + s->written = 0; + } + } + return 0; +} + +/* Maps a sz_chain failure onto the zipx contract. 7zAES gets its own code: + "password required or wrong" is actionable, "corrupt archive" is not. A + wrong password decrypts to plausible rubbish, so an encrypted folder whose + data then fails to decode is reported as a password problem, with the same + hint the chain itself appends. */ +static int +szx_decode_error(szx_ctx_t *c, UInt32 folder, const sz_chain_err_t *derr, + const char *desc, int encrypted) { + switch(derr->status) { + case SZ_CHAIN_ERR_PASSWORD: + return szx_fail(c, ZIPX_ERR_PASSWORD, NULL, "folder %u (%s): %s", + (unsigned)folder, desc, derr->message); + case SZ_CHAIN_ERR_METHOD: + case SZ_CHAIN_ERR_LAYOUT: + return szx_fail(c, ZIPX_ERR_UNSUPPORTED, NULL, "folder %u (%s): %s", + (unsigned)folder, desc, + sz_chain_status_string(derr->status)); + case SZ_CHAIN_ERR_LIMIT: + return szx_fail(c, ZIPX_ERR_LIMIT_FILE, NULL, "folder %u (%s): %s", + (unsigned)folder, desc, derr->message); + case SZ_CHAIN_ERR_CANCELED: + return szx_fail(c, ZIPX_ERR_CANCELED, NULL, NULL); + case SZ_CHAIN_ERR_MEM: + return szx_fail(c, ZIPX_ERR_INTERNAL, NULL, "folder %u (%s): %s", + (unsigned)folder, desc, derr->message); + case SZ_CHAIN_ERR_DATA: + if(encrypted) { + return szx_fail(c, ZIPX_ERR_PASSWORD, NULL, + "folder %u (%s): %s (the password is wrong, or the " + "archive is damaged)", + (unsigned)folder, desc, derr->message); + } + return szx_fail(c, ZIPX_ERR_FORMAT, NULL, "folder %u (%s): %s", + (unsigned)folder, desc, derr->message); + default: + return szx_fail(c, ZIPX_ERR_FORMAT, NULL, "folder %u (%s): %s", + (unsigned)folder, desc, derr->message); + } +} + +/* Decodes every folder into the staging tree. */ +static int +extract_folders(const CSzArEx *db, szx_ctx_t *c, szx_reader_t *reader) { + szx_sink_t sink; + UInt32 folder; + int ret = 0; + + memset(&sink, 0, sizeof(sink)); + sink.c = c; + sink.db = db; + sink.fd = -1; + + for(folder = 0; folder < db->db.NumFolders; folder++) { + uint64_t pack_positions[SZ_CHAIN_MAX_STREAMS + 1]; + UInt32 pack_count = 0; + const uint8_t *blob = db->db.CodersData + db->db.FoCodersOffsets[folder]; + size_t blob_size = db->db.FoCodersOffsets[(size_t)folder + 1] - + db->db.FoCodersOffsets[folder]; + const uint64_t *cus = + &db->db.CoderUnpackSizes[db->db.FoToCoderUnpackSizes[folder]]; + uint64_t unpack_size = SzAr_GetFolderUnpackSize(&db->db, folder); + sz_chain *chain = NULL; + sz_chain_err_t cerr; + char desc[256]; + uint32_t folder_crc = 0; + int encrypted = 0; + + if(szx_canceled(c)) { + ret = szx_fail(c, ZIPX_ERR_CANCELED, NULL, NULL); + break; + } + if(check_folder_ratio(db, folder, c)) { + ret = (int)c->result->status; + break; + } + if(szx_folder_packs(db, folder, pack_positions, &pack_count, c)) { + ret = (int)c->result->status; + break; + } + if(sz_chain_parse(&chain, blob, blob_size, pack_positions, pack_count, cus, + unpack_size, sz_chain_default_limits(), &cerr) != 0) { + ret = szx_decode_error(c, folder, &cerr, "parse", 0); + break; + } + sz_chain_describe(chain, desc, sizeof(desc)); + encrypted = sz_chain_needs_password(chain); + + /* An empty folder has no bytes to split and no entries to place; its + (zero length) files arrive through the empty-entry pass below. */ + if(unpack_size == 0) { + sz_chain_free(chain); + continue; + } + if(szx_plan_build(&sink, folder)) { + szx_plan_free(&sink); + sz_chain_free(chain); + ret = (int)c->result->status; + break; + } + + if(sz_chain_decode(chain, szx_read_at, reader, szx_sink_write, &sink, NULL, + NULL, c->password, &folder_crc, &cerr) != 0) { + /* A failure the sink already explained wins over the decoder's summary, + which can only say that the sink rejected data. */ + if(c->result->status == ZIPX_OK) { + szx_decode_error(c, folder, &cerr, desc, encrypted); + } + szx_sink_close(&sink); + szx_plan_free(&sink); + sz_chain_free(chain); + ret = (int)c->result->status; + break; + } + szx_sink_close(&sink); + + if(sink.plan_pos != sink.plan_len) { + ret = szx_fail(c, ZIPX_ERR_FORMAT, NULL, + "folder %u: only %u of %u entries were produced", + (unsigned)folder, (unsigned)sink.plan_pos, + (unsigned)sink.plan_len); + szx_plan_free(&sink); + sz_chain_free(chain); + break; + } + if(SzBitWithVals_Check(&db->db.FolderCRCs, folder) && + folder_crc != db->db.FolderCRCs.Vals[folder]) { + ret = szx_fail(c, ZIPX_ERR_CRC, NULL, + "folder %u: checksum mismatch (%08X, expected %08X)", + (unsigned)folder, (unsigned)folder_crc, + (unsigned)db->db.FolderCRCs.Vals[folder]); + szx_plan_free(&sink); + sz_chain_free(chain); + break; + } + + szx_plan_free(&sink); + sz_chain_free(chain); + } + + szx_plan_free(&sink); + return ret; +} + +/* Creates everything a folder stream does not carry: the directories and the + zero-byte files. They have no bytes to place, so they go straight into the + staging tree. */ +static int +extract_empty_entries(const CSzArEx *db, szx_ctx_t *c) { + UInt32 i; + + for(i = 0; i < db->NumFiles; i++) { + char raw[SZX_NAME_MAX]; + char rel[ZIPX_PATH_MAX]; + char full[ZIPX_PATH_MAX]; + UInt32 attrib = szx_entry_attrib(db, i); + uint32_t depth = 0; + int is_dir = szx_entry_is_dir(db, i, attrib); + + if(szx_entry_size(db, i) != 0) { + continue; + } + if(szx_entry_name(db, i, raw, sizeof(raw), c)) { + return (int)c->result->status; + } + if(szx_normalize_name(raw, rel, sizeof(rel), &depth, c)) { + return (int)c->result->status; + } + (void)depth; + + if(szx_make_dirs(c, rel)) { + return (int)c->result->status; + } + if(snprintf(full, sizeof(full), "%s/%s", c->staging, rel) >= + (int)sizeof(full)) { + return szx_fail(c, ZIPX_ERR_LIMIT_NAME, rel, "path is too long"); + } + + if(is_dir) { + if(mkdir(full, 0777) && errno != EEXIST) { + return szx_fail(c, ZIPX_ERR_IO, rel, "cannot create directory: %s", + strerror(errno)); + } + c->dirs_created++; + } else { + int fd = open(full, O_WRONLY | O_CREAT | O_TRUNC | O_CLOEXEC, 0666); + + if(fd < 0) { + return szx_fail(c, ZIPX_ERR_IO, rel, "cannot create: %s", + strerror(errno)); + } + close(fd); + c->files_created++; + } + szx_apply_metadata(c, rel, attrib, szx_entry_mtime(db, i)); + c->entries_done++; + szx_report(c, ZIPX_PHASE_EXTRACT, rel, 0); + } + return 0; +} + +/************************************************************************** + * publish phase (identical to zip_extract.c / rar_extract.c) + **************************************************************************/ + +static int szx_publish_dir(const char *src, const char *dst, int depth, + szx_ctx_t *c); + +static int +szx_publish_entry(const char *src, const char *dst, const char *name, int depth, + szx_ctx_t *c) { + char src_child[ZIPX_PATH_MAX]; + char dst_child[ZIPX_PATH_MAX]; + struct stat st; + struct stat src_st; + int src_is_dir; + + if(snprintf(src_child, sizeof(src_child), "%s/%s", src, name) >= + (int)sizeof(src_child) || + snprintf(dst_child, sizeof(dst_child), "%s/%s", dst, name) >= + (int)sizeof(dst_child)) { + return szx_fail(c, ZIPX_ERR_LIMIT_NAME, name, "path is too long"); + } + if(lstat(src_child, &src_st)) { + return szx_fail(c, ZIPX_ERR_IO, src_child, "cannot read staging: %s", + strerror(errno)); + } + src_is_dir = S_ISDIR(src_st.st_mode) ? 1 : 0; + + if(lstat(dst_child, &st)) { + if(errno != ENOENT) { + return szx_fail(c, ZIPX_ERR_IO, dst_child, "cannot check target: %s", + strerror(errno)); + } + if(rename(src_child, dst_child)) { + return szx_fail(c, ZIPX_ERR_IO, dst_child, "cannot publish: %s", + strerror(errno)); + } + return szx_remember_published(c, dst_child, src_is_dir); + } + + if(S_ISDIR(st.st_mode)) { + if(!src_is_dir) { + return szx_fail(c, ZIPX_ERR_CONFLICT, dst_child, + "file collides with an existing directory"); + } + /* Both MERGE and OVERWRITE treat a matching directory tree as a recursion: + the user's choice between them only matters at the file leaves (see + below). Refusing up front would make OVERWRITE useless for any archive + that overlaps a directory already on disk. */ + if(c->conflict != ZIPX_CONFLICT_MERGE && + c->conflict != ZIPX_CONFLICT_OVERWRITE) { + return szx_fail(c, ZIPX_ERR_CONFLICT, dst_child, + "directory already exists"); + } + if(depth >= SZX_PUBLISH_MAX_DEPTH) { + return szx_fail(c, ZIPX_ERR_LIMIT_DEPTH, dst_child, "path is too deep"); + } + if(szx_publish_dir(src_child, dst_child, depth + 1, c)) { + return -1; + } + rmdir(src_child); + return 0; + } + + if(S_ISREG(st.st_mode) && !src_is_dir) { + if(c->conflict == ZIPX_CONFLICT_OVERWRITE) { + if(rename(src_child, dst_child)) { + return szx_fail(c, ZIPX_ERR_IO, dst_child, "cannot replace: %s", + strerror(errno)); + } + return szx_remember_published(c, dst_child, 0); + } + if(c->conflict == ZIPX_CONFLICT_MERGE) { + if(unlink(src_child)) { + return szx_fail(c, ZIPX_ERR_IO, dst_child, "cannot drop staged file: %s", + strerror(errno)); + } + return 0; + } + } + return szx_fail(c, ZIPX_ERR_CONFLICT, dst_child, "target already exists"); +} + +static int +szx_publish_dir(const char *src, const char *dst, int depth, szx_ctx_t *c) { + DIR *dir = opendir(src); + struct dirent *ent; + char **names = NULL; + size_t count = 0; + size_t cap = 0; + size_t i; + int ret = 0; + + if(!dir) { + return szx_fail(c, ZIPX_ERR_IO, src, "cannot read staging: %s", + strerror(errno)); + } + while((ent = readdir(dir))) { + if(!strcmp(ent->d_name, ".") || !strcmp(ent->d_name, "..")) { + continue; + } + if(count == cap) { + size_t next = cap ? cap * 2 : 64; + char **grown = realloc(names, next * sizeof(*grown)); + + if(!grown) { + ret = szx_fail(c, ZIPX_ERR_INTERNAL, src, "out of memory"); + break; + } + names = grown; + cap = next; + } + if(!(names[count] = strdup(ent->d_name))) { + ret = szx_fail(c, ZIPX_ERR_INTERNAL, src, "out of memory"); + break; + } + count++; + } + closedir(dir); + + for(i = 0; i < count && !ret; i++) { + szx_report(c, ZIPX_PHASE_PUBLISH, names[i], 0); + ret = szx_publish_entry(src, dst, names[i], depth, c); + } + for(i = 0; i < count; i++) { + free(names[i]); + } + free(names); + return ret; +} + +static int +szx_publish_staging(szx_ctx_t *c, const char *dst_dir, int dst_existed) { + int ret; + + szx_report(c, ZIPX_PHASE_PUBLISH, dst_dir, 1); + if(!dst_existed) { + if(rename(c->staging, dst_dir)) { + return szx_fail(c, ZIPX_ERR_IO, dst_dir, "cannot publish: %s", + strerror(errno)); + } + c->staging_created = 0; + return 0; + } + ret = szx_publish_dir(c->staging, dst_dir, 0, c); + if(ret) { + szx_rollback_published(c); + } + return ret; +} + +/************************************************************************** + * public entry point + **************************************************************************/ + +zipx_status_t +sevenz_extract(const char *sevenz_path, const char *dst_dir, + zipx_conflict_t conflict, const zipx_limits_t *limits, + zipx_cancel_fn cancel, zipx_progress_fn progress, void *userdata, + const char *password, zipx_result_t *result) { + szx_ctx_t ctx; + szx_ctx_t *c = &ctx; + char parent[ZIPX_PATH_MAX]; + char dst_copy[ZIPX_PATH_MAX]; + struct stat st; + CLookToRead2 look_stream; + Byte *look_buf = NULL; + CSzArEx db; + int db_open = 0; + sevenz_volstream *vol = NULL; + char *vol_err = NULL; + char vol_desc[512] = ""; + szx_reader_t reader; + SRes res; + int dst_existed = 0; + int status; + static int tables_ready; + + if(!result || !sevenz_path || !sevenz_path[0] || !dst_dir || !dst_dir[0]) { + if(result) { + memset(result, 0, sizeof(*result)); + result->status = ZIPX_ERR_INTERNAL; + snprintf(result->message, sizeof(result->message), "invalid argument"); + } + return ZIPX_ERR_INTERNAL; + } + + memset(&ctx, 0, sizeof(ctx)); + memset(result, 0, sizeof(*result)); + c->result = result; + c->conflict = conflict; + c->limits = limits ? *limits : *zipx_default_limits(); + c->cancel = cancel; + c->progress = progress; + c->userdata = userdata; + c->password = password; + + snprintf(dst_copy, sizeof(dst_copy), "%s", dst_dir); + { + size_t len = strlen(dst_copy); + + while(len > 1 && dst_copy[len - 1] == '/') { + dst_copy[--len] = 0; + } + } + if(szx_path_parent(dst_copy, parent, sizeof(parent))) { + status = szx_fail(c, ZIPX_ERR_INTERNAL, dst_dir, "invalid destination"); + goto done; + } + if(stat(parent, &st) || !S_ISDIR(st.st_mode)) { + status = szx_fail(c, ZIPX_ERR_IO, parent, "destination parent is missing"); + goto done; + } + dst_existed = !lstat(dst_copy, &st); + if(dst_existed && !S_ISDIR(st.st_mode)) { + status = szx_fail(c, ZIPX_ERR_CONFLICT, dst_copy, + "destination is not a directory"); + goto done; + } + + if(sevenz_volstream_open(&vol, sevenz_path, NULL, &vol_err) != 0) { + /* The volume detector knows which part is missing and names it; the + generic message is only for the cases it has no opinion about. */ + status = szx_fail(c, ZIPX_ERR_OPEN, sevenz_path, "%s", + vol_err ? vol_err : "cannot open the archive"); + free(vol_err); + goto done; + } + free(vol_err); + sevenz_volstream_describe(vol, vol_desc, sizeof(vol_desc)); + + if(!tables_ready) { + CrcGenerateTable(); + tables_ready = 1; + } + LookToRead2_CreateVTable(&look_stream, 0); + look_buf = (Byte *)ISzAlloc_Alloc(&g_sz_alloc, SZX_INPUT_BUF_SIZE); + if(!look_buf) { + status = szx_fail(c, ZIPX_ERR_INTERNAL, NULL, "out of memory"); + goto done; + } + look_stream.buf = look_buf; + look_stream.bufSize = SZX_INPUT_BUF_SIZE; + look_stream.realStream = sevenz_volstream_stream(vol); + LookToRead2_INIT(&look_stream); + + SzArEx_Init(&db); + res = SzArEx_Open(&db, &look_stream.vt, &g_sz_alloc, &g_sz_alloc); + if(res != SZ_OK) { + /* SzArEx_Open already released everything it allocated. */ + if(res == SZ_ERROR_UNSUPPORTED) { + status = szx_fail(c, ZIPX_ERR_UNSUPPORTED, sevenz_path, + "%s: the archive header is encrypted (-mhe=on); only " + "archives whose header is readable can be unpacked", + vol_desc); + } else { + status = szx_fail(c, ZIPX_ERR_FORMAT, sevenz_path, + "%s: cannot read the 7z header (code %d)", vol_desc, + (int)res); + } + goto done; + } + db_open = 1; + + if(scan_entries(&db, c)) { + status = (int)c->result->status; + goto done; + } + if(szx_canceled(c)) { + status = szx_fail(c, ZIPX_ERR_CANCELED, NULL, NULL); + goto done; + } + if(precheck_folders(&db, c)) { + status = (int)c->result->status; + goto done; + } + if(szx_check_space(c, dst_existed ? dst_copy : parent)) { + status = (int)c->result->status; + goto done; + } + if(szx_make_staging(c, parent)) { + status = (int)c->result->status; + goto done; + } + + reader.stream = sevenz_volstream_stream(vol); + reader.base = db.dataPos; + + if(extract_folders(&db, c, &reader)) { + status = (int)c->result->status; + goto done; + } + if(extract_empty_entries(&db, c)) { + status = (int)c->result->status; + goto done; + } + if(szx_canceled(c)) { + status = szx_fail(c, ZIPX_ERR_CANCELED, NULL, NULL); + goto done; + } + if(c->entries_done != c->entries_total) { + status = szx_fail(c, ZIPX_ERR_FORMAT, NULL, + "%llu of %llu entries were written; the archive's entry " + "table and its folders disagree", + (unsigned long long)c->entries_done, + (unsigned long long)c->entries_total); + goto done; + } + + SzArEx_Free(&db, &g_sz_alloc); + db_open = 0; + sevenz_volstream_free(vol); + vol = NULL; + + if(szx_publish_staging(c, dst_copy, dst_existed)) { + status = (int)c->result->status; + goto done; + } + + result->status = ZIPX_OK; + status = ZIPX_OK; + +done: + if(db_open) { + SzArEx_Free(&db, &g_sz_alloc); + } + if(look_buf) { + ISzAlloc_Free(&g_sz_alloc, look_buf); + } + sevenz_volstream_free(vol); + szx_cleanup_staging(c); + result->entries_total = c->entries_total; + result->entries_done = c->entries_done; + result->bytes_total = c->bytes_total; + result->files_created = c->files_created; + result->dirs_created = c->dirs_created; + if(status != ZIPX_OK && !result->message[0]) { + snprintf(result->message, sizeof(result->message), "%s", + zipx_status_string(result->status)); + } + if(status != ZIPX_OK) { + result->status = (zipx_status_t)status; + } + szx_report(c, ZIPX_PHASE_CLEANUP, NULL, 1); + szx_free_published(c); + return result->status; +} diff --git a/src/sevenz_extract.h b/src/sevenz_extract.h new file mode 100644 index 0000000..eb9dcb4 --- /dev/null +++ b/src/sevenz_extract.h @@ -0,0 +1,37 @@ +#pragma once + +/* Standalone 7z extraction engine, the third sibling of zip_extract.c and + rar_extract.c. Like them it has no HTTP or task dependencies, and it fills + in the same zipx_result_t so a caller can treat every format alike. + + Input may be a single `name.7z` or a byte-split set (`name.7z.001`, ...): + both reach the decoder through src/sevenz_volstream.c. + + The publish / staging / rollback / name-validation machinery is mirrored + from rar_extract.c on purpose -- three self-contained engines is the shape + this project has settled on, so that a format's bugs stay inside its file. + + Backend notes (LZMA SDK 26.03 + src/sevenz_chain.c): + * Copy / LZMA / LZMA2 / PPMd, the Delta filter and the x86 / PPC / IA64 / + ARM / ARMT / SPARC branch converters, BCJ2, and 7zAES. + * An encrypted *header* (`-mhe=on`) is not readable: the SDK refuses it + before any folder is known, and we report exactly that. +*/ + +#include "zip_extract.h" + +/* Extract sevenz_path into dst_dir. + `password` is the archive password as UTF-8, or NULL / "" when the caller + has none. It is only consulted by archives that encrypt their streams. + + Returns ZIPX_OK or an error code; *result is always filled in. A missing or + wrong password comes back as ZIPX_ERR_PASSWORD so the caller can ask for one + and retry. On any failure the staging directory is removed and dst_dir is + left as it was, except for objects already published under the overwrite + policy. */ +zipx_status_t sevenz_extract(const char *sevenz_path, const char *dst_dir, + zipx_conflict_t conflict, + const zipx_limits_t *limits, + zipx_cancel_fn cancel, + zipx_progress_fn progress, void *userdata, + const char *password, zipx_result_t *result); diff --git a/src/zip_extract.c b/src/zip_extract.c index 64de918..ab5c58a 100644 --- a/src/zip_extract.c +++ b/src/zip_extract.c @@ -36,93 +36,8 @@ #define ZIPX_PUBLISH_MAX_DEPTH 128 #define ZIPX_SPACE_SLACK_PER_ENTRY 512 -/* Default limits. - * - * Tuned to cover real-world PS5 workloads without prompting: - * - PS5 system backup ZIPs (~200-300 GiB total, individual chunks <64 GiB) - * - 3A-game archives with a single ~300 GiB uncompressed file - * - * Safety against zip bombs is delegated to: - * 1. `check_space()` (statvfs-based real disk space check) before extract - * 2. `max_ratio` below (declared compression ratio cap) - * The size caps here are an early-fail UX guard, not a security boundary. - */ -static const zipx_limits_t k_default_limits = { - .max_entries = 200000, - .max_total_bytes = 2ULL * 1024 * 1024 * 1024 * 1024, - .max_file_bytes = 512ULL * 1024 * 1024 * 1024, - .max_ratio = 500, - /* Only entries that would individually materialise >=1 GiB are screened - by ratio; anything smaller is harmless (bounded by declared size + the - real free-space check) and is commonly highly compressible in - legitimate archives. */ - .ratio_min_bytes = 1ULL * 1024 * 1024 * 1024, - .max_depth = 32, - .max_name_len = 255, - .max_path_len = 1024 -}; - -/* Large profile for archives that exceed the default cap. - * - * - max_file_bytes = 1 TiB (single uncompressed file) - * - max_total_bytes = 4 TiB (whole archive) - * - max_ratio = 1000 (relaxed ratio cap; check_space still applies) - * - * Requires the user to opt in via the web UI (large=1) before these take - * effect. Default limits must always be strictly smaller than large so the - * large profile is unambiguously a relaxation. - */ -static const zipx_limits_t k_large_limits = { - .max_entries = 500000, - .max_total_bytes = 4ULL * 1024 * 1024 * 1024 * 1024, - .max_file_bytes = 1ULL * 1024 * 1024 * 1024 * 1024, - .max_ratio = 1000, - .ratio_min_bytes = 1ULL * 1024 * 1024 * 1024, - .max_depth = 32, - .max_name_len = 255, - .max_path_len = 1024 -}; - -const zipx_limits_t * -zipx_default_limits(void) { - return &k_default_limits; -} - -const zipx_limits_t * -zipx_limits_profile(int profile) { - switch(profile) { - case ZIPX_LIMITS_LARGE: - return &k_large_limits; - case ZIPX_LIMITS_DEFAULT: - default: - return &k_default_limits; - } -} - -const char * -zipx_status_string(zipx_status_t status) { - switch(status) { - case ZIPX_OK: return "ok"; - case ZIPX_ERR_CANCELED: return "canceled"; - case ZIPX_ERR_OPEN: return "cannot open archive"; - case ZIPX_ERR_FORMAT: return "corrupt archive"; - case ZIPX_ERR_UNSUPPORTED: return "unsupported archive"; - case ZIPX_ERR_UNSAFE_NAME: return "unsafe entry name"; - case ZIPX_ERR_SPECIAL: return "unsupported entry type"; - case ZIPX_ERR_DUPLICATE: return "duplicate entry name"; - case ZIPX_ERR_LIMIT_ENTRIES: return "too many entries"; - case ZIPX_ERR_LIMIT_FILE: return "entry too large"; - case ZIPX_ERR_LIMIT_TOTAL: return "archive contents too large"; - case ZIPX_ERR_LIMIT_RATIO: return "compression ratio too high"; - case ZIPX_ERR_LIMIT_DEPTH: return "path too deep"; - case ZIPX_ERR_LIMIT_NAME: return "path too long"; - case ZIPX_ERR_CONFLICT: return "target already exists"; - case ZIPX_ERR_SPACE: return "not enough space"; - case ZIPX_ERR_IO: return "read or write failed"; - case ZIPX_ERR_CRC: return "crc mismatch"; - default: return "internal error"; - } -} +/* The limit profiles and zipx_status_string() are format independent and live + in src/zipx_common.c, which every engine links. */ /************************************************************************** * small helpers @@ -1354,7 +1269,7 @@ zipx_extract(const char *zip_path, const char *dst_dir, memset(result, 0, sizeof(*result)); c->result = result; c->conflict = conflict; - c->limits = limits ? *limits : k_default_limits; + c->limits = limits ? *limits : *zipx_default_limits(); c->cancel = cancel; c->progress = progress; c->userdata = userdata; diff --git a/src/zip_extract.h b/src/zip_extract.h index b1e86b0..ac5b6c4 100644 --- a/src/zip_extract.h +++ b/src/zip_extract.h @@ -31,6 +31,8 @@ typedef enum { ZIPX_ERR_LIMIT_DEPTH, ZIPX_ERR_LIMIT_NAME, ZIPX_ERR_CONFLICT, /* target already exists for the chosen policy */ + ZIPX_ERR_PASSWORD, /* the archive is encrypted and the password is missing + or wrong; the caller can prompt and retry */ ZIPX_ERR_SPACE, ZIPX_ERR_IO, ZIPX_ERR_CRC, diff --git a/src/zipx_common.c b/src/zipx_common.c new file mode 100644 index 0000000..8a3cffc --- /dev/null +++ b/src/zipx_common.c @@ -0,0 +1,99 @@ +/* Bits of the zipx_* contract that are not specific to a container format. + + The limit profiles and the status-to-text mapping describe the *engine + family*, not ZIP, so they live here rather than inside zip_extract.c. All + three engines (ZIP, RAR, 7z) link this one object; keeping them in the ZIP + file would force the RAR and 7z test builds to drag in minizip-ng and zlib + for the sake of three functions. */ + +#include "zip_extract.h" + +/* Default limits. + * + * Tuned to cover real-world PS5 workloads without prompting: + * - PS5 system backup archives (~200-300 GiB total, individual chunks + * well under 64 GiB) + * - 3A-game archives with a single ~300 GiB uncompressed file + * + * Safety against decompression bombs is delegated to: + * 1. `check_space()` (statvfs-based real disk space check) before extract + * 2. `max_ratio` below (declared compression ratio cap) + * The size caps here are an early-fail UX guard, not a security boundary. + */ +static const zipx_limits_t k_default_limits = { + .max_entries = 200000, + .max_total_bytes = 2ULL * 1024 * 1024 * 1024 * 1024, + .max_file_bytes = 512ULL * 1024 * 1024 * 1024, + .max_ratio = 500, + /* Only entries that would individually materialise >=1 GiB are screened + by ratio; anything smaller is harmless (bounded by declared size + the + real free-space check) and is commonly highly compressible in + legitimate archives. */ + .ratio_min_bytes = 1ULL * 1024 * 1024 * 1024, + .max_depth = 32, + .max_name_len = 255, + .max_path_len = 1024 +}; + +/* Large profile for archives that exceed the default cap. + * + * - max_file_bytes = 1 TiB (single uncompressed file) + * - max_total_bytes = 4 TiB (whole archive) + * - max_ratio = 1000 (relaxed ratio cap; check_space still applies) + * + * Requires the user to opt in via the web UI (large=1) before these take + * effect. Default limits must always be strictly smaller than large so the + * large profile is unambiguously a relaxation. + */ +static const zipx_limits_t k_large_limits = { + .max_entries = 500000, + .max_total_bytes = 4ULL * 1024 * 1024 * 1024 * 1024, + .max_file_bytes = 1ULL * 1024 * 1024 * 1024 * 1024, + .max_ratio = 1000, + .ratio_min_bytes = 1ULL * 1024 * 1024 * 1024, + .max_depth = 32, + .max_name_len = 255, + .max_path_len = 1024 +}; + +const zipx_limits_t * +zipx_default_limits(void) { + return &k_default_limits; +} + +const zipx_limits_t * +zipx_limits_profile(int profile) { + switch(profile) { + case ZIPX_LIMITS_LARGE: + return &k_large_limits; + case ZIPX_LIMITS_DEFAULT: + default: + return &k_default_limits; + } +} + +const char * +zipx_status_string(zipx_status_t status) { + switch(status) { + case ZIPX_OK: return "ok"; + case ZIPX_ERR_CANCELED: return "canceled"; + case ZIPX_ERR_OPEN: return "cannot open archive"; + case ZIPX_ERR_FORMAT: return "corrupt archive"; + case ZIPX_ERR_UNSUPPORTED: return "unsupported archive"; + case ZIPX_ERR_UNSAFE_NAME: return "unsafe entry name"; + case ZIPX_ERR_SPECIAL: return "unsupported entry type"; + case ZIPX_ERR_DUPLICATE: return "duplicate entry name"; + case ZIPX_ERR_LIMIT_ENTRIES: return "too many entries"; + case ZIPX_ERR_LIMIT_FILE: return "entry too large"; + case ZIPX_ERR_LIMIT_TOTAL: return "archive contents too large"; + case ZIPX_ERR_LIMIT_RATIO: return "compression ratio too high"; + case ZIPX_ERR_LIMIT_DEPTH: return "path too deep"; + case ZIPX_ERR_LIMIT_NAME: return "path too long"; + case ZIPX_ERR_CONFLICT: return "target already exists"; + case ZIPX_ERR_PASSWORD: return "password required or wrong"; + case ZIPX_ERR_SPACE: return "not enough space"; + case ZIPX_ERR_IO: return "read or write failed"; + case ZIPX_ERR_CRC: return "crc mismatch"; + default: return "internal error"; + } +} diff --git a/tests/fixtures/broken.zip.001 b/tests/fixtures/broken.zip.001 new file mode 100644 index 0000000..5f415aa Binary files /dev/null and b/tests/fixtures/broken.zip.001 differ diff --git a/tests/fixtures/disks.z01 b/tests/fixtures/disks.z01 new file mode 100644 index 0000000..bbc4dd5 Binary files /dev/null and b/tests/fixtures/disks.z01 differ diff --git a/tests/fixtures/disks.zip b/tests/fixtures/disks.zip new file mode 100644 index 0000000..c082181 Binary files /dev/null and b/tests/fixtures/disks.zip differ diff --git a/tests/fixtures/gap.zip.001 b/tests/fixtures/gap.zip.001 new file mode 100644 index 0000000..5f415aa Binary files /dev/null and b/tests/fixtures/gap.zip.001 differ diff --git a/tests/fixtures/gap.zip.003 b/tests/fixtures/gap.zip.003 new file mode 100644 index 0000000..5ac15ff Binary files /dev/null and b/tests/fixtures/gap.zip.003 differ diff --git a/tests/fixtures/parts.part1.zip b/tests/fixtures/parts.part1.zip new file mode 100644 index 0000000..5f415aa Binary files /dev/null and b/tests/fixtures/parts.part1.zip differ diff --git a/tests/fixtures/parts.part2.zip b/tests/fixtures/parts.part2.zip new file mode 100644 index 0000000..b0ff256 Binary files /dev/null and b/tests/fixtures/parts.part2.zip differ diff --git a/tests/fixtures/parts.part3.zip b/tests/fixtures/parts.part3.zip new file mode 100644 index 0000000..5ac15ff Binary files /dev/null and b/tests/fixtures/parts.part3.zip differ diff --git a/tests/fixtures/plain.zip.001 b/tests/fixtures/plain.zip.001 new file mode 100644 index 0000000..5f415aa Binary files /dev/null and b/tests/fixtures/plain.zip.001 differ diff --git a/tests/fixtures/plain.zip.002 b/tests/fixtures/plain.zip.002 new file mode 100644 index 0000000..b0ff256 Binary files /dev/null and b/tests/fixtures/plain.zip.002 differ diff --git a/tests/fixtures/plain.zip.003 b/tests/fixtures/plain.zip.003 new file mode 100644 index 0000000..5ac15ff Binary files /dev/null and b/tests/fixtures/plain.zip.003 differ diff --git a/tests/fixtures/split_single.zip b/tests/fixtures/split_single.zip new file mode 100644 index 0000000..697a086 Binary files /dev/null and b/tests/fixtures/split_single.zip differ diff --git a/tests/posix_compat.h b/tests/posix_compat.h index c18a60f..ff6cef6 100644 --- a/tests/posix_compat.h +++ b/tests/posix_compat.h @@ -1,262 +1,474 @@ -/* Host test shim: lets the POSIX extraction engine build and run on MinGW. - Injected with gcc -include for the test build only; never compiled into the - PS5 payload. It maps the *at() calls onto plain paths and fakes the few - POSIX bits Windows lacks (symlinks and O_NOFOLLOW have no Windows - equivalent, which is why the symlink tests are skipped there). */ - -#ifndef WFM_TEST_POSIX_COMPAT_H -#define WFM_TEST_POSIX_COMPAT_H - -#if defined(__MINGW32__) || defined(_WIN32) - -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include - -#ifndef PATH_MAX -#define PATH_MAX 4096 -#endif - -#define O_NOFOLLOW 0 -#define O_CLOEXEC 0 -/* Windows cannot open a directory with _open(); a non-zero sentinel lets the - shim detect directory opens and hand back a synthetic dirfd. */ -#define O_DIRECTORY 0x10000 -#define AT_SYMLINK_NOFOLLOW 0 -#define AT_REMOVEDIR 0x0200 -#ifndef S_IFLNK -#define S_IFLNK 0xA000 -#endif -#ifndef S_ISLNK -#define S_ISLNK(m) (((m) & S_IFMT) == S_IFLNK) -#endif - -#ifndef CLOCK_MONOTONIC -#define CLOCK_MONOTONIC 1 -#endif - -#define WFM_FD_SLOTS 512 - -#define open(...) wfm_open(__VA_ARGS__) - -static struct { - int fd; - char path[PATH_MAX]; -} wfm_fd_slots[WFM_FD_SLOTS]; - -static void __attribute__((unused)) -wfm_fd_set(int fd, const char *path) { - int i; - - if(fd < 0) { - return; - } - for(i = 0; i < WFM_FD_SLOTS; i++) { - if(wfm_fd_slots[i].fd == fd || !wfm_fd_slots[i].path[0]) { - wfm_fd_slots[i].fd = fd; - snprintf(wfm_fd_slots[i].path, PATH_MAX, "%s", path); - return; - } - } -} - -static void __attribute__((unused)) -wfm_fd_clear(int fd) { - int i; - - for(i = 0; i < WFM_FD_SLOTS; i++) { - if(wfm_fd_slots[i].fd == fd) { - wfm_fd_slots[i].fd = -1; - wfm_fd_slots[i].path[0] = 0; - return; - } - } -} - -static const char * -wfm_fd_path(int fd) { - int i; - - for(i = 0; i < WFM_FD_SLOTS; i++) { - if(wfm_fd_slots[i].fd == fd) { - return wfm_fd_slots[i].path; - } - } - return NULL; -} - -static int -wfm_join(int dirfd, const char *rel, char *out, size_t out_size) { - const char *base = wfm_fd_path(dirfd); - - if(!base) { - errno = EBADF; - return -1; - } - if(snprintf(out, out_size, "%s/%s", base, rel) >= (int)out_size) { - errno = ENAMETOOLONG; - return -1; - } - return 0; -} - -static int __attribute__((unused)) -wfm_open(const char *path, int flags, ...) { - int mode = 0; - int fd; - - if(flags & O_CREAT) { - va_list ap; - - va_start(ap, flags); - mode = va_arg(ap, int); - va_end(ap); - } - /* Directory opens become synthetic fds so openat/mkdirat can resolve them - to paths; _open() returns EACCES for directories on Windows. */ - if(flags & O_DIRECTORY) { - static int next_dirfd = 0x10000; - - fd = next_dirfd++; - wfm_fd_set(fd, path); - return fd; - } - /* Force O_BINARY: MinGW's _open defaults to text mode, which would - translate LF -> CRLF on write and corrupt binary payloads. */ - fd = _open(path, (flags & ~(O_NOFOLLOW | O_DIRECTORY | O_CLOEXEC)) | O_BINARY, - mode); - if(fd >= 0) { - wfm_fd_set(fd, path); - } - return fd; -} - -static int __attribute__((unused)) -wfm_openat(int dirfd, const char *path, int flags, ...) { - char full[PATH_MAX]; - int mode = 0; - - if(wfm_join(dirfd, path, full, sizeof(full))) { - return -1; - } - if(flags & O_CREAT) { - va_list ap; - - va_start(ap, flags); - mode = va_arg(ap, int); - va_end(ap); - } - return wfm_open(full, flags, mode); -} - -static int __attribute__((unused)) -wfm_mkdirat(int dirfd, const char *path, mode_t mode) { - char full[PATH_MAX]; - - (void)mode; - if(wfm_join(dirfd, path, full, sizeof(full))) { - return -1; - } - return mkdir(full); -} - -static int __attribute__((unused)) -wfm_renameat(int from_fd, const char *from, int to_fd, const char *to) { - char src[PATH_MAX]; - char dst[PATH_MAX]; - - if(wfm_join(from_fd, from, src, sizeof(src)) || - wfm_join(to_fd, to, dst, sizeof(dst))) { - return -1; - } - /* Windows rename() refuses to replace an existing file. */ - if(_access(dst, 0) == 0) { - if(remove(dst)) { - return -1; - } - } - return rename(src, dst); -} - -static int __attribute__((unused)) -wfm_rename(const char *from, const char *to) { - /* Windows rename() refuses to replace an existing file, unlike POSIX. */ - if(_access(to, 0) == 0) { - if(remove(to)) { - return -1; - } - } - return rename(from, to); -} - -static int __attribute__((unused)) -wfm_unlinkat(int dirfd, const char *path, int flags) { - char full[PATH_MAX]; - - if(wfm_join(dirfd, path, full, sizeof(full))) { - return -1; - } - return (flags & AT_REMOVEDIR) ? rmdir(full) : unlink(full); -} - -static int __attribute__((unused)) -wfm_mkdir1(const char *path) { - return mkdir(path); /* MinGW's mkdir() takes a single argument. */ -} - -static int __attribute__((unused)) -wfm_fstatat(int dirfd, const char *path, struct stat *st, int flags) { - char full[PATH_MAX]; - - (void)flags; - if(wfm_join(dirfd, path, full, sizeof(full))) { - return -1; - } - return stat(full, st); -} - -static int __attribute__((unused)) -wfm_fsync(int fd) { - return _commit(fd); -} - -static int __attribute__((unused)) -wfm_fchmod(int fd, mode_t mode) { - (void)fd; - (void)mode; - return 0; /* Windows has no Unix modes; the engine ignores this failure. */ -} - -static int __attribute__((unused)) -wfm_close(int fd) { - wfm_fd_clear(fd); - if(fd >= 0x10000) { - return 0; /* synthetic dirfd, nothing to close */ - } - return _close(fd); -} - -#define mkdir(p, ...) wfm_mkdir1(p) -#define open(...) wfm_open(__VA_ARGS__) -#define openat(...) wfm_openat(__VA_ARGS__) -#define mkdirat(d, p, m) wfm_mkdirat(d, p, m) -#define rename(a, b) wfm_rename(a, b) -#define renameat(sd, sp, dd, dp) wfm_renameat(sd, sp, dd, dp) -#define unlinkat(d, p, f) wfm_unlinkat(d, p, f) -#define fstatat(d, p, s, f) wfm_fstatat(d, p, s, f) -#define fsync(fd) wfm_fsync(fd) -#define fchmod(fd, mode) wfm_fchmod(fd, mode) -#define close(fd) wfm_close(fd) -#define lstat(p, s) stat(p, s) - -#endif /* _WIN32 */ - -#endif /* WFM_TEST_POSIX_COMPAT_H */ +/* Host test shim: lets the POSIX extraction engine build and run on MinGW. + Injected with gcc -include for the test build only; never compiled into the + PS5 payload. It maps the *at() calls onto plain paths and fakes the few + POSIX bits Windows lacks (symlinks and O_NOFOLLOW have no Windows + equivalent, which is why the symlink tests are skipped there). */ + +#ifndef WFM_TEST_POSIX_COMPAT_H +#define WFM_TEST_POSIX_COMPAT_H + +/* _wopendir / struct _wdirent require Vista+; pull the SDK level up before any + system header touches the type definitions. */ +#ifndef _WIN32_WINNT +#define _WIN32_WINNT 0x0600 +#endif + +#if defined(__MINGW32__) || defined(_WIN32) + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#ifndef PATH_MAX +#define PATH_MAX 4096 +#endif + +#define O_NOFOLLOW 0 +#define O_CLOEXEC 0 +/* Windows cannot open a directory with _open(); a non-zero sentinel lets the + shim detect directory opens and hand back a synthetic dirfd. */ +#define O_DIRECTORY 0x10000 +#define AT_SYMLINK_NOFOLLOW 0 +#define AT_REMOVEDIR 0x0200 +#ifndef S_IFLNK +#define S_IFLNK 0xA000 +#endif +#ifndef S_ISLNK +#define S_ISLNK(m) (((m) & S_IFMT) == S_IFLNK) +#endif + +#ifndef CLOCK_MONOTONIC +#define CLOCK_MONOTONIC 1 +#endif + +#define WFM_FD_SLOTS 512 + +#define open(...) wfm_open(__VA_ARGS__) + +static struct { + int fd; + char path[PATH_MAX]; +} wfm_fd_slots[WFM_FD_SLOTS]; + +static void __attribute__((unused)) +wfm_fd_set(int fd, const char *path) { + int i; + + if(fd < 0) { + return; + } + for(i = 0; i < WFM_FD_SLOTS; i++) { + if(wfm_fd_slots[i].fd == fd || !wfm_fd_slots[i].path[0]) { + wfm_fd_slots[i].fd = fd; + snprintf(wfm_fd_slots[i].path, PATH_MAX, "%s", path); + return; + } + } +} + +static void __attribute__((unused)) +wfm_fd_clear(int fd) { + int i; + + for(i = 0; i < WFM_FD_SLOTS; i++) { + if(wfm_fd_slots[i].fd == fd) { + wfm_fd_slots[i].fd = -1; + wfm_fd_slots[i].path[0] = 0; + return; + } + } +} + +static const char * +wfm_fd_path(int fd) { + int i; + + for(i = 0; i < WFM_FD_SLOTS; i++) { + if(wfm_fd_slots[i].fd == fd) { + return wfm_fd_slots[i].path; + } + } + return NULL; +} + +static int +wfm_join(int dirfd, const char *rel, char *out, size_t out_size) { + const char *base = wfm_fd_path(dirfd); + + if(!base) { + errno = EBADF; + return -1; + } + if(snprintf(out, out_size, "%s/%s", base, rel) >= (int)out_size) { + errno = ENAMETOOLONG; + return -1; + } + return 0; +} + +/* UTF-8 to UTF-16, for the wide entry points below. The engines speak UTF-8 + (that is what an archive stores), but MinGW's ANSI entry points decode their + argument in the system code page: on a CP936 or CP1252 host a name such as + "中文-テスト.txt" is either mangled or rejected outright with the + unhelpful errno -1. Going through the wide API keeps the on-disk name + identical to the archive's. */ +static void __attribute__((unused)) +wfm_wide(const char *src, wchar_t *dst, size_t cap) { + const unsigned char *p = (const unsigned char *)src; + size_t out = 0; + + while(*p && out + 2 < cap) { + unsigned long cp = *p++; + + if(cp >= 0x80) { + unsigned extra = 0; + unsigned i; + + if((cp & 0xE0) == 0xC0) { + cp &= 0x1F; + extra = 1; + } else if((cp & 0xF0) == 0xE0) { + cp &= 0x0F; + extra = 2; + } else if((cp & 0xF8) == 0xF0) { + cp &= 0x07; + extra = 3; + } else { + cp = '?'; + extra = 0; + } + for(i = 0; i < extra; i++) { + if((*p & 0xC0) != 0x80) { + cp = '?'; + break; + } + cp = (cp << 6) | (unsigned long)(*p++ & 0x3F); + } + } + if(cp >= 0x10000) { + cp -= 0x10000; + dst[out++] = (wchar_t)(0xD800 | (cp >> 10)); + dst[out++] = (wchar_t)(0xDC00 | (cp & 0x3FF)); + } else { + dst[out++] = (wchar_t)cp; + } + } + dst[out] = 0; +} + +/* Defined further down; the *at() shims above call them. */ +static int wfm_mkdir1(const char *path); +static int wfm_unlink(const char *path); +static int wfm_rmdir(const char *path); +static int wfm_stat(const char *path, struct stat *st); +static int wfm_rename(const char *from, const char *to); + +static int __attribute__((unused)) +wfm_open(const char *path, int flags, ...) { + int mode = 0; + int fd; + + if(flags & O_CREAT) { + va_list ap; + + va_start(ap, flags); + mode = va_arg(ap, int); + va_end(ap); + } + /* Directory opens become synthetic fds so openat/mkdirat can resolve them + to paths; _open() returns EACCES for directories on Windows. */ + if(flags & O_DIRECTORY) { + static int next_dirfd = 0x10000; + + fd = next_dirfd++; + wfm_fd_set(fd, path); + return fd; + } + /* Force O_BINARY: MinGW's _open defaults to text mode, which would + translate LF -> CRLF on write and corrupt binary payloads. The wide + call keeps a non-ASCII name intact (see wfm_wide). */ + { + wchar_t wide[PATH_MAX]; + + wfm_wide(path, wide, PATH_MAX); + fd = _wopen(wide, (flags & ~(O_NOFOLLOW | O_DIRECTORY | O_CLOEXEC)) | + O_BINARY, + mode); + } + if(fd >= 0) { + wfm_fd_set(fd, path); + } + return fd; +} + +static int __attribute__((unused)) +wfm_openat(int dirfd, const char *path, int flags, ...) { + char full[PATH_MAX]; + int mode = 0; + + if(wfm_join(dirfd, path, full, sizeof(full))) { + return -1; + } + if(flags & O_CREAT) { + va_list ap; + + va_start(ap, flags); + mode = va_arg(ap, int); + va_end(ap); + } + return wfm_open(full, flags, mode); +} + +static int __attribute__((unused)) +wfm_mkdirat(int dirfd, const char *path, mode_t mode) { + char full[PATH_MAX]; + + (void)mode; + if(wfm_join(dirfd, path, full, sizeof(full))) { + return -1; + } + return wfm_mkdir1(full); +} + +static int __attribute__((unused)) +wfm_renameat(int from_fd, const char *from, int to_fd, const char *to) { + char src[PATH_MAX]; + char dst[PATH_MAX]; + + if(wfm_join(from_fd, from, src, sizeof(src)) || + wfm_join(to_fd, to, dst, sizeof(dst))) { + return -1; + } + return wfm_rename(src, dst); +} + +static int __attribute__((unused)) +wfm_rename(const char *from, const char *to) { + wchar_t wfrom[PATH_MAX]; + wchar_t wto[PATH_MAX]; + + wfm_wide(from, wfrom, PATH_MAX); + wfm_wide(to, wto, PATH_MAX); + /* Windows rename() refuses to replace an existing file, unlike POSIX. */ + if(_waccess(wto, 0) == 0) { + if(_wremove(wto)) { + return -1; + } + } + return _wrename(wfrom, wto); +} + +static int __attribute__((unused)) +wfm_unlinkat(int dirfd, const char *path, int flags) { + char full[PATH_MAX]; + + if(wfm_join(dirfd, path, full, sizeof(full))) { + return -1; + } + return (flags & AT_REMOVEDIR) ? wfm_rmdir(full) : wfm_unlink(full); +} + +static int __attribute__((unused)) +wfm_mkdir1(const char *path) { + wchar_t wide[PATH_MAX]; + + wfm_wide(path, wide, PATH_MAX); + return _wmkdir(wide); /* MinGW's mkdir() takes a single argument. */ +} + +static int __attribute__((unused)) +wfm_unlink(const char *path) { + wchar_t wide[PATH_MAX]; + + wfm_wide(path, wide, PATH_MAX); + return _wunlink(wide); +} + +static int __attribute__((unused)) +wfm_rmdir(const char *path) { + wchar_t wide[PATH_MAX]; + + wfm_wide(path, wide, PATH_MAX); + return _wrmdir(wide); +} + +/* _wstati64 fills its own struct; the engines only ever read st_mode, st_size + and st_mtime, so copying those across is safe and avoids depending on how + this toolchain happens to alias `struct stat`. */ +static int __attribute__((unused)) +wfm_stat(const char *path, struct stat *st) { + wchar_t wide[PATH_MAX]; + struct _stati64 wst; + + wfm_wide(path, wide, PATH_MAX); + if(_wstati64(wide, &wst)) { + return -1; + } + memset(st, 0, sizeof(*st)); + st->st_mode = (mode_t)wst.st_mode; + st->st_size = (off_t)wst.st_size; + st->st_mtime = (time_t)wst.st_mtime; + return 0; +} + +static int __attribute__((unused)) +wfm_fstatat(int dirfd, const char *path, struct stat *st, int flags) { + char full[PATH_MAX]; + + (void)flags; + if(wfm_join(dirfd, path, full, sizeof(full))) { + return -1; + } + return wfm_stat(full, st); +} + +static int __attribute__((unused)) +wfm_fsync(int fd) { + return _commit(fd); +} + +static int __attribute__((unused)) +wfm_fchmod(int fd, mode_t mode) { + (void)fd; + (void)mode; + return 0; /* Windows has no Unix modes; the engine ignores this failure. */ +} + +/* MinGW has no utimes(); _utime() is the same thing with second precision, + which is all the 7z engine asks for (it feeds the extractor both fields). */ +static int __attribute__((unused)) +wfm_utimes(const char *path, const struct timeval tv[2]) { + struct _utimbuf ut; + + ut.actime = tv[0].tv_sec; + ut.modtime = tv[1].tv_sec; + return _utime(path, &ut); +} + +static int __attribute__((unused)) +wfm_close(int fd) { + wfm_fd_clear(fd); + if(fd >= 0x10000) { + return 0; /* synthetic dirfd, nothing to close */ + } + return _close(fd); +} + +/* MinGW's fopen() decodes the path in the system code page, the same way + stat() does. Redirect to _wfopen so a UTF-8 path goes through the wide + API and round-trips back to the on-disk name regardless of the host's + ACP. The translation units that include this shim may not call fopen() + themselves, so mark the wrapper as unused to keep -Werror quiet. */ +__attribute__((unused)) +static FILE *wfm_fopen(const char *path, const char *mode) { + wchar_t wide_path[PATH_MAX]; + wchar_t wide_mode[16]; + size_t i; + + wfm_wide(path, wide_path, PATH_MAX); + for(i = 0; i + 1 < sizeof(wide_mode) && mode[i]; i++) { + wide_mode[i] = (wchar_t)(unsigned char)mode[i]; + } + wide_mode[i] = 0; + return _wfopen(wide_path, wide_mode); +} + +#define fopen(p, m) wfm_fopen(p, m) + +#define mkdir(p, ...) wfm_mkdir1(p) +#define rmdir(p) wfm_rmdir(p) +#define unlink(p) wfm_unlink(p) +#define open(...) wfm_open(__VA_ARGS__) +#define openat(...) wfm_openat(__VA_ARGS__) +#define mkdirat(d, p, m) wfm_mkdirat(d, p, m) +#define rename(a, b) wfm_rename(a, b) +#define renameat(sd, sp, dd, dp) wfm_renameat(sd, sp, dd, dp) +#define unlinkat(d, p, f) wfm_unlinkat(d, p, f) +#define fstatat(d, p, s, f) wfm_fstatat(d, p, s, f) +#define fsync(fd) wfm_fsync(fd) +#define fchmod(fd, mode) wfm_fchmod(fd, mode) +#define close(fd) wfm_close(fd) +/* Direct lstat/stat to the wide-path shim so non-ASCII archive entries survive + a CP936 or CP1252 host. MinGW's stat() defaults to the ANSI entry point + and silently truncates names it cannot represent. */ +#define lstat(p, s) wfm_stat(p, s) +#define stat(p, s) wfm_stat(p, s) +#define utimes(p, tv) wfm_utimes(p, tv) +/* opendir/readdir/closedir go through the wide variants so the names we get + back are real UTF-8; otherwise MinGW hands us whatever the system code page + made of the filename, which round-trips through a non-ASCII UTF-8 entry as + a garbage string that no later wfm_stat() call can resolve. */ +typedef struct { + _WDIR *wd; + struct dirent de; +} WFM_DIR; + +static DIR * __attribute__((unused)) +wfm_opendir(const char *path) { + wchar_t wide[PATH_MAX]; + WFM_DIR *wfm; + + wfm_wide(path, wide, PATH_MAX); + wfm = (WFM_DIR *)malloc(sizeof(*wfm)); + if(!wfm) { + return NULL; + } + wfm->wd = _wopendir(wide); + if(!wfm->wd) { + free(wfm); + return NULL; + } + return (DIR *)wfm; +} + +static struct dirent * __attribute__((unused)) +wfm_readdir(DIR *d) { + WFM_DIR *wfm = (WFM_DIR *)d; + struct _wdirent *we; + + if(!wfm) { + return NULL; + } + we = _wreaddir(wfm->wd); + if(!we) { + return NULL; + } + WideCharToMultiByte(CP_UTF8, 0, we->d_name, -1, wfm->de.d_name, + sizeof(wfm->de.d_name), NULL, NULL); + wfm->de.d_ino = we->d_ino; + wfm->de.d_reclen = (unsigned short)strlen(wfm->de.d_name); + return &wfm->de; +} + +static int __attribute__((unused)) +wfm_closedir(DIR *d) { + WFM_DIR *wfm = (WFM_DIR *)d; + + if(!wfm) { + return -1; + } + _wclosedir(wfm->wd); + free(wfm); + return 0; +} + +#define opendir(p) wfm_opendir(p) +#define readdir(d) wfm_readdir(d) +#define closedir(d) wfm_closedir(d) + +#endif /* _WIN32 */ + +#endif /* WFM_TEST_POSIX_COMPAT_H */ diff --git a/tests/run-sevenz-tests.sh b/tests/run-sevenz-tests.sh index 8dcd55d..bd3f56b 100644 --- a/tests/run-sevenz-tests.sh +++ b/tests/run-sevenz-tests.sh @@ -18,6 +18,7 @@ ROOT="$(cd "$(dirname "$0")/.." && pwd -W 2>/dev/null || pwd)" BUILD="$ROOT/.build/sevenz-test" SEVENZ_DIR="$ROOT/third_party/7z" FIXTURES="$ROOT/tests/fixtures-7z" +COMPAT_INC="$ROOT/tests/compat" PYTHON="${PYTHON:-python3}" CC="${CC:-gcc}" @@ -58,8 +59,21 @@ done "$CC" -c -O2 -Wall -Wextra -Werror -D_FILE_OFFSET_BITS=64 -D_LARGEFILE_SOURCE \ -I"$ROOT/src" -o "$BUILD/zipx_volume.o" "$ROOT/src/zipx_volume.c" +# The extraction facade is engine code too, so it gets the host POSIX shim as +# well as the same strictness (see tests/run-tests.sh for the ZIP/RAR pair). +# zipx_common carries the limit profiles and the status text that all three +# engines share. +"$CC" -c -O2 -Wall -Wextra -Werror -I"$ROOT/src" -o "$BUILD/zipx_common.o" \ + "$ROOT/src/zipx_common.c" +"$CC" -c -O2 -Wall -Wextra -Werror -D_FILE_OFFSET_BITS=64 -D_LARGEFILE_SOURCE \ + -I"$SEVENZ_DIR" -I"$ROOT/src" -I"$COMPAT_INC" \ + -include "$ROOT/tests/posix_compat.h" \ + -o "$BUILD/sevenz_extract.o" "$ROOT/src/sevenz_extract.c" + ENGINE_OBJS=("$BUILD/sevenz_chain.o" "$BUILD/sevenz_volstream.o" - "$BUILD/zipx_volume.o") + "$BUILD/zipx_volume.o" "$BUILD/zipx_common.o") + +FACADE_OBJS=("$BUILD/sevenz_extract.o" "${ENGINE_OBJS[@]}") # unrar-style extra libs are only needed by the Windows path of 7zFile.c. EXTRA_LIBS=() @@ -71,6 +85,13 @@ esac "$ROOT/tests/sevenz_chain_e2e.c" "${ENGINE_OBJS[@]}" "${VENDOR_OBJS[@]}" \ "${EXTRA_LIBS[@]}" +# The facade driver: the same strict flags as the engine, plus the POSIX shim, +# because a MinGW host has neither statvfs() nor a two-argument mkdir(). +"$CC" -O2 -Wall -Wextra -I"$SEVENZ_DIR" -I"$ROOT/src" -I"$COMPAT_INC" \ + -include "$ROOT/tests/posix_compat.h" \ + -o "$BUILD/test_sevenz_extract" "$ROOT/tests/test_sevenz_extract.c" \ + "${FACADE_OBJS[@]}" "${VENDOR_OBJS[@]}" "${EXTRA_LIBS[@]}" + # The SDK-baseline driver is kept buildable: it is the fastest way to tell an # engine bug from an SDK one when a fixture starts failing. "$CC" -O2 -w -I"$SEVENZ_DIR" -o "$BUILD/sevenz_e2e" \ @@ -146,6 +167,63 @@ if [ -f "$FIXTURES/vol.7z.001" ]; then run_case "vol.7z.001" "$FIXTURES/vol.7z.001" fi +# ------------------------------------------------- extraction facade +# The same fixtures again, but through src/sevenz_extract.c: staging, publish, +# limits, conflict policy and name validation all have to agree with the +# decoder before the format is wired into the server. +echo +echo "== 7z extraction facade (engine: src/sevenz_extract.c) ==" +mkdir -p "$BUILD/fx" +for a in store lzma2 lzma ppmd bcj delta utf8 bcj2 solidoff bcj2off aes; do + [ -f "$FIXTURES/$a.7z" ] || continue + out="$(mktemp -d "$BUILD/fx/XXXXXX")" || continue + target="$out/$a" + mkdir -p "$target" + rc=0 + case "$a" in + aes) "$BUILD/test_sevenz_extract" "$FIXTURES/$a.7z" "$target" \ + "$FIXTURE_PASSWORD" >"$out.log" 2>&1 || rc=$? ;; + *) "$BUILD/test_sevenz_extract" "$FIXTURES/$a.7z" "$target" \ + >"$out.log" 2>&1 || rc=$? ;; + esac + if [ "$rc" -eq 0 ] && diff -r "$FIXTURES/_src" "$target/_src" >/dev/null 2>&1; then + printf ' %-12s ok\n' "$a" + pass=$((pass + 1)) + else + printf ' %-12s FAIL\n' "$a" + sed -n '1,20p' "$out.log" | sed 's/^/ /' + fail=$((fail + 1)) + fi +done + +# A byte-split set goes through the same facade. +if [ -f "$FIXTURES/vol.7z.001" ]; then + out="$(mktemp -d "$BUILD/fx/XXXXXX")" || out= + if [ -n "$out" ]; then + rc=0 + "$BUILD/test_sevenz_extract" "$FIXTURES/vol.7z.001" "$out/vol" \ + >"$out.log" 2>&1 || rc=$? + if [ "$rc" -eq 0 ] && diff -r "$FIXTURES/_src" "$out/vol/_src" >/dev/null 2>&1; then + printf ' %-12s ok\n' "vol.7z.001" + pass=$((pass + 1)) + else + printf ' %-12s FAIL\n' "vol.7z.001" + sed -n '1,20p' "$out.log" | sed 's/^/ /' + fail=$((fail + 1)) + fi + fi +fi + +echo +echo "== 7z error and policy paths ==" +mkdir -p "$BUILD/cases" +cases_work="$(mktemp -d "$BUILD/cases/XXXXXX")" +if "$BUILD/test_sevenz_extract" --cases "$FIXTURES" "$cases_work"; then + pass=$((pass + 1)) +else + fail=$((fail + 1)) +fi + echo echo "$pass passed, $fail failed" [ "$fail" -eq 0 ] diff --git a/tests/run-tests.sh b/tests/run-tests.sh index f689b64..d18624e 100644 --- a/tests/run-tests.sh +++ b/tests/run-tests.sh @@ -67,6 +67,11 @@ COMPAT_INC="$ROOT/tests/compat" -include "$ROOT/tests/posix_compat.h" \ -o "$BUILD/zip_extract.o" "$ROOT/src/zip_extract.c" +# Format-independent helpers (limits profiles + status string) live here. +"$CC" -c -O2 -Wall -Wextra -Wno-unused-parameter -I"$ROOT/src" \ + -I"$COMPAT_INC" -include "$ROOT/tests/posix_compat.h" \ + -o "$BUILD/zipx_common.o" "$ROOT/src/zipx_common.c" + # Volume support: the concatenating stream and the volume set detector. "$CC" -c -O2 -Wall -Wextra -Wno-unused-parameter \ -I"$ROOT/third_party/minizip-ng/include" -I"$ROOT/src" -I"$COMPAT_INC" \ @@ -94,14 +99,14 @@ objs=() rar_objs=() for obj in "$BUILD"/*.o; do case "$obj" in - */zip_extract.o|*/rar_extract.o|*/test_zip_extract.o|*/test_rar_extract.o) continue ;; + */zip_extract.o|*/zipx_common.o|*/rar_extract.o|*/test_zip_extract.o|*/test_rar_extract.o) continue ;; */unrar7_*.o) rar_objs+=("$obj"); continue ;; esac objs+=("$obj") done "$CC" -O2 -o "$BUILD/test-zip-extract" \ - "$BUILD/zip_extract.o" "$BUILD/test_zip_extract.o" "${objs[@]}" + "$BUILD/zip_extract.o" "$BUILD/zipx_common.o" "$BUILD/test_zip_extract.o" "${objs[@]}" # The RAR test links the unrar7 objects, so it needs the C++ driver. # Windows unrar system.cpp references SetSuspendState (PowrProf). @@ -109,7 +114,7 @@ RAR_LIBS=() [ "$HOST_KIND" = windows ] && RAR_LIBS=(-lpowrprof) "$CXX" -O2 -o "$BUILD/test-rar-extract" \ "$BUILD/rar_extract.o" "$BUILD/test_rar_extract.o" \ - "$BUILD/zip_extract.o" "${objs[@]}" "${rar_objs[@]}" "${RAR_LIBS[@]}" + "$BUILD/zip_extract.o" "$BUILD/zipx_common.o" "${objs[@]}" "${rar_objs[@]}" "${RAR_LIBS[@]}" "$BUILD/test-zip-extract" "$ROOT/tests/fixtures" "$BUILD/work-zip" # Real RAR fixtures (v6 / multi-volume / encrypted) live in fixtures-real/, diff --git a/tests/test_sevenz_extract.c b/tests/test_sevenz_extract.c new file mode 100644 index 0000000..8c32ec4 --- /dev/null +++ b/tests/test_sevenz_extract.c @@ -0,0 +1,216 @@ +/* + * Test driver for the 7z extraction facade (src/sevenz_extract.c). + * + * test_sevenz_extract [password] + * Extract one archive. Exit status 0 means ZIPX_OK; the shell compares + * the result against the fixture's source tree byte for byte. + * + * test_sevenz_extract --cases + * Exercise the error and policy paths that need no byte comparison: + * a missing/wrong password, an encrypted header, conflicts under each + * policy, cancellation, limits, a missing destination parent, and the + * guarantee that nothing is published and no staging tree survives a + * failure. + * + * The host build injects tests/posix_compat.h (see run-sevenz-tests.sh), so + * the engine can stay plain POSIX. + */ + +#include +#include +#include +#include +#include + +#include "sevenz_extract.h" + +#define PASSWORD "Secret123" +#define PATH_MAX_LOCAL 4096 + +static int g_checks = 0; +static int g_failures = 0; + +static void +check(int ok, const char *what) { + g_checks++; + if(!ok) { + g_failures++; + printf(" FAIL %s\n", what); + } +} + +static int +cancel_always(void *userdata) { + (void)userdata; + return 1; +} + +static int +has_staging_leftover(const char *dir) { + DIR *d = opendir(dir); + struct dirent *ent; + int found = 0; + + if(!d) { + return 0; + } + while((ent = readdir(d))) { + if(!strncmp(ent->d_name, ".wfm-extract-", 13)) { + found = 1; + break; + } + } + closedir(d); + return found; +} + +static int +extract_one(const char *archive, const char *dst, const char *password) { + zipx_result_t r; + zipx_status_t st = sevenz_extract(archive, dst, ZIPX_CONFLICT_FAIL, + zipx_default_limits(), NULL, NULL, NULL, + password, &r); + + if(st != ZIPX_OK) { + fprintf(stderr, " %s: %s: %s\n", archive, zipx_status_string(st), + r.message); + return 1; + } + return 0; +} + +/* Runs a case that is expected to fail, and checks the status and message. */ +static void +expect_fail(const char *label, const char *archive, const char *dst, + const char *password, zipx_conflict_t conflict, + const zipx_limits_t *limits, zipx_cancel_fn cancel, + zipx_status_t want, const char *needle) { + zipx_result_t r; + zipx_status_t st; + char buf[256]; + + st = sevenz_extract(archive, dst, conflict, limits, cancel, NULL, NULL, + password, &r); + snprintf(buf, sizeof(buf), "%s: status is %s, not %s", label, + zipx_status_string(st), zipx_status_string(want)); + check(st == want, buf); + if(needle) { + snprintf(buf, sizeof(buf), "%s: message mentions \"%s\" (got \"%s\")", + label, needle, r.message); + check(strstr(r.message, needle) != NULL, buf); + } +} + +/* Extracts store.7z into `dst` under the given policy, expecting `want`. */ +static void +policy_case(const char *label, const char *archive, const char *dst, + zipx_conflict_t conflict, int run, zipx_status_t want) { + zipx_result_t r; + zipx_status_t st; + char buf[256]; + + st = sevenz_extract(archive, dst, conflict, zipx_default_limits(), NULL, NULL, + NULL, NULL, &r); + snprintf(buf, sizeof(buf), "%s (run %d): status is %s, not %s", label, run, + zipx_status_string(st), zipx_status_string(want)); + check(st == want, buf); +} + +static int +run_cases(const char *fx, const char *work) { + char arc[PATH_MAX_LOCAL]; + char dst[PATH_MAX_LOCAL]; + zipx_limits_t tight; + + mkdir(work, 0777); + + /* --- passwords ----------------------------------------------------- */ + snprintf(arc, sizeof(arc), "%s/aes.7z", fx); + snprintf(dst, sizeof(dst), "%s/pw-missing", work); + mkdir(dst, 0777); + expect_fail("aes with no password", arc, dst, NULL, ZIPX_CONFLICT_FAIL, + zipx_default_limits(), NULL, ZIPX_ERR_PASSWORD, "encrypted"); + + snprintf(dst, sizeof(dst), "%s/pw-wrong", work); + mkdir(dst, 0777); + expect_fail("aes with the wrong password", arc, dst, "NotThePassword", + ZIPX_CONFLICT_FAIL, zipx_default_limits(), NULL, + ZIPX_ERR_PASSWORD, "7zAES"); + check(!has_staging_leftover(work), "no staging tree survives a wrong password"); + + snprintf(dst, sizeof(dst), "%s/pw-ok", work); + mkdir(dst, 0777); + check(extract_one(arc, dst, PASSWORD) == 0, + "aes extracts with the right password"); + + /* --- encrypted header ---------------------------------------------- */ + snprintf(arc, sizeof(arc), "%s/aeshe.7z", fx); + snprintf(dst, sizeof(dst), "%s/he", work); + mkdir(dst, 0777); + expect_fail("aeshe (encrypted header)", arc, dst, PASSWORD, ZIPX_CONFLICT_FAIL, + zipx_default_limits(), NULL, ZIPX_ERR_UNSUPPORTED, "-mhe=on"); + + /* --- open failures -------------------------------------------------- */ + snprintf(dst, sizeof(dst), "%s/missing-file", work); + mkdir(dst, 0777); + expect_fail("a missing archive", "/no/such/archive.7z", dst, NULL, + ZIPX_CONFLICT_FAIL, zipx_default_limits(), NULL, ZIPX_ERR_OPEN, + "cannot open"); + + snprintf(arc, sizeof(arc), "%s/store.7z", fx); + snprintf(dst, sizeof(dst), "%s/no-parent/deeper", work); + expect_fail("a destination whose parent is missing", arc, dst, NULL, + ZIPX_CONFLICT_FAIL, zipx_default_limits(), NULL, ZIPX_ERR_IO, + "destination parent is missing"); + + /* --- conflict policies ---------------------------------------------- */ + snprintf(dst, sizeof(dst), "%s/policy-fail", work); + mkdir(dst, 0777); + policy_case("fail", arc, dst, ZIPX_CONFLICT_FAIL, 1, ZIPX_OK); + policy_case("fail", arc, dst, ZIPX_CONFLICT_FAIL, 2, ZIPX_ERR_CONFLICT); + + snprintf(dst, sizeof(dst), "%s/policy-overwrite", work); + mkdir(dst, 0777); + policy_case("overwrite", arc, dst, ZIPX_CONFLICT_OVERWRITE, 1, ZIPX_OK); + policy_case("overwrite", arc, dst, ZIPX_CONFLICT_OVERWRITE, 2, ZIPX_OK); + + snprintf(dst, sizeof(dst), "%s/policy-merge", work); + mkdir(dst, 0777); + policy_case("merge", arc, dst, ZIPX_CONFLICT_MERGE, 1, ZIPX_OK); + policy_case("merge", arc, dst, ZIPX_CONFLICT_MERGE, 2, ZIPX_OK); + + /* --- cancellation --------------------------------------------------- */ + snprintf(dst, sizeof(dst), "%s/cancel", work); + mkdir(dst, 0777); + expect_fail("a canceled extraction", arc, dst, NULL, ZIPX_CONFLICT_FAIL, + zipx_default_limits(), cancel_always, ZIPX_ERR_CANCELED, NULL); + check(!has_staging_leftover(work), "no staging tree survives a cancellation"); + + /* --- limits --------------------------------------------------------- */ + tight = *zipx_default_limits(); + tight.max_entries = 1; + snprintf(dst, sizeof(dst), "%s/limits", work); + mkdir(dst, 0777); + expect_fail("an archive over the entry limit", arc, dst, NULL, + ZIPX_CONFLICT_FAIL, &tight, NULL, ZIPX_ERR_LIMIT_ENTRIES, + "more than 1 entries"); + + printf(" cases: %d checks, %d failures\n", g_checks, g_failures); + return g_failures == 0 ? 0 : 1; +} + +int +main(int argc, char **argv) { + setvbuf(stdout, NULL, _IONBF, 0); + if(argc >= 4 && !strcmp(argv[1], "--cases")) { + return run_cases(argv[2], argv[3]); + } + if(argc < 3) { + fprintf(stderr, + "usage: %s [password]\n" + " %s --cases \n", + argv[0], argv[0]); + return 2; + } + return extract_one(argv[1], argv[2], argc > 3 ? argv[3] : NULL); +}