Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
ba668ade50 | ||
|
|
3f80eb4692 | ||
|
|
807d129d8d | ||
|
|
70eaa1027d | ||
|
|
575f94cd39 | ||
|
|
2059c0e4b5 | ||
|
|
9a36c3cb30 | ||
|
|
5e8b56f23e | ||
|
|
5cc493d152 | ||
|
|
8766178aa1 | ||
|
|
fd48232b6c | ||
|
|
49636b0b8a | ||
|
|
f1633321d2 | ||
|
|
9058d5a97e | ||
|
|
5d88914674 | ||
|
|
0d036a74a7 | ||
|
|
1fa2f0953f | ||
|
|
f4fd464353 | ||
|
|
5229cd59df | ||
|
|
a54f34bcab | ||
|
|
98679423e0 | ||
|
|
00b750d80b | ||
|
|
112f8a6b72 | ||
|
|
27eec8c399 | ||
|
|
200b38604f | ||
|
|
021c9cb339 | ||
|
|
4da345a6e8 | ||
|
|
9b2f5a07c3 | ||
|
|
2748b383bd | ||
|
|
8bee84bd49 | ||
|
|
da565ccb7c | ||
|
|
01e27f3825 | ||
|
|
e0bc4a6ae0 | ||
|
|
f820016de3 | ||
|
|
95578fb2d2 | ||
|
|
4695295b8e | ||
|
|
76c4eadbf5 | ||
|
|
ad765776f7 | ||
|
|
ae41434089 | ||
|
|
af6b440f33 | ||
|
|
01a6616472 | ||
|
|
068983b062 | ||
|
|
da67bfa284 | ||
|
|
452b9b0187 | ||
|
|
91757bfd02 | ||
|
|
30d0decf58 | ||
|
|
22a4b5c76a | ||
|
|
ff010add68 | ||
|
|
68056f2d9f | ||
|
|
3db921e408 | ||
|
|
c58c145733 | ||
|
|
a0482a41ed | ||
|
|
c68c0def35 | ||
|
|
8344b9bae0 | ||
|
|
cb82ee439d | ||
|
|
9e9830a214 | ||
|
|
687ef6b297 | ||
|
|
36ea055e70 | ||
|
|
2a346d694c | ||
|
|
bf55a4e4fd | ||
|
|
848b774333 | ||
|
|
5af5acf4ab | ||
|
|
96286cb9f2 | ||
|
|
50eb1734c1 | ||
|
|
0d24359d80 | ||
|
|
bfe522e56b | ||
|
|
c7d9e965b8 | ||
|
|
3f516b51ad | ||
|
|
aef4a44ab3 | ||
|
|
5cb0b76e7a |
No files matched your search
@@ -0,0 +1,168 @@
|
||||
#!/usr/bin/env bash
|
||||
# ===========================================================================
|
||||
# ps5-web-file-manager -- 一键在 WSL 里生成 PS5 ELF
|
||||
#
|
||||
# 这个脚本在 WSL (Ubuntu-22.04) 里执行,做四件事:
|
||||
# 1. 把 Windows 仓库的源码 rsync 到 WSL 项目目录(增量,跳过构建缓存)
|
||||
# 2. make all (PS5_PAYLOAD_SDK = /opt/ps5-payload-sdk)
|
||||
# 3. 验证产物:size / sha256 / e_machine
|
||||
# 4. 把 ELF 拷回 Windows 项目根
|
||||
#
|
||||
# 从 Windows 的 Git Bash / MinGW64 bash 里这样跑:
|
||||
# wsl.exe -d Ubuntu-22.04 -- bash < \
|
||||
# "/c/Users/songl/Desktop/Web File Manager/ps5-web-file-manager/.build/build-elf-wsl.sh"
|
||||
#
|
||||
# 注意:必须用 stdin 重定向 `bash < script`,不要 `bash -c '...'` ——
|
||||
# 路径含空格时 -c 的参数会被 wsl.exe 拆断。
|
||||
# ===========================================================================
|
||||
|
||||
set -uo pipefail
|
||||
|
||||
# ---------------------------------------------------------------- 配置 -----
|
||||
SRC_WIN='/mnt/c/Users/songl/Desktop/Web File Manager/ps5-web-file-manager'
|
||||
PROJ='/home/song/ps5-web-file-manager'
|
||||
SDK='/opt/ps5-payload-sdk'
|
||||
|
||||
log() { printf '\033[1;36m%s\033[0m\n' "$*"; }
|
||||
warn() { printf '\033[1;33m[WARN] %s\033[0m\n' "$*"; }
|
||||
fail() { printf '\033[1;31m[FAIL] %s\033[0m\n' "$*"; exit 1; }
|
||||
|
||||
# ------------------------------------------------------------ jwasm --------
|
||||
# The LZMA decoder has an asm implementation that is 26% faster than the C one
|
||||
# (see docs/EXTRACTION-PERF.md). It is MASM syntax, so a MASM-compatible
|
||||
# assembler is needed. Makefile only enables the optimisation when jwasm is on
|
||||
# PATH, so a failure here downgrades rather than breaks the build.
|
||||
JWASM_HOME="$HOME/.cache/wfm-jwasm"
|
||||
JWASM_BIN="$JWASM_HOME/jwasm"
|
||||
|
||||
ensure_jwasm() {
|
||||
if command -v jwasm >/dev/null 2>&1; then
|
||||
echo " jwasm: $(command -v jwasm)"
|
||||
return 0
|
||||
fi
|
||||
if [ -x "$JWASM_BIN" ]; then
|
||||
export PATH="$JWASM_HOME:$PATH"
|
||||
echo " jwasm: $JWASM_BIN (缓存)"
|
||||
return 0
|
||||
fi
|
||||
|
||||
echo " jwasm 不在,正在从源码编译(首次约 30 秒)..."
|
||||
mkdir -p "$JWASM_HOME" || return 1
|
||||
if [ ! -d "$JWASM_HOME/src" ]; then
|
||||
git clone --depth 1 https://github.com/Baron-von-Riedesel/JWasm.git \
|
||||
"$JWASM_HOME/src" >/dev/null 2>&1 || return 1
|
||||
fi
|
||||
make -C "$JWASM_HOME/src" -f GccUnix.mak -j4 >/dev/null 2>&1 || return 1
|
||||
cp -f "$JWASM_HOME/src/build/GccUnixR/jwasm" "$JWASM_BIN" || return 1
|
||||
chmod +x "$JWASM_BIN" || return 1
|
||||
export PATH="$JWASM_HOME:$PATH"
|
||||
echo " jwasm: $JWASM_BIN (已编译)"
|
||||
}
|
||||
|
||||
# ------------------------------------------------------- 1/5 环境检查 -----
|
||||
log "[1/5] 环境检查"
|
||||
|
||||
[ -d "$SRC_WIN" ] || fail "Windows 源码目录不可见: $SRC_WIN (/mnt/c 挂载了吗?)"
|
||||
[ -x "$SDK/bin/prospero-clang" ] || fail "SDK 缺失: $SDK/bin/prospero-clang"
|
||||
|
||||
export PS5_PAYLOAD_SDK="$SDK"
|
||||
# shellcheck disable=SC1091
|
||||
source "$SDK/toolchain/prospero.sh" 2>/dev/null || true
|
||||
export CC="$SDK/bin/prospero-clang"
|
||||
export CXX="$SDK/bin/prospero-clang++"
|
||||
export PKG_CONFIG="$SDK/bin/prospero-pkg-config"
|
||||
|
||||
"$CC" --version | head -1
|
||||
"$PKG_CONFIG" --modversion libmicrohttpd 2>/dev/null \
|
||||
|| fail "libmicrohttpd 未装到 sysroot —— 先跑一次完整的 build-elf.sh v4"
|
||||
|
||||
ensure_jwasm || warn "jwasm 不可用 —— 将退回纯 C 解码器(约慢 26%)"
|
||||
echo
|
||||
|
||||
# ------------------------------------------------------- 2/5 同步源码 -----
|
||||
log "[2/5] 同步源码 Windows -> WSL (rsync 增量)"
|
||||
mkdir -p "$PROJ"
|
||||
|
||||
rsync -a --delete \
|
||||
--exclude='/ps5-obj' --exclude='/linux-obj' \
|
||||
--exclude='/web-file-mgr-*.elf' --exclude='/web-file-mgr-linux*' \
|
||||
--exclude='/gen' --exclude='/.build' --exclude='/tests' \
|
||||
--exclude='/docs' --exclude='/HANDOVER.md' \
|
||||
--exclude='/README.md' --exclude='/CHANGELOG.md' \
|
||||
--exclude='/erssongl*' \
|
||||
--include='/src' --include='/assets' \
|
||||
--include='/third_party' --include='/Makefile' \
|
||||
--include='/gen-asset-module.py' --include='/.gitignore' \
|
||||
--exclude='/*' \
|
||||
"$SRC_WIN/" "$PROJ/" || fail "rsync 失败"
|
||||
|
||||
# rsync 的 --include 只放行目录本身,这几个顶层文件再单独 cp 一次
|
||||
for f in Makefile gen-asset-module.py .gitignore; do
|
||||
[ -f "$SRC_WIN/$f" ] && cp -f "$SRC_WIN/$f" "$PROJ/$f"
|
||||
done
|
||||
|
||||
# 冒烟:今天的关键文件都在不在
|
||||
for f in src/sevenz_extract.c src/zipx_common.c src/sevenz_volstream.c \
|
||||
src/sevenz_chain.c Makefile gen-asset-module.py; do
|
||||
[ -f "$PROJ/$f" ] || fail "同步后缺失: $PROJ/$f"
|
||||
done
|
||||
[ -d "$PROJ/third_party/7z" ] || fail "同步后缺失: $PROJ/third_party/7z"
|
||||
|
||||
# 输出文件名由 Makefile 的 VERSION_TAG 决定(web-file-mgr-<ver>.elf)。
|
||||
# 从 Makefile 里读,别在脚本里硬编 —— 否则改了版本号脚本还在找旧名字。
|
||||
VERSION=$(sed -n 's/^VERSION_TAG *[?:]*= *//p' "$PROJ/Makefile" | head -1)
|
||||
[ -n "$VERSION" ] || fail "读不到 VERSION_TAG: $PROJ/Makefile"
|
||||
BIN_NAME="web-file-mgr-${VERSION}.elf"
|
||||
ELF="$PROJ/$BIN_NAME"
|
||||
|
||||
echo " src/ + assets/ + third_party/ + Makefile OK"
|
||||
echo " 版本: $VERSION -> 输出: $BIN_NAME"
|
||||
echo
|
||||
|
||||
# ---------------------------------------------------------- 3/5 编译 ------
|
||||
log "[3/5] make all"
|
||||
cd "$PROJ" || fail "cd $PROJ"
|
||||
|
||||
make all 2>&1 | tail -120
|
||||
# make 的退出码被管道吃了,用 PIPESTATUS 取回来
|
||||
if [ "${PIPESTATUS[0]}" -ne 0 ]; then
|
||||
fail "make all 失败(详见上方输出)"
|
||||
fi
|
||||
echo
|
||||
|
||||
# ---------------------------------------------------------- 4/5 验证 ------
|
||||
log "[4/5] 验证产物"
|
||||
[ -f "$ELF" ] || fail "ELF 未生成: $ELF"
|
||||
|
||||
SIZE=$(stat -c%s "$ELF")
|
||||
HASH=$(sha256sum "$ELF" | cut -d' ' -f1)
|
||||
# ELF header: offset 18 起 2 字节 = e_machine。
|
||||
# `od -tx2` 按 2 字节小端解释成一个 short 后打印其**值**,所以文件里的
|
||||
# 字节序 "3e 00" 会输出成 "003e"(不是 "3e00")。别拿字节序去比对。
|
||||
EM=$(od -An -tx2 -j 18 -N 2 "$ELF" | tr -d ' \n')
|
||||
EM_NUM=$((16#$EM))
|
||||
|
||||
ls -lh "$ELF"
|
||||
echo " size: $SIZE bytes (~$((SIZE / 1024)) KiB)"
|
||||
echo " sha256: $HASH"
|
||||
echo " e_machine = 0x$EM ($EM_NUM)"
|
||||
|
||||
case "$EM_NUM" in
|
||||
62) echo " -> x86-64 / PS5 [OK]" ;;
|
||||
183) fail "e_machine=$EM_NUM (0x$EM) 是 aarch64!PS5 是 x86-64,target 三元组错了" ;;
|
||||
*) fail "e_machine=$EM_NUM (0x$EM) 非预期(期望 62 = 0x003e = x86-64)" ;;
|
||||
esac
|
||||
echo
|
||||
|
||||
# ------------------------------------------------- 5/5 拷回 Windows -------
|
||||
log "[5/5] 拷回 Windows"
|
||||
cp -f "$ELF" "$SRC_WIN/$BIN_NAME" || fail "拷回 Windows 失败"
|
||||
ls -lh "$SRC_WIN/$BIN_NAME"
|
||||
echo
|
||||
printf '\033[1;32m[DONE]\033[0m %s\n' "$SRC_WIN/$BIN_NAME"
|
||||
echo " $SIZE bytes / sha256 $HASH / e_machine 0x$EM"
|
||||
|
||||
# 顺便报告 Windows 侧现在有哪些版本化 ELF,方便挑一个拷进 U 盘
|
||||
echo
|
||||
echo " Windows 项目根现有的 ELF:"
|
||||
ls -1 "$SRC_WIN"/web-file-mgr-*.elf 2>/dev/null | sed 's#.*/##' | sed 's/^/ /' || true
|
||||
@@ -0,0 +1,74 @@
|
||||
==========================================
|
||||
PS5 Web File Manager -- ELF 构建 v4
|
||||
时间: Fri Sep 4 12:45:52 CST 2026
|
||||
PS5_PAYLOAD_SDK=/opt/ps5-payload-sdk
|
||||
user: song uid=1000
|
||||
==========================================
|
||||
[0/7] 预热 sudo (会提示输 1 次密码)...
|
||||
[1/7] SDK OK: /opt/ps5-payload-sdk
|
||||
[2/7] LLVM bindir: /usr/lib/llvm-18/bin
|
||||
[3/7] prospero-clang --version ...
|
||||
Ubuntu clang version 18.1.8 (++20240731024944+3b5b5c1ec4a3-1~exp1~20240731145000.144)
|
||||
[4/7] 同步源码 (Windows -> WSL, 增量) ...
|
||||
已同步自 /mnt/c/Users/songl/Desktop/Web File Manager/ps5-web-file-manager
|
||||
[5/7] libmicrohttpd (staging + 一次性 sudo cp)...
|
||||
下载 libmicrohttpd-1.0.1.tar.gz...
|
||||
tarball: /home/song/.cache/libmicrohttpd-1.0.1.tar.gz (2.2M)
|
||||
configure (host=x86_64-pc-freebsd)...
|
||||
make -j14...
|
||||
mv -f $depbase.Tpo $depbase.Po
|
||||
/bin/bash ../../libtool --tag=CC --mode=link /opt/ps5-payload-sdk/bin/prospero-clang -fno-strict-aliasing -O1 -w -o perf_replies perf_replies.o mhd_tool_get_cpu_count.o ../../src/microhttpd/libmicrohttpd.la
|
||||
libtool: link: /opt/ps5-payload-sdk/bin/prospero-clang -fno-strict-aliasing -O1 -w -o perf_replies perf_replies.o mhd_tool_get_cpu_count.o ../../src/microhttpd/.libs/libmicrohttpd.a -lpthread -pthread
|
||||
make[4]: Leaving directory '/tmp/tmp.1uwmf0LJPq/libmicrohttpd-1.0.1/src/tools'
|
||||
make[3]: Leaving directory '/tmp/tmp.1uwmf0LJPq/libmicrohttpd-1.0.1/src/tools'
|
||||
make[2]: Leaving directory '/tmp/tmp.1uwmf0LJPq/libmicrohttpd-1.0.1/src'
|
||||
Making all in .
|
||||
make[2]: Entering directory '/tmp/tmp.1uwmf0LJPq/libmicrohttpd-1.0.1'
|
||||
make[2]: Leaving directory '/tmp/tmp.1uwmf0LJPq/libmicrohttpd-1.0.1'
|
||||
make[1]: Leaving directory '/tmp/tmp.1uwmf0LJPq/libmicrohttpd-1.0.1'
|
||||
make install DESTDIR=/tmp/ps5-homebrew-ext4 ...
|
||||
make[2]: Nothing to be done for 'install-exec-am'.
|
||||
/usr/bin/mkdir -p '/tmp/ps5-homebrew-ext4/user/homebrew/lib/pkgconfig'
|
||||
/usr/bin/install -c -m 644 libmicrohttpd.pc '/tmp/ps5-homebrew-ext4/user/homebrew/lib/pkgconfig'
|
||||
make[2]: Leaving directory '/tmp/tmp.1uwmf0LJPq/libmicrohttpd-1.0.1'
|
||||
make[1]: Leaving directory '/tmp/tmp.1uwmf0LJPq/libmicrohttpd-1.0.1'
|
||||
sudo cp stage -> SDK homebrew...
|
||||
验证 prospero-pkg-config:
|
||||
1.0.1
|
||||
-L/opt/ps5-payload-sdk/target/user/homebrew/lib -lmicrohttpd -lpthread
|
||||
[6/7] make all (增量编译, 复用第三方 obj)...
|
||||
mkdir gen
|
||||
python3 gen-asset-module.py --path icon-back.png assets/icon-back.png > gen/icon-back.png.c
|
||||
python3 gen-asset-module.py --path icon-file.png assets/icon-file.png > gen/icon-file.png.c
|
||||
python3 gen-asset-module.py --path icon-folder.png assets/icon-folder.png > gen/icon-folder.png.c
|
||||
python3 gen-asset-module.py --path icon-generic.png assets/icon-generic.png > gen/icon-generic.png.c
|
||||
python3 gen-asset-module.py --path icon-image.png assets/icon-image.png > gen/icon-image.png.c
|
||||
python3 gen-asset-module.py --path icon-pkg.png assets/icon-pkg.png > gen/icon-pkg.png.c
|
||||
python3 gen-asset-module.py --path icon-up.png assets/icon-up.png > gen/icon-up.png.c
|
||||
python3 gen-asset-module.py --path index.html assets/index.html > gen/index.html.c
|
||||
python3 gen-asset-module.py --path lang-en.js assets/lang-en.js > gen/lang-en.js.c
|
||||
python3 gen-asset-module.py --path lang-zh.js assets/lang-zh.js > gen/lang-zh.js.c
|
||||
python3 gen-asset-module.py --path main.css assets/main.css > gen/main.css.c
|
||||
python3 gen-asset-module.py --path main.js assets/main.js > gen/main.js.c
|
||||
python3 gen-asset-module.py --path param.json assets/param.json > gen/param.json.c
|
||||
/opt/ps5-payload-sdk/bin/prospero-clang -O2 -w -Ithird_party/zlib/include -Ithird_party/minizip-ng/include -DHAVE_ZLIB -DZLIB_COMPAT -DHAVE_UNISTD_H=1 -D_FILE_OFFSET_BITS=64 -D_LARGEFILE64_SOURCE -c -o ps5-obj/third_party/zlib/src/adler32.o third_party/zlib/src/adler32.c
|
||||
/opt/ps5-payload-sdk/bin/prospero-clang -O2 -w -Ithird_party/zlib/include -Ithird_party/minizip-ng/include -DHAVE_ZLIB -DZLIB_COMPAT -DHAVE_UNISTD_H=1 -D_FILE_OFFSET_BITS=64 -D_LARGEFILE64_SOURCE -c -o ps5-obj/third_party/zlib/src/crc32.o third_party/zlib/src/crc32.c
|
||||
/opt/ps5-payload-sdk/bin/prospero-clang -O2 -w -Ithird_party/zlib/include -Ithird_party/minizip-ng/include -DHAVE_ZLIB -DZLIB_COMPAT -DHAVE_UNISTD_H=1 -D_FILE_OFFSET_BITS=64 -D_LARGEFILE64_SOURCE -c -o ps5-obj/third_party/zlib/src/deflate.o third_party/zlib/src/deflate.c
|
||||
/opt/ps5-payload-sdk/bin/prospero-clang -O2 -w -Ithird_party/zlib/include -Ithird_party/minizip-ng/include -DHAVE_ZLIB -DZLIB_COMPAT -DHAVE_UNISTD_H=1 -D_FILE_OFFSET_BITS=64 -D_LARGEFILE64_SOURCE -c -o ps5-obj/third_party/zlib/src/inffast.o third_party/zlib/src/inffast.c
|
||||
/opt/ps5-payload-sdk/bin/prospero-clang -O2 -w -Ithird_party/zlib/include -Ithird_party/minizip-ng/include -DHAVE_ZLIB -DZLIB_COMPAT -DHAVE_UNISTD_H=1 -D_FILE_OFFSET_BITS=64 -D_LARGEFILE64_SOURCE -c -o ps5-obj/third_party/zlib/src/inflate.o third_party/zlib/src/inflate.c
|
||||
/opt/ps5-payload-sdk/bin/prospero-clang -O2 -w -Ithird_party/zlib/include -Ithird_party/minizip-ng/include -DHAVE_ZLIB -DZLIB_COMPAT -DHAVE_UNISTD_H=1 -D_FILE_OFFSET_BITS=64 -D_LARGEFILE64_SOURCE -c -o ps5-obj/third_party/zlib/src/inftrees.o third_party/zlib/src/inftrees.c
|
||||
/opt/ps5-payload-sdk/bin/prospero-clang -O2 -w -Ithird_party/zlib/include -Ithird_party/minizip-ng/include -DHAVE_ZLIB -DZLIB_COMPAT -DHAVE_UNISTD_H=1 -D_FILE_OFFSET_BITS=64 -D_LARGEFILE64_SOURCE -c -o ps5-obj/third_party/zlib/src/trees.o third_party/zlib/src/trees.c
|
||||
/opt/ps5-payload-sdk/bin/prospero-clang -O2 -w -Ithird_party/zlib/include -Ithird_party/minizip-ng/include -DHAVE_ZLIB -DZLIB_COMPAT -DHAVE_UNISTD_H=1 -D_FILE_OFFSET_BITS=64 -D_LARGEFILE64_SOURCE -c -o ps5-obj/third_party/zlib/src/zutil.o third_party/zlib/src/zutil.c
|
||||
/opt/ps5-payload-sdk/bin/prospero-clang -O2 -w -Ithird_party/zlib/include -Ithird_party/minizip-ng/include -DHAVE_ZLIB -DZLIB_COMPAT -DHAVE_UNISTD_H=1 -D_FILE_OFFSET_BITS=64 -D_LARGEFILE64_SOURCE -c -o ps5-obj/third_party/minizip-ng/src/mz_crypt.o third_party/minizip-ng/src/mz_crypt.c
|
||||
/opt/ps5-payload-sdk/bin/prospero-clang -O2 -w -Ithird_party/zlib/include -Ithird_party/minizip-ng/include -DHAVE_ZLIB -DZLIB_COMPAT -DHAVE_UNISTD_H=1 -D_FILE_OFFSET_BITS=64 -D_LARGEFILE64_SOURCE -c -o ps5-obj/third_party/minizip-ng/src/mz_os.o third_party/minizip-ng/src/mz_os.c
|
||||
/opt/ps5-payload-sdk/bin/prospero-clang -O2 -w -Ithird_party/zlib/include -Ithird_party/minizip-ng/include -DHAVE_ZLIB -DZLIB_COMPAT -DHAVE_UNISTD_H=1 -D_FILE_OFFSET_BITS=64 -D_LARGEFILE64_SOURCE -c -o ps5-obj/third_party/minizip-ng/src/mz_os_posix.o third_party/minizip-ng/src/mz_os_posix.c
|
||||
/opt/ps5-payload-sdk/bin/prospero-clang -O2 -w -Ithird_party/zlib/include -Ithird_party/minizip-ng/include -DHAVE_ZLIB -DZLIB_COMPAT -DHAVE_UNISTD_H=1 -D_FILE_OFFSET_BITS=64 -D_LARGEFILE64_SOURCE -c -o ps5-obj/third_party/minizip-ng/src/mz_strm.o third_party/minizip-ng/src/mz_strm.c
|
||||
/opt/ps5-payload-sdk/bin/prospero-clang -O2 -w -Ithird_party/zlib/include -Ithird_party/minizip-ng/include -DHAVE_ZLIB -DZLIB_COMPAT -DHAVE_UNISTD_H=1 -D_FILE_OFFSET_BITS=64 -D_LARGEFILE64_SOURCE -c -o ps5-obj/third_party/minizip-ng/src/mz_strm_mem.o third_party/minizip-ng/src/mz_strm_mem.c
|
||||
/opt/ps5-payload-sdk/bin/prospero-clang -O2 -w -Ithird_party/zlib/include -Ithird_party/minizip-ng/include -DHAVE_ZLIB -DZLIB_COMPAT -DHAVE_UNISTD_H=1 -D_FILE_OFFSET_BITS=64 -D_LARGEFILE64_SOURCE -c -o ps5-obj/third_party/minizip-ng/src/mz_strm_os_posix.o third_party/minizip-ng/src/mz_strm_os_posix.c
|
||||
third_party/minizip-ng/src/mz_strm_os_posix.c:95:33: error: use of undeclared identifier 'O_BINARY'
|
||||
95 | fd = open(path, mode_open | O_BINARY, S_IRUSR | S_IWUSR | S_IRGRP);
|
||||
| ^
|
||||
1 error generated.
|
||||
make: *** [Makefile:78: ps5-obj/third_party/minizip-ng/src/mz_strm_os_posix.o] Error 1
|
||||
[7/7] 验证产物 + 同步...
|
||||
FAIL: ELF 未生成 -- 日志: /home/song/build-elf.log
|
||||
@@ -0,0 +1,213 @@
|
||||
#!/usr/bin/env bash
|
||||
# PS5 Web File Manager 一键构建 (v4)
|
||||
# 修复:
|
||||
# (1) staging 模式装 libmicrohttpd -> make install 到 /tmp, 再 sudo cp 到 SDK
|
||||
# (2) 显式 sudo -v 刷密码缓存 (v3 的 chown 静默失败了)
|
||||
# (3) 增量: 保留 zlib/minizip-ng OBJ 缓存 (不每次 make clean)
|
||||
# (4) libmicrohttpd tarball 缓存 ~/.cache/, 失败可重试不重下
|
||||
# (5) 同步源用 rsync 增量
|
||||
# (6) 失败时把 log 同步到 Windows .build/build-elf.log
|
||||
#
|
||||
# 跑法: bash /home/song/build-elf.sh
|
||||
|
||||
set -uo pipefail
|
||||
|
||||
export PS5_PAYLOAD_SDK=/opt/ps5-payload-sdk
|
||||
SRC_WIN="/mnt/c/Users/songl/Desktop/Web File Manager/ps5-web-file-manager"
|
||||
PROJ="/home/song/ps5-web-file-manager"
|
||||
LOG="/home/song/build-elf.log"
|
||||
STAGE="/tmp/ps5-homebrew-ext4"
|
||||
CACHE="${HOME}/.cache/libmicrohttpd-1.0.1.tar.gz"
|
||||
|
||||
mkdir -p "$(dirname "$LOG")" "$STAGE" "$(dirname "$CACHE")"
|
||||
: >"$LOG"
|
||||
exec > >(tee -a "$LOG") 2>&1
|
||||
|
||||
echo "=========================================="
|
||||
echo "PS5 Web File Manager -- ELF 构建 v4"
|
||||
echo "时间: $(date)"
|
||||
echo "PS5_PAYLOAD_SDK=$PS5_PAYLOAD_SDK"
|
||||
echo "user: $(whoami) uid=$(id -u)"
|
||||
echo "=========================================="
|
||||
|
||||
# 0. sudo 密码缓存 (脚本会跑 sudo 几次, 在开头集中要求密码)
|
||||
echo "[0/7] 预热 sudo (会提示输 1 次密码)..."
|
||||
sudo -v 2>&1 || { echo "FAIL: sudo 不可用"; exit 99; }
|
||||
|
||||
# 1. SDK 验证
|
||||
if [ ! -x "$PS5_PAYLOAD_SDK/bin/prospero-clang" ]; then
|
||||
if [ -f /home/song/ps5-payload-sdk.zip ]; then
|
||||
echo "[1/7] 展开 SDK zip 到 $PS5_PAYLOAD_SDK ..."
|
||||
sudo mkdir -p "$PS5_PAYLOAD_SDK"
|
||||
sudo unzip -q -o /home/song/ps5-payload-sdk.zip -d /tmp/sdk-stage
|
||||
sudo cp -r /tmp/sdk-stage/. "$PS5_PAYLOAD_SDK/"
|
||||
sudo chmod -R a+rx "$PS5_PAYLOAD_SDK"
|
||||
else
|
||||
echo "FAIL: 未发现 /home/song/ps5-payload-sdk.zip 也无 $PS5_PAYLOAD_SDK/bin/prospero-clang"
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
echo "[1/7] SDK OK: $PS5_PAYLOAD_SDK"
|
||||
|
||||
# 2. LLVM 工具链
|
||||
if ! command -v llvm-config-18 >/dev/null 2>&1; then
|
||||
echo "[2/7] 装 LLVM 18 (apt.llvm.org) ..."
|
||||
if [ ! -x /tmp/llvm.sh ]; then
|
||||
wget -q https://apt.llvm.org/llvm.sh -O /tmp/llvm.sh
|
||||
chmod +x /tmp/llvm.sh
|
||||
fi
|
||||
sudo /tmp/llvm.sh 18 all >/dev/null 2>&1 || true
|
||||
fi
|
||||
if ! command -v pkg-config >/dev/null 2>&1; then
|
||||
echo "[2/7] 装 pkg-config ..."
|
||||
sudo apt-get install -y pkg-config rsync
|
||||
fi
|
||||
LLVM_BINDIR="$(llvm-config-18 --bindir 2>/dev/null || llvm-config --bindir)"
|
||||
echo "[2/7] LLVM bindir: $LLVM_BINDIR"
|
||||
|
||||
# 3. 工具链验证
|
||||
echo "[3/7] prospero-clang --version ..."
|
||||
"$PS5_PAYLOAD_SDK/bin/prospero-clang" --version | head -1
|
||||
|
||||
# 4. 同步源码 (rsync 增量)
|
||||
echo "[4/7] 同步源码 (Windows -> WSL, 增量) ..."
|
||||
mkdir -p "$PROJ"
|
||||
if [ -d "$SRC_WIN/src" ]; then
|
||||
rsync -a --delete \
|
||||
--exclude='ps5-obj' --exclude='linux-obj' \
|
||||
--exclude='web-file-mgr.elf' --exclude='web-file-mgr-linux' \
|
||||
--exclude='.build/stub' --exclude='.build/obj' \
|
||||
--exclude='.build/probe*' --exclude='.build/probe-work' \
|
||||
--exclude='.build/host-test' --exclude='.build/tp' \
|
||||
--exclude='.build/*.exe' --exclude='.build/*.txt' \
|
||||
--exclude='gen' \
|
||||
"$SRC_WIN/" "$PROJ/"
|
||||
echo " 已同步自 $SRC_WIN"
|
||||
else
|
||||
echo " (Windows 端不可见 -- /mnt/c 是否挂载?)"
|
||||
fi
|
||||
|
||||
# 5. 装 libmicrohttpd (staging 模式: make install 到 /tmp, 再 sudo cp 过去)
|
||||
echo "[5/7] libmicrohttpd (staging + 一次性 sudo cp)..."
|
||||
if "$PS5_PAYLOAD_SDK/bin/prospero-pkg-config" --modversion libmicrohttpd >/dev/null 2>&1; then
|
||||
echo " 已装: $($PS5_PAYLOAD_SDK/bin/prospero-pkg-config --modversion libmicrohttpd) -- 跳过"
|
||||
else
|
||||
# 5a) 下载 tarball (缓存复用)
|
||||
if [ ! -f "$CACHE" ] || [ ! -s "$CACHE" ]; then
|
||||
echo " 下载 libmicrohttpd-1.0.1.tar.gz..."
|
||||
if command -v wget >/dev/null; then
|
||||
wget -q https://ftp.gnu.org/gnu/libmicrohttpd/libmicrohttpd-1.0.1.tar.gz -O "$CACHE"
|
||||
else
|
||||
curl -fsSL https://ftp.gnu.org/gnu/libmicrohttpd/libmicrohttpd-1.0.1.tar.gz -o "$CACHE"
|
||||
fi
|
||||
fi
|
||||
[ -s "$CACHE" ] || { echo "FAIL: 下载失败: $CACHE"; exit 2; }
|
||||
echo " tarball: $CACHE ($(du -h "$CACHE" | cut -f1))"
|
||||
|
||||
# 5b) 解压 + configure + make (普通 user 跑, 输出到临时目录)
|
||||
TMPSRC=$(mktemp -d)
|
||||
trap 'rm -rf -- "$TMPSRC"' EXIT
|
||||
tar xf "$CACHE" -C "$TMPSRC"
|
||||
cd "$TMPSRC/libmicrohttpd-1.0.1"
|
||||
source "${PS5_PAYLOAD_SDK}/toolchain/prospero.sh"
|
||||
export CFLAGS="${CFLAGS:-} -O1 -w"
|
||||
echo " configure (host=x86_64-pc-freebsd)..."
|
||||
./configure --prefix="/user/homebrew" \
|
||||
--host=x86_64-pc-freebsd \
|
||||
--enable-static --disable-shared \
|
||||
--disable-doc --disable-curl --disable-examples \
|
||||
--disable-https --disable-openssl \
|
||||
>/dev/null 2>&1
|
||||
echo " make -j$(nproc)..."
|
||||
make -j"$(nproc)" 2>&1 | tail -10
|
||||
|
||||
# 5c) install 到 STAGE (普通 user 写到 /tmp)
|
||||
rm -rf "$STAGE"
|
||||
mkdir -p "$STAGE"
|
||||
echo " make install DESTDIR=$STAGE ..."
|
||||
make install DESTDIR="$STAGE" 2>&1 | tail -5
|
||||
|
||||
# 5d) 验证 stage 里有产物
|
||||
if [ ! -f "$STAGE/user/homebrew/include/microhttpd.h" ]; then
|
||||
echo "FAIL: stage 中无 microhttpd.h -- stage 内容:"
|
||||
find "$STAGE" -maxdepth 4 -type f | head -10
|
||||
exit 2
|
||||
fi
|
||||
|
||||
# 5e) 一次性 sudo 拷到 SDK
|
||||
echo " sudo cp stage -> SDK homebrew..."
|
||||
sudo mkdir -p "$PS5_PAYLOAD_SDK/target/user/homebrew"
|
||||
sudo cp -r "$STAGE/user/homebrew/." "$PS5_PAYLOAD_SDK/target/user/homebrew/"
|
||||
echo " 验证 prospero-pkg-config:"
|
||||
"$PS5_PAYLOAD_SDK/bin/prospero-pkg-config" --modversion libmicrohttpd
|
||||
"$PS5_PAYLOAD_SDK/bin/prospero-pkg-config" --libs libmicrohttpd
|
||||
fi
|
||||
|
||||
# 6. 编译 (不 make clean, 复用 zlib/minizip-ng OBJ 缓存)
|
||||
echo "[6/7] make all (增量编译, 复用第三方 obj)..."
|
||||
cd "$PROJ"
|
||||
export PS5_PAYLOAD_SDK="/opt/ps5-payload-sdk"
|
||||
# 仅当二进制缺失或源码变更才全量重编
|
||||
# 重要: 必须包含 assets/* 和 gen-asset-module.py —— 它们经 gen-asset-module.py
|
||||
# 生成 gen/*.c 进而影响 ELF, 不在列表里就会跳过 make 产生伪"无变更"(v1.8.3
|
||||
# 被这个 bug 坑过, ELF sha256 没变)。
|
||||
if [ -f web-file-mgr.elf ]; then
|
||||
echo " 已存在 web-file-mgr.elf, 检查源码变更..."
|
||||
NEEDS_REBUILD=""
|
||||
for src in src/*.c Makefile assets/* gen-asset-module.py \
|
||||
third_party/minizip-ng/include/*.h third_party/zlib/include/*.h; do
|
||||
[ -e "$src" ] || continue
|
||||
if [ "$src" -nt web-file-mgr.elf ]; then
|
||||
NEEDS_REBUILD="$src"
|
||||
break
|
||||
fi
|
||||
done
|
||||
if [ -z "$NEEDS_REBUILD" ]; then
|
||||
echo " 无源码变更 -- 跳过 make"
|
||||
else
|
||||
echo " 源码变更: $NEEDS_REBUILD"
|
||||
if ! make all 2>&1 | tail -60; then
|
||||
echo "FAIL: make all 失败(见上)。注意: 旧 ELF 仍留在原地, 但不算新产物"
|
||||
exit 6
|
||||
fi
|
||||
fi
|
||||
else
|
||||
if ! make all 2>&1 | tail -60; then
|
||||
echo "FAIL: make all 失败(见上)"
|
||||
exit 6
|
||||
fi
|
||||
fi
|
||||
|
||||
# 7. 验证 + 同步回 Windows
|
||||
echo "[7/7] 验证产物 + 同步..."
|
||||
ELF="$PROJ/web-file-mgr.elf"
|
||||
if [ ! -f "$ELF" ]; then
|
||||
echo "FAIL: ELF 未生成 -- 日志: $LOG"
|
||||
# 把日志同步到 Windows .build/ 方便排查
|
||||
cp -f "$LOG" "$SRC_WIN/.build/build-elf.log" 2>/dev/null && \
|
||||
echo " 日志已同步: $SRC_WIN/.build/build-elf.log"
|
||||
exit 3
|
||||
fi
|
||||
SIZE=$(stat -c%s "$ELF")
|
||||
HASH=$(sha256sum "$ELF" | cut -d' ' -f1)
|
||||
echo "OK: $ELF"
|
||||
echo " size: $SIZE bytes (~$((SIZE/1024)) KiB)"
|
||||
echo " sha256: $HASH"
|
||||
echo " header bytes (期望 ELF64 magic 7f454c46 + class 2 + little-endian 1 + e_machine 0x003e=x86-64):"
|
||||
head -c 20 "$ELF" | xxd | head -2
|
||||
echo ""
|
||||
# 用 od 直接读 offset 18 起的 1 个 16-bit 小端 word = e_machine
|
||||
EM_RAW=$(od -An -tx2 -j 18 -N 2 "$ELF" | tr -d ' \n') # e.g. "3e00"
|
||||
EM_NUM=$((16#${EM_RAW})) # decimal
|
||||
echo "e_machine = 0x$EM_RAW ($EM_NUM)"
|
||||
case "$EM_NUM" in
|
||||
62) echo " -> x86-64 (PS5 真机 target: x86_64-sie-ps5)" ;;
|
||||
183) echo " -> aarch64 (注意: PS5 实际是 x86-64, 此结果可疑)" ;;
|
||||
*) echo " -> 未知架构 (期望 62=0x3e, 当前 e_machine=$EM_NUM)" ;;
|
||||
esac
|
||||
|
||||
# 同步 ELF 回项目根
|
||||
mkdir -p "$SRC_WIN" 2>/dev/null
|
||||
cp -f "$ELF" "$SRC_WIN/web-file-mgr.elf" 2>&1 && \
|
||||
echo "" && echo "[bonus] 已同步回: $SRC_WIN/web-file-mgr.elf" && \
|
||||
ls -lh "$SRC_WIN/web-file-mgr.elf"
|
||||
@@ -0,0 +1,58 @@
|
||||
#!/usr/bin/env bash
|
||||
# ===========================================================================
|
||||
# ps5-web-file-manager -- Windows 端一键构建入口
|
||||
#
|
||||
# 在 Windows 的 Git Bash / MinGW64 bash 里这样跑(注意用 /usr/bin/bash):
|
||||
# /usr/bin/bash .build/build-win.sh
|
||||
#
|
||||
# 它只做一件事:把 WSL 脚本喂给 wsl.exe 执行,然后透传退出码。
|
||||
# 真正的 sync / make / verify / 拷回都在 .build/build-elf-wsl.sh 里。
|
||||
#
|
||||
# 注意:
|
||||
# * 用 `bash < script` 走 stdin,不要用 `bash -c '...'` ——
|
||||
# 路径含空格时 -c 的参数会被 wsl.exe 拆断。
|
||||
# * 别写裸 `bash .build/build-win.sh`,那个 bash 可能解析到
|
||||
# C:\Windows\System32\bash.exe(WSL 启动器),脚本会跑进 Linux 环境。
|
||||
# ===========================================================================
|
||||
|
||||
set -uo pipefail
|
||||
|
||||
# Git Bash / MSYS2 rewrites anything that looks like a POSIX path before the
|
||||
# argument reaches wsl.exe, so `/mnt/c/...` becomes
|
||||
# `<msys-root>/mnt/c/...` and the WSL side reports "cannot stat" (this is how
|
||||
# the script first failed under PortableGit). WSL sees Linux paths, so the
|
||||
# translation must be off for the whole script. Harmless on a real Linux shell.
|
||||
export MSYS_NO_PATHCONV=1
|
||||
export MSYS2_ARG_CONV_EXCL='*'
|
||||
|
||||
REPO='/c/Users/songl/Desktop/Web File Manager/ps5-web-file-manager'
|
||||
WSL_DISTRO='Ubuntu-22.04'
|
||||
WSL_DIR='/home/song/ps5-web-file-manager/.build'
|
||||
HOST_SCRIPT="$REPO/.build/build-elf-wsl.sh"
|
||||
|
||||
[ -f "$HOST_SCRIPT" ] || { echo "[FAIL] 找不到 $HOST_SCRIPT"; exit 1; }
|
||||
|
||||
# 每次都把最新的 WSL 脚本推进去(.build/ 被 rsync 排除,WSL 侧不会自己更新;
|
||||
# 只做一次会导致改了脚本还在跑旧版 —— 这个坑踩过)
|
||||
wsl.exe -d "$WSL_DISTRO" -- mkdir -p "$WSL_DIR" || exit 1
|
||||
wsl.exe -d "$WSL_DISTRO" -- cp \
|
||||
'/mnt/c/Users/songl/Desktop/Web File Manager/ps5-web-file-manager/.build/build-elf-wsl.sh' \
|
||||
"$WSL_DIR/build-elf-wsl.sh" || {
|
||||
echo "[FAIL] 无法写入 WSL 文件系统"
|
||||
exit 1
|
||||
}
|
||||
|
||||
echo "[run] wsl.exe -d $WSL_DISTRO -- bash < build-elf-wsl.sh"
|
||||
echo "=================================================================="
|
||||
wsl.exe -d "$WSL_DISTRO" -- bash < "$HOST_SCRIPT"
|
||||
rc=$?
|
||||
echo "=================================================================="
|
||||
|
||||
if [ "$rc" -eq 0 ]; then
|
||||
# 输出文件带版本号(web-file-mgr-<VERSION_TAG>.elf),列出实际产物
|
||||
echo "[OK] 构建完成,产物:"
|
||||
ls -1 "$REPO"/web-file-mgr-*.elf 2>/dev/null | sed 's#^# #'
|
||||
else
|
||||
echo "[FAIL] 构建失败 (exit=$rc)"
|
||||
fi
|
||||
exit "$rc"
|
||||
@@ -0,0 +1,62 @@
|
||||
"""Verify that specific identifiers made it into the ELF's embedded assets.
|
||||
|
||||
`gen-asset-module.py` gzips every JS/CSS/HTML asset (compresslevel=9, mtime=0),
|
||||
so plain `strings` finds none of their text -- a zero hit means "compressed",
|
||||
not "missing". This script walks the file for gzip streams, decompresses each
|
||||
one and reports where each key landed.
|
||||
|
||||
Usage:
|
||||
python .build/check-elf-gzip.py <elf> [key ...]
|
||||
|
||||
With no keys it defaults to the current round: the v1.9.3M fork marker and the
|
||||
footer tooltip that explains it. Pass your own after the ELF path when reusing
|
||||
this for another change.
|
||||
|
||||
Scope note: this covers **assets only**. C-side string literals (e.g.
|
||||
`extract_dict_too_large`, `versionsTooltip` in a header) are not compressed and
|
||||
belong to a plain `strings` check.
|
||||
"""
|
||||
import sys
|
||||
import zlib
|
||||
|
||||
if len(sys.argv) < 2:
|
||||
sys.exit(__doc__)
|
||||
|
||||
f = sys.argv[1]
|
||||
data = open(f, "rb").read()
|
||||
|
||||
keys = [k.encode() for k in sys.argv[2:]] or [
|
||||
b"v1.9.3M", # assets/main.js footer fallback
|
||||
b"versionTooltip", # main.js + both lang files
|
||||
"本版为 LisherSong 改版".encode(), # assets/lang-zh.js
|
||||
b"Modified build by LisherSong", # assets/lang-en.js
|
||||
]
|
||||
|
||||
seen = {}
|
||||
streams = 0
|
||||
i = data.find(b"\x1f\x8b\x08")
|
||||
while i >= 0:
|
||||
# Try to decompress starting here, capped at 2 MiB.
|
||||
end_cap = min(len(data), i + 2 * 1024 * 1024)
|
||||
try:
|
||||
dec = zlib.decompressobj(zlib.MAX_WBITS | 16)
|
||||
chunk = dec.decompress(data[i:end_cap], 2 * 1024 * 1024)
|
||||
if dec.eof and chunk:
|
||||
streams += 1
|
||||
for k in keys:
|
||||
if k in chunk and k not in seen:
|
||||
idx = chunk.find(k)
|
||||
ctx = chunk[max(0, idx - 24):idx + len(k) + 48]
|
||||
seen[k] = ctx.decode("utf-8", errors="replace")
|
||||
except Exception:
|
||||
pass
|
||||
i = data.find(b"\x1f\x8b\x08", i + 1)
|
||||
|
||||
print(f"{f}: {streams} gzip streams scanned, keys found: {len(seen)} / {len(keys)}")
|
||||
for k, v in seen.items():
|
||||
print(f" OK {k.decode()}")
|
||||
print(f" context: {v[:160]}")
|
||||
missing = [k for k in keys if k not in seen]
|
||||
for m in missing:
|
||||
print(f" MISS {m.decode()}")
|
||||
sys.exit(1 if missing else 0)
|
||||
@@ -0,0 +1,29 @@
|
||||
import re, sys
|
||||
f = sys.argv[1]
|
||||
data = open(f, "rb").read()
|
||||
runs = re.findall(rb"[\x20-\x7e]{6,}", data)
|
||||
hits = set()
|
||||
keys = [
|
||||
b"zipx_limits_profile",
|
||||
b"k_large_limits",
|
||||
b"ZIPX_LIMITS_LARGE",
|
||||
b"ZIPX_LIMITS_DEFAULT",
|
||||
b"large-file profile",
|
||||
b"large-file",
|
||||
b"LargeFile",
|
||||
b"large_file",
|
||||
b"extractLargeAsk",
|
||||
b"extractLargeActive",
|
||||
b"large",
|
||||
]
|
||||
for r in runs:
|
||||
for k in keys:
|
||||
if k in r and k.decode() not in hits:
|
||||
# Trim the context a bit
|
||||
i = r.find(k)
|
||||
ctx = r[max(0, i - 8):i + len(k) + 24].strip()
|
||||
if len(ctx) < 80:
|
||||
hits.add(ctx.decode(errors="replace"))
|
||||
for h in sorted(hits):
|
||||
print(repr(h))
|
||||
print("total:", len(hits))
|
||||
@@ -1,5 +1,41 @@
|
||||
gen/
|
||||
# Build artifacts
|
||||
*.elf
|
||||
web-file-mgr-linux
|
||||
*.o
|
||||
*.d
|
||||
gen/
|
||||
web-file-mgr-linux
|
||||
|
||||
# Sandbox scratch under .build/: ignore the whole directory, then re-allow the
|
||||
# handful of files that are actually part of the repo (build scripts + the
|
||||
# ELF checker). A whitelist is the only thing that survives -- every debugging
|
||||
# session drops a new probe directory in here.
|
||||
.build/*
|
||||
!.build/build-elf.sh
|
||||
!.build/build-elf-wsl.sh
|
||||
!.build/build-win.sh
|
||||
!.build/check-elf-*.py
|
||||
!.build/extract-demo.html
|
||||
!.build/build-elf.log
|
||||
|
||||
# Generated 7z fixtures (tests/make_sevenz_fixtures.py rebuilds them).
|
||||
tests/fixtures-7z/
|
||||
|
||||
# Staging trees the fixture generators copy from (make-rar-fixtures.bat uses
|
||||
# fixture-stage, make-zip-enc-fixtures.bat uses fixture-stage-zip). The
|
||||
# archives they produce under tests/fixtures-real/ ARE committed.
|
||||
tests/fixture-stage/
|
||||
tests/fixture-stage-zip/
|
||||
|
||||
# Python bytecode (the fixture generators and bench_driver are hand-run)
|
||||
__pycache__/
|
||||
*.pyc
|
||||
|
||||
# Editor / OS noise
|
||||
.vscode/
|
||||
.idea/
|
||||
*.swp
|
||||
.DS_Store
|
||||
Thumbs.db
|
||||
|
||||
# WorkBuddy session data (never commit; also never delete).
|
||||
.workbuddy/
|
||||
@@ -0,0 +1,863 @@
|
||||
# Changelog
|
||||
|
||||
All notable changes to **PS5 Web File Manager** are documented in this file.
|
||||
The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
|
||||
and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
||||
|
||||
> Release artifact for v1.9.3M — **built and verified locally, not published yet**:
|
||||
> `web-file-mgr-v1.9.3M.elf` — size 903 448 bytes (~882 KiB)
|
||||
> sha256 `8ca47d5aaca75085b32641300cce30fadb7df7749cb6b53d04f129bcecc286b7`
|
||||
> ELF class 64, little-endian, e_machine `0x003e` (x86_64-sie-ps5)
|
||||
>
|
||||
> **The trailing `M` is the fork marker** (Modified build, maintained by
|
||||
> LisherSong). Upstream owendswang ships plain `vX.Y.Z`, so a version string on
|
||||
> its own says which of the two you are looking at. The marker rides on
|
||||
> `VERSION_TAG` rather than on a display-only constant, so `/api/version`, the
|
||||
> PS5 start-up notification, the stdout banner, the UI footer and the ELF file
|
||||
> name all gain it in one move — nothing can be forgotten in one of the five.
|
||||
> A second benefit is that a fork build can no longer collide with an upstream
|
||||
> artifact of the same upstream version, a mix-up that has already happened
|
||||
> twice. The footer additionally carries a tooltip spelling the marker out
|
||||
> (`versionTooltip`, en + zh). This is the first release using the convention;
|
||||
> the older entries below keep their original plain numbers.
|
||||
>
|
||||
> This is the first binary to carry the encrypted-archive work and the
|
||||
> dictionary-reporting fix (`[v1.9.3M]` below). It exists for on-device testing:
|
||||
> the GitHub release is still v1.9.2 and its binary contains none of it.
|
||||
>
|
||||
> Its size is **unchanged yet again** (903 448 B) although the content grew, for
|
||||
> the sixth build in a row. Every round of this release has only moved
|
||||
> `.rodata`, and by less than the 16 KiB section alignment absorbs:
|
||||
>
|
||||
> | build | `.rodata` | delta | what changed |
|
||||
> |---|---|---|---|
|
||||
> | `v1.9.3` (pre-marker) | 0x025F80 | — | — |
|
||||
> | `v1.9.3M` + encrypted archives | 0x0260C0 | +0x140 | the `M` marker, its tooltip, re-gzipped assets |
|
||||
> | `+` upload menu and i18n names | 0x026A40 | +0x980 | the menu, the hint, the new copy |
|
||||
> | `+` hint move, status clamp, retry key | 0x026B40 | +0x100 | final copy and CSS |
|
||||
> | `+` menu row highlight fix | 0x026CC0 | +0x180 | the scoped row highlight rules and their comment |
|
||||
> | `+` always-on extract button | 0x026F00 | +0x240 | the un-hidden button, three disabled reasons, the tooltip fix |
|
||||
>
|
||||
> `.text` is byte-for-byte the same size across all six, which is the expected
|
||||
> shape for a change that touches no C logic. **Never infer "nothing changed"
|
||||
> from the file size** — compare sections with `readelf -SW`. Each of the six
|
||||
> carries a different sha256 despite the identical size, so the digest, not the
|
||||
> byte count, is what identifies a build.
|
||||
>
|
||||
> Putting the extract button on screen at all times is not free: it widens the
|
||||
> resting toolbar by its own width, so the width at which the toolbar wraps onto
|
||||
> a second row moves out from 1080px to 1190px in Chinese, and from 1230px to
|
||||
> 1350px in English, where the labels are longer. The console is 1920px wide and
|
||||
> 1280px still fits in Chinese, so the trade was accepted: the alternative was
|
||||
> leaving the entry hidden until an archive happened to be selected, which is
|
||||
> what made it undiscoverable in the first place. The measured threshold is now
|
||||
> pinned by the headless harness rather than left to be rediscovered.
|
||||
>
|
||||
> Release artifact for v1.9.2:
|
||||
> `web-file-mgr-v1.9.2.elf` — size 870 488 bytes (~850 KiB)
|
||||
> sha256 `177e90fecf93a0251e83f67884fba4551051be248330b0d70fda8ea732f88e84`
|
||||
> ELF class 64, little-endian, e_machine `0x003e` (x86_64-sie-ps5)
|
||||
>
|
||||
> Behaviourally identical to the published v1.9.1 binary — the only source
|
||||
> delta is the version literal itself (`VERSION_TAG` in the Makefile, plus the
|
||||
> UI footer fallback in `assets/main.js`). The build is reproducible: reverting
|
||||
> those two literals and rebuilding reproduces the v1.9.1 ELF byte for byte, so
|
||||
> nothing else differs. See [v1.9.2] below for why the version moved at all.
|
||||
>
|
||||
> Release artifact for v1.9.1 (superseded — the tag pointed four commits behind
|
||||
> the tree that actually produced this binary):
|
||||
> `web-file-mgr-v1.9.1.elf` — size 870 488 bytes (~850 KiB)
|
||||
> sha256 `24392aff6ddcca4dc0ea969cce356bd693ac52efe8a117d61ee1c814aa43cd07`
|
||||
> ELF class 64, little-endian, e_machine `0x003e` (x86_64-sie-ps5)
|
||||
>
|
||||
> Built on the 7z-complete tree: LZMA SDK decode subset + self-written codec
|
||||
> chain, 7zAES, and ZIP/RAR/7z volume support. Build-system-only delta vs the
|
||||
> first v1.9.1 artifact (1 017 864 B): `src/demangle_stub.c` keeps libc++abi's
|
||||
> Itanium name demangler (105 KiB, only reachable from the uncaught-exception
|
||||
> path) out of the link, and `-Wl,--icf=all` folds identical functions.
|
||||
> −15.8% overall with no change to functionality or decompression throughput.
|
||||
> See `docs/SIZE-OPTIMIZATION.md`.
|
||||
>
|
||||
> Release artifact for v1.9:
|
||||
> `web-file-mgr.elf` — size 919 440 bytes (~897 KiB)
|
||||
> sha256 `bb8f17e9addc6a9984f611503ca01b51f8984b773d353630da1f25bd1a28a997`
|
||||
> ELF class 64, little-endian, e_machine `0x003e` (x86_64-sie-ps5)
|
||||
>
|
||||
> Source delta vs v1.8.3: RAR engine replaced (dmc_unrar 1.7.0 → rarlab
|
||||
> UnRAR 7.20.1, `third_party/unrar/` → `third_party/unrar7/`), new
|
||||
> `src/rar_extract.c` scan/extract implementation, Makefile + host-test
|
||||
> C++ rules, 5 real RAR fixtures committed. See [v1.9] below.
|
||||
>
|
||||
> Release artifact for v1.8.3:
|
||||
> `web-file-mgr.elf` — size 509 704 bytes (~497 KiB)
|
||||
> sha256 `fdcf7b09b69e2160e77dfa084c0e890ba0696d4dd478b1d5ff499cdc9f527955`
|
||||
> ELF class 64, little-endian, e_machine `0x003e` (x86_64-sie-ps5)
|
||||
>
|
||||
> Source delta vs v1.8.2: 5 files touched (4 user-facing + 1 build pipeline) —
|
||||
> see [v1.8.3] below for details.
|
||||
>
|
||||
> Release artifact for v1.8.2:
|
||||
> `web-file-mgr.elf` — size 509 704 bytes (~497 KiB)
|
||||
> sha256 `1b2c3d68b35e32737105f17d14a80a3c159ceca0cabd274ee168cbcd81906f65`
|
||||
> ELF class 64, little-endian, e_machine `0x003e` (x86_64-sie-ps5)
|
||||
>
|
||||
> Source delta vs v1.8.1: 5 files relaxed (`src/zip_extract.c` k_default_limits
|
||||
> + k_large_limits, `assets/main.js` `LARGE_FILE_THRESHOLD_BYTES`,
|
||||
> `assets/lang-{en,zh}.js` copy, `tests/test_zip_extract.c` advertised-number
|
||||
> assertion, README/HANDOVER numeric references) + 2 PS5-only build fixes
|
||||
> (`Makefile` CFLAGS `-Ithird_party/unrar`, `src/extract.c` forward
|
||||
> declaration of `extract_progress`); no vendored or engine changes.
|
||||
|
||||
## [v1.9.3M] — 2026-09-23
|
||||
|
||||
**Encrypted archives now extract end to end — ZIP (both schemes), RAR, and 7z
|
||||
with an encrypted header.**
|
||||
|
||||
Until now every encrypted archive was refused up front, even though the
|
||||
password field, the prompt and `extract_password` copy had been in place
|
||||
since v1.9. The gap was in the engines, not the UI:
|
||||
|
||||
- **ZIP**: the vendored minizip-ng had been trimmed past its crypto
|
||||
backend, so the `-DHAVE_WZAES` / `-DHAVE_PKCRYPT` branches inside the
|
||||
(unmodified) `mz_zip.c` had no implementation to call.
|
||||
- **RAR**: rarlab UnRAR could decrypt, but `RARSetPassword` was never
|
||||
called.
|
||||
- **7z**: `-mhe=on` put the file names and the folder table inside the
|
||||
encrypted header, so the archive could not even be listed.
|
||||
|
||||
All three are wired now. A missing or wrong password is reported as
|
||||
`extract_password` (`ZIPX_ERR_PASSWORD` at the engine level), which is the
|
||||
code the task overlay's existing password prompt already reacts to.
|
||||
|
||||
### Changed
|
||||
|
||||
- **The version string gains a fork marker: `v1.9.3` → `v1.9.3M`.** Upstream
|
||||
owendswang releases are plain `vX.Y.Z`, so the `M` (Modified) is what tells a
|
||||
user which of the two they are holding. It is part of `VERSION_TAG` in the
|
||||
Makefile, which means `/api/version`, the PS5 start-up notification, the
|
||||
stdout banner, the UI footer **and the ELF file name** all carry it at once.
|
||||
Because the file name changes too, a fork build can no longer shadow an
|
||||
upstream artifact of the same upstream version.
|
||||
- The UI footer tooltip (`versionTooltip`, en + zh) spells the marker out, so
|
||||
`v1.9.3M` is not left unexplained for someone who has not read this file.
|
||||
`loadVersion()` only ever rewrites the footer's text, so the tooltip survives
|
||||
the `/api/version` round trip.
|
||||
- **The upload button opens a menu instead of hiding half of itself behind a
|
||||
caret.** It used to be a main button plus a small arrow: users read the arrow
|
||||
as decoration and never found "upload a folder" at all. One click now lists
|
||||
**Upload Files / Upload Folder** (`#uploadMenu`, `role="menu"`), with Escape,
|
||||
an outside click and arrow keys handled, and focus moved into the list when it
|
||||
opens. The button keeps a caret so it is obvious that a list is coming.
|
||||
- The footer status line is now clamped to a single line with an ellipsis. It
|
||||
is a fixed-height row that also carries the version and the new drag hint, and
|
||||
a long "uploading 3/12: some-name.zip" used to wrap to two lines and spill out
|
||||
of the 46 px footer.
|
||||
- **The extract button is always on screen and greys out instead of being
|
||||
hidden.** It used to appear only once an extractable archive was selected, so
|
||||
the resting toolbar had no extract entry at all — the same discoverability
|
||||
problem the upload button was changed for. It is now always present, disabled
|
||||
and at 45% opacity whenever the selection cannot be extracted, and its tooltip
|
||||
names the reason: *"Select one archive to extract (ZIP / RAR / 7z)"*, the
|
||||
existing "please select the main volume" when only a `.partNN.rar` sub-volume
|
||||
is selected, and a new "only one archive can be extracted at a time" when
|
||||
several are selected — the old code answered *that* case with the main-volume
|
||||
message, which was simply the wrong sentence. Because `button:disabled` sets
|
||||
`pointer-events: none`, a disabled button cannot be hovered and its tooltip
|
||||
never appears at all, so `.extract-action:disabled` restores hit-testing
|
||||
without restoring clicks; the parent-directory button needed the same fix
|
||||
earlier.
|
||||
- The extract button's label is the short `extract` ("解压" / "Extract"),
|
||||
matching the other toolbar verbs, instead of `extractToCurrent` ("解压到当前
|
||||
目录" / "Extract to current folder"). A button that is on screen permanently
|
||||
should not also be the widest one in the toolbar, and the target folder is
|
||||
still named in the tooltip and in the confirmation dialog. The measured cost
|
||||
of the always-on button: the width at which the toolbar wraps onto a second
|
||||
row moves from 1080px to 1190px in Chinese and from 1230px to 1350px in
|
||||
English, where the labels are longer. 1280px still fits in Chinese and the
|
||||
console is 1920px wide.
|
||||
|
||||
### Added
|
||||
|
||||
- **Encrypted ZIP** — traditional PKWARE ("ZipCrypto", what `zip -e`
|
||||
writes) and WinZip AES-128/192/256 (method `99` + the `0x9901` extra
|
||||
field, what `7z -mem=AES256` writes), for stored and deflated entries.
|
||||
- **Encrypted RAR** — `-p` data encryption and `-hp` header encryption.
|
||||
`RARSetPassword` now runs right after `RAROpenArchiveEx` and before the
|
||||
first `RARReadHeaderEx`, which is the order unrar needs to decrypt a
|
||||
RAR5 header.
|
||||
- `password=` on `/api/extract` now reaches a real decrypt path for both
|
||||
engines. An empty or absent value means "no password", so the raw form
|
||||
field can be passed straight through.
|
||||
- **The UI retries a failed extraction with a password.** An
|
||||
`extract_password` failure used to end in an error box, which for ZIP and
|
||||
RAR meant the password could never be supplied at all — the prompt only
|
||||
existed for 7z. The failed task is now re-sent with whatever the user types,
|
||||
up to three times, and the remembered request keeps the original conflict
|
||||
policy and large-file opt-in. Cancelling or submitting an empty box falls
|
||||
back to the previous failure report. 7z keeps its up-front prompt so an
|
||||
encrypted header does not cost a wasted scan.
|
||||
- `extractPasswordRetryAsk` (en + zh) is the retry prompt's wording, distinct
|
||||
from the up-front `extractPasswordAsk`.
|
||||
- `third_party/minizip-ng/src/mz_crypt_wfm.c` — a local crypto provider for
|
||||
the trimmed minizip-ng: SHA-1, HMAC-SHA1 and AES-128/192/256, with the
|
||||
S-box and the GF(2^8) tables derived on first use so the binary gains no
|
||||
new `.rodata` lookup tables. PBKDF2 comes from the vendored `mz_crypt.c`;
|
||||
the CSPRNG reads `/dev/urandom` rather than `mz_os_rand()`, which keeps
|
||||
`rand`/`srand` out of the import table. Restored verbatim from upstream
|
||||
4.2.2: `mz_strm_wzaes.{c,h}`, `mz_strm_pkcrypt.{c,h}`.
|
||||
- `tests/make-zip-enc-fixtures.bat` and three real fixtures under
|
||||
`tests/fixtures-real/` (`enc-zipcrypto.zip`, `enc-aes256.zip`,
|
||||
`enc-aes256-store.zip`, password `secret123`).
|
||||
- **Encrypted 7z headers (`-mhe=on`)** — the last remaining format gap. With
|
||||
`-mhe=on` the header is itself a folder holding the file names, the folder
|
||||
table and every entry size, so the vendored SDK (whose C decoder has no
|
||||
7zAES coder at all) abandons the archive with `SZ_ERROR_UNSUPPORTED` before
|
||||
it can list a single entry. `src/sevenz_header.c` now reads the
|
||||
`k7zIdEncodedHeader` record, decodes its one folder through the project's
|
||||
own 7zAES path (`src/sevenz_chain.c`) and then gives the SDK a small virtual
|
||||
`ISeekInStream` in which that record has been replaced by the plaintext —
|
||||
the rewritten start header, the decrypted header at the offset the encoded
|
||||
one already occupied, and the real archive everywhere else, so every offset
|
||||
the archive stores still points where it did. Nothing on disk is written to.
|
||||
Archives whose header is only *compressed* (`-mhc=on`, the default) are
|
||||
detected from one byte and never touched, and a wrong password comes back as
|
||||
`ZIPX_ERR_PASSWORD` like any other encrypted archive.
|
||||
- **A visible drag-and-drop hint** (`dropUploadHint`, en + zh) in the footer:
|
||||
*"Drag files or folders into this window to upload"*. Dropping already worked,
|
||||
but nothing but the drop overlay itself ever said so, and that only appears
|
||||
once a drag is under way. It is `remote-only`, like the upload button it
|
||||
describes, and starts hidden so the console browser never flashes it.
|
||||
- `uploadFiles` (en + zh) for the menu's file entry, and
|
||||
`extractPasswordFirstAsk` — a first-failure prompt that says the archive is
|
||||
encrypted rather than blaming a password the user was never asked for.
|
||||
- `extractSelectArchive` and `extractOneAtATime` (en + zh): the two reasons the
|
||||
always-on extract button can be greyed out with nothing useful selected, and
|
||||
with several archives selected. The third reason, `extractSelectMainVolume`,
|
||||
already existed.
|
||||
|
||||
### Fixed
|
||||
|
||||
- **Compiler-flag changes now invalidate objects.** `make` cannot see a
|
||||
flag change, so adding `-DHAVE_WZAES -DHAVE_PKCRYPT` left the existing
|
||||
`mz_zip.o` / `mz_crypt.o` untouched — and since nothing referenced the
|
||||
new streams any more, `--gc-sections` dropped the encryption code again
|
||||
while the link still reported success (the first build of this change was
|
||||
byte-for-byte the published release). The Makefile now records the
|
||||
third-party flag set in `ps5-obj/.third_party_cflags` /
|
||||
`linux-obj/.third_party_cflags` and rebuilds only when it really changes —
|
||||
the same trap the older `LzmaDec.o` rule was working around, generalised.
|
||||
- `ZIPX_ERR_UNSUPPORTED` no longer covers encryption — it is now "multipart
|
||||
or unsupported compression method" only.
|
||||
|
||||
- **An oversized archive dictionary is now reported as such.** An archive whose
|
||||
dictionary exceeds what the build allows used to fail with
|
||||
`extract_entry_too_large` / "A file inside the archive is too large" naming the
|
||||
entry that happened to be in flight — for the 8 GiB-dictionary fixture the
|
||||
message blamed a 7 KB text file. The dictionary is a property of the archive,
|
||||
not of the entry, so it now has its own status (`ZIPX_ERR_LIMIT_DICT`), its own
|
||||
i18n code (`extract_dict_too_large`, en + zh) and the real numbers, which
|
||||
unrar hands over in the `UCM_LARGEDICT` callback: *"needs 8192 MiB (limit
|
||||
4096 MiB)"*.
|
||||
- The **behaviour is unchanged on purpose**: we still refuse, matching what
|
||||
rarlab's own CLI does by default ("8 GB dictionary exceeds the 4 GB limit and
|
||||
needs more than 8 GB of memory; use -md8g or -mdx8g"). Answering `1` to
|
||||
`UCM_LARGEDICT` would only move the problem into `Unpack::Init()`, which
|
||||
allocates the whole dictionary in one block — on a 16 GB console that trades a
|
||||
clean error for an OOM-kill of the payload mid-extraction. Note also that
|
||||
**RAR 5.0 headers cannot express more than 4 GiB** (four dictionary bits,
|
||||
`arcread.cpp:871`), so only RAR7 headers can reach this path at all.
|
||||
- `tests/test_rar_extract.c:test_dict_limit` + a new synthetic fixture
|
||||
`tests/fixtures/dict-8g.rar` (built by `tests/make_fixtures.py:bigdict()`,
|
||||
which writes a minimal valid RAR5 archive by hand: no compressor can produce
|
||||
such a header, and `Rar.exe 7.23` refuses to create RAR7 archives — `-ma4`,
|
||||
`-ma6`, `-ma7` all exit 7). Verified independently with rarlab's own tools:
|
||||
`UnRAR lt` reports `-md=8g` and `UnRAR t -mdx12g` extracts it cleanly.
|
||||
|
||||
- **`err_extract_unsupported` copy was two releases out of date.** It read
|
||||
*"only plain ZIP, single-volume RAR, and 7z are supported"* — the exact
|
||||
inverse of what the build does now, because encrypted archives and
|
||||
multi-volume RAR are both supported. The label now names what is actually
|
||||
accepted (`.zip` / `.rar` / `.7z`, including their multi-volume and
|
||||
encrypted forms); the backend's own sentence, which carries the real cause,
|
||||
is still appended after it. The same stale wording in both READMEs is
|
||||
corrected too. `.rodata` +0x40 (64 B), nothing else changed.
|
||||
|
||||
- **The password prompt never appeared for an archive in a non-ASCII folder
|
||||
— the user only ever saw the failure alert.** The retry that asks for a
|
||||
password was looked up by *path*, and paths are not stable across the wire:
|
||||
the server escapes every byte ≥ 0x80 as `\u00XX` when it serialises a name
|
||||
and `fs_path_value()` maps those back to raw bytes when it receives one, so
|
||||
the string the page holds for a directory and the string the task reports back
|
||||
differ for every name that is not pure ASCII. The lookup missed, the retry
|
||||
returned false, and the plain "extract failed" box was shown instead — the
|
||||
archive had to be extracted again by hand before the prompt would appear. The
|
||||
remembered request is now keyed by **task id**, which the server assigns and
|
||||
which survives the round trip untouched; the retry also re-sends the path the
|
||||
*server* reported rather than the one the page was holding. On-device symptom:
|
||||
upload `x.zip` into a Chinese-named folder, choose "upload and extract" → no
|
||||
prompt, then the raw error.
|
||||
|
||||
- **Entry names in error messages were unreadable mojibake** —
|
||||
`解压失败: 密码错误,或压缩包未使用所提供的密码加密: â®…ç§.psd (entry is
|
||||
encrypted and no password was given)`. The listing has always translated the
|
||||
byte-mapped names back through `decodeFsText()` for display, but
|
||||
`backendErrorText()` used the raw `error_arg` — the one place where the name
|
||||
matters most. Both `error_arg` and the backend's own sentence now go through
|
||||
the same translation, so a GBK name stored inside a ZIP reads as Chinese
|
||||
again.
|
||||
|
||||
- **An archive with a non-ASCII name inside a non-ASCII folder could not be
|
||||
extracted after upload.** `uploadAndExtractFile()` joined the byte-mapped
|
||||
directory with the real Unicode file name, producing a mixed path; the server
|
||||
only byte-repairs a string when *every* non-ASCII code point in it is ≤ 0xFF,
|
||||
so one CJK character made it skip the repair and look for a path that does not
|
||||
exist. `encodeFsText()` (the inverse of `decodeFsText()`) now byte-maps the
|
||||
name before the join, which is also applied to the path `New Text` hands to
|
||||
the editor. Pure-ASCII paths and pure-real-Unicode paths were unaffected,
|
||||
which is why this survived until someone hit the mixed case.
|
||||
|
||||
- **The upload menu's row highlight painted as a broken shape.** Selecting a
|
||||
row drew a 3px blue ring at `outline-offset: 2px` that cleared the panel's
|
||||
6px padding, overlapped the row above, and kept the row's own 6px corner
|
||||
radius instead of growing with the offset — so the highlight read as a
|
||||
detached outline with two arcs hanging off its sides rather than a selected
|
||||
row. Two rules were fighting. The ring came from the generic `button:focus`
|
||||
rule, which is sized for a 54px toolbar button and was never meant to reach a
|
||||
46px list row. Underneath it, the panel's own
|
||||
`.upload-menu-list button:hover:not(:disabled)` fill had **never applied at
|
||||
all**: it ties on specificity (0,3,1) with the generic
|
||||
`button:not(.row-action):hover:not(:disabled)` rule and that one comes later
|
||||
in the file, so it won the cascade — hover and focus therefore ended up two
|
||||
different colours (`#303945` against `#2b343e`), and a hovered row lit up
|
||||
while the keyboard-focused row stayed lit, which is what made two rows look
|
||||
selected at once. Both rules are now scoped to the panel id, rows highlight
|
||||
by fill alone, and the keyboard cue is a 2px **inset** ring drawn inside the
|
||||
row, where it cannot cross the panel edge at any row height.
|
||||
|
||||
### Tests
|
||||
|
||||
- `tests/test_zip_extract.c` runs each encrypted fixture four ways (no
|
||||
password → `PASSWORD`, empty → `PASSWORD`, wrong → `PASSWORD`, correct →
|
||||
`ZIPX_OK` with a byte-level content check), plus a case proving the
|
||||
limits still apply when a password has been handed over.
|
||||
- `tests/test_rar_extract.c` exercises `enc-v6.rar` the same way, including
|
||||
that nothing is published on the failing paths.
|
||||
- `tests/test_sevenz_extract.c` runs `aeshe.7z` three ways: no password →
|
||||
`ZIPX_ERR_PASSWORD`, wrong password → `ZIPX_ERR_PASSWORD`, correct password
|
||||
→ success with the content compared byte for byte, and no staging tree left
|
||||
behind on any of them.
|
||||
- `.build/ui_retry_test.mjs` loads the real `assets/main.js` into a stubbed DOM
|
||||
and checks the frontend retry flow: the request is remembered with its
|
||||
conflict policy and large-file flag, a failure retries with the typed
|
||||
password, cancel and empty input give up, and the retry count caps at three.
|
||||
It also carries the regression case for the missing prompt: a task whose
|
||||
`src`/`dst` come back byte-repaired (a non-ASCII folder) must still be retried,
|
||||
and a structural check that no retry entry is keyed by a path. **40 checks,
|
||||
0 failures.**
|
||||
- `.build/ui_upload_menu_test.mjs` is the markup-side counterpart: it asserts
|
||||
that every `data-i18n` key in `index.html` exists in both language files, that
|
||||
the two language files carry the same keys, that the upload menu and the drag
|
||||
hint exist and start hidden, that the removed file/folder split is gone from
|
||||
both the markup and the stylesheet, that the click handlers are bound to the
|
||||
new ids, and that the classes the markup uses are actually styled. Four of its
|
||||
checks pin the row-highlight cascade: the hover and focus rules must be scoped
|
||||
to the panel, the row ring must be suppressed, the keyboard cue must be an
|
||||
inset shadow, and the unscoped forms must not come back — a rule that silently
|
||||
loses the cascade is exactly the kind of thing a static check can still catch.
|
||||
A further group covers the extract button, which is now always on screen: the
|
||||
markup must not hide it, main.js must never assign to `extractBtn.hidden`, the
|
||||
disabled rule must keep the tooltip hoverable, and there must be a message for
|
||||
each reason it can be unavailable. One check sweeps the other direction —
|
||||
every `t("...")` literal in `main.js` (117 keys) must exist in both language
|
||||
files — because a key reached only from script code is invisible to the markup
|
||||
sweep, which is how a message goes missing unnoticed.
|
||||
**40 checks, 0 failures.** Both scripts are hand-run — the project has no
|
||||
browser test runner — but each one exits non-zero on failure.
|
||||
- `.build/preview_build.py` + `.build/preview_check.mjs` render the real page
|
||||
against a fixture API in headless Chromium and assert what a screenshot alone
|
||||
cannot: the menu is hidden at rest, opens on click, moves focus into the list,
|
||||
reaches the hidden `<input type="file">`, and closes on a choice, an outside
|
||||
click and Escape. It reads back the *computed* highlight for a focused, a
|
||||
hovered and a keyboard-focused row, which is the only way to settle a cascade
|
||||
question — a rule losing to a generic one and a ring leaking out of its
|
||||
container both look fine in the source. It is how the three layout/highlight
|
||||
mistakes of this round were caught (a hint that pushed the toolbar onto a
|
||||
second row, a status line that wrapped out of the footer, and the broken row
|
||||
highlight described under Fixed).
|
||||
- Host totals: **140 ZIP + 37 RAR = 177 checks**, 0 failures.
|
||||
- 7z totals: **27 cases, 0 failures** (`tests/run-sevenz-tests.sh`), and the
|
||||
`KNOWN_GAPS` list that held `aeshe` is now empty — the encrypted-header
|
||||
fixture passes through both the folder decoder and the extraction facade.
|
||||
|
||||
### Still open
|
||||
|
||||
- End-to-end validation of the built ELF on a real console.
|
||||
|
||||
## [v1.9] — 2026-09-05
|
||||
|
||||
**RAR engine replaced: rarlab UnRAR 7.20.1 (v6 / multi-volume / decryption-capable).**
|
||||
|
||||
The vendored dmc_unrar 1.7.0 only dispatched RAR5 compression version 5
|
||||
(`switch(file->version)` case `0x5000`). Archives written by WinRAR 6.x /
|
||||
7.x (algorithm string `v6`, version field `0x5001`) hit the default branch
|
||||
and surfaced as "corrupt archive" — confirmed on a real `v6:8M` archive.
|
||||
v1.9 swaps in the official rarlab UnRAR source (7.20.1) via its
|
||||
C-compatible DLL API, compiled as a static library (`-DRARDLL`, PS5 uses
|
||||
the toolchain's default `libc++`).
|
||||
|
||||
What this enables:
|
||||
|
||||
- **RAR5 "v6" compression** (WinRAR 6/7 archives) — the v1.9 trigger.
|
||||
- **Multi-volume RAR** (`.partNN.rar`): unrar stitches volumes by name when
|
||||
all parts sit next to the opened volume. Select the first volume
|
||||
(`name.part1.rar`) and extract as usual.
|
||||
- RAR4 and older RAR5 remain supported.
|
||||
- The engine *can* decrypt encrypted archives (`RARSetPassword`), but the
|
||||
password channel (API + UI) is not wired yet — encrypted headers/entries
|
||||
still fail up front with `err_extract_unsupported`. Planned for a follow-up.
|
||||
|
||||
Engine changes:
|
||||
|
||||
- `src/rar_extract.c` rewritten to unrar's sequential DLL API
|
||||
(`RAROpenArchiveEx → RARReadHeaderEx → RARProcessFile`); scan and extract
|
||||
each re-open the archive. Multi-volume continuation segments
|
||||
(`RHDF_SPLITBEFORE`) are advanced but not re-counted/deduped.
|
||||
- The three-phase scan → staging → publish/rollback machinery is unchanged.
|
||||
- Bug fix: `normalize_name()` no longer clears the caller's directory flag
|
||||
(a real v6 archive with an explicit directory header after its files
|
||||
tripped the duplicate detector).
|
||||
|
||||
Build & test:
|
||||
|
||||
- Makefile: `.cpp` rules for the unrar RARDLL source set (49 files, mirrors
|
||||
`UnRARDll.vcxproj`); links through the C++ driver; `VERSION_TAG` v1.9.
|
||||
- tests: 5 real RAR fixtures committed under `tests/fixtures-real/`
|
||||
(generated with `tests/make-rar-fixtures.bat` + WinRAR); new happy-path
|
||||
checks extract a real v6 archive, verify files on disk, auto-merge a
|
||||
3-volume split, and reject encrypted archives. Total: **70 ZIP + 24 RAR
|
||||
= 94 checks** (up from 70 + 14; the old 14 RAR checks never ran a real
|
||||
archive).
|
||||
|
||||
Credits: unrar (c) Alexander Roshal, freeware license — see
|
||||
`third_party/unrar7/license.txt` and `THIRD_PARTY_NOTICES`.
|
||||
|
||||
## [v1.8.3] — 2026-09-05
|
||||
|
||||
**Hotfix: "Upload and extract" now accepts `.rar` files.**
|
||||
|
||||
The "upload and extract" entry was hard-coded to accept only `.zip`,
|
||||
even though the server-side dispatch in `src/extract.c:79` already
|
||||
correctly routes `.rar` to `rar_extract()`. v1.8.3 fixes the frontend
|
||||
filter so users can select a single-volume plaintext `.rar` from the
|
||||
file picker and have it uploaded + extracted in one click (the same
|
||||
flow that already worked for `.zip`).
|
||||
|
||||
What this enables:
|
||||
|
||||
- Choose a single-volume `.rar` from "Upload and extract"
|
||||
- Server extracts it via the existing `rar_extract()` engine
|
||||
- Uploaded `.rar` is auto-deleted after a successful extract (same as
|
||||
`.zip` since v1.7)
|
||||
|
||||
What this does **not** enable (planned for v1.9.0):
|
||||
|
||||
- **Multi-volume RAR** (e.g. `name.part01.rar` + `name.part02.rar` …)
|
||||
- **Encrypted RAR** (password-protected headers or entries)
|
||||
|
||||
Both still return `extract_unsupported` "single-volume RAR only" /
|
||||
"encrypted RAR is not supported; please extract on a PC first" — see
|
||||
the underlying engine limit in `third_party/unrar/dmc_unrar` (GPL-2.0,
|
||||
1.7.0). v1.9 will swap the vendor to **opello/unrar 7.20.1** (UnRAR
|
||||
License) which natively supports both.
|
||||
|
||||
Changed:
|
||||
|
||||
- `assets/index.html` — `<input id="uploadZip" accept>` now lists
|
||||
`.rar` + the two RAR MIME types next to the existing ZIP entries.
|
||||
- `assets/main.js:2340` — `/\.zip$/i` → `/\.(zip|rar)$/i` (the upload
|
||||
pre-check), plus a local `isRar` flag so the next step branches.
|
||||
- `assets/lang-en.js` — `extractUploadConfirm`: "uploaded ZIP" →
|
||||
"uploaded archive".
|
||||
- `assets/lang-zh.js` — `extractUploadConfirm` & `extractLargeAsk`
|
||||
drop the "ZIP" wording so the copy reads sensibly for RAR uploads.
|
||||
- `assets/main.js:38` — `APP_VERSION` `"v1.7"` → `"v1.8.3"` (footer
|
||||
version string had been hard-coded to v1.7 since the frontend was
|
||||
first imported; it no longer misleads about which build is running).
|
||||
|
||||
No backend changes — the server side was already correct. No test
|
||||
changes — the existing RAR happy-path test in `tests/test_rar_extract.c`
|
||||
passes against the same backend.
|
||||
|
||||
Build pipeline (also v1.8.3):
|
||||
|
||||
- `.build/build-elf.sh` step 6 "no source change → skip make" check now
|
||||
also watches `assets/*` and `gen-asset-module.py`, not just `src/*.c`
|
||||
and the `Makefile`. Without this, v1.8.3 (which touched no backend,
|
||||
only frontend assets feeding `gen/*.c`) was misclassified as "no
|
||||
change" and `make` was skipped — the result was that the v1.8.2 ELF
|
||||
was reported as v1.8.3 with the same sha256. With this fix, only
|
||||
frontend changes correctly trigger a rebuild. Users running the WSL
|
||||
build need to re-copy `.build/build-elf.sh` to `/home/song/build-elf.sh`
|
||||
(canonical source is on the Windows side).
|
||||
|
||||
## [v1.8.2] — 2026-09-05
|
||||
|
||||
**Hotfix: default ZIP extraction limits cover 3A-game single-file archives.**
|
||||
|
||||
The default `k_default_limits` profile is now **2 TiB total / 512 GiB per
|
||||
entry / 500 : 1 ratio** (was 1 TiB / 256 GiB / 500 : 1 in v1.8.1). The
|
||||
frontend threshold `LARGE_FILE_THRESHOLD_BYTES` is bumped from 240 GiB to
|
||||
**480 GiB** to match. The `large=1` profile is widened to **4 TiB total
|
||||
/ 1 TiB per entry / 1000 : 1 ratio** (was 2 TiB / 1 TiB / 1000 : 1); the
|
||||
large profile must always be strictly more permissive than default.
|
||||
|
||||
Why: a 3A-game archive with a single ~300 GiB uncompressed file was
|
||||
**silently rejected by the default profile** (`scan_archive` returns
|
||||
`ZIPX_ERR_LIMIT_FILE_SIZE` in `src/zip_extract.c` line ~717 — the request
|
||||
never reaches the frontend confirmation prompt, so the user just sees
|
||||
"卡壳"). The 256 GiB default cap was tuned for PS5 system backups (which
|
||||
have many smaller entries, not a single huge file) and was wrong for the
|
||||
3A-game single-file case. The default cap is now 512 GiB so a typical
|
||||
3A archive extracts under the default profile without prompting.
|
||||
|
||||
The safety argument is unchanged from v1.8.1: zip-bomb defence is
|
||||
`check_space()` (`statvfs`-based real disk space check before staging) +
|
||||
`max_ratio` (declared compression ratio cap). The size caps are a UX
|
||||
guard, not a security boundary.
|
||||
|
||||
RAR extraction inherits the new defaults automatically — rar_extract.c
|
||||
threads `c->limits` through from the engine, so no rar-side change is
|
||||
required.
|
||||
|
||||
### Changed
|
||||
|
||||
- `src/zip_extract.c` — `k_default_limits` relaxed:
|
||||
- `max_total_bytes`: 1 TiB → **2 TiB**
|
||||
- `max_file_bytes`: 256 GiB → **512 GiB**
|
||||
- `max_ratio`: 500 → **500** (unchanged)
|
||||
- `src/zip_extract.c` — `k_large_limits` widened (must stay > default):
|
||||
- `max_total_bytes`: 2 TiB → **4 TiB**
|
||||
- `max_file_bytes`: 1 TiB → **1 TiB** (unchanged)
|
||||
- `max_ratio`: 1000 → **1000** (unchanged)
|
||||
- `assets/main.js` — `LARGE_FILE_THRESHOLD_BYTES`: 240 GiB → **480 GiB**
|
||||
- `assets/lang-{en,zh}.js` — `extractLargeAsk` copy updated to reflect
|
||||
the new numbers (default 512 GiB / 2 TiB; large 1 TiB / 4 TiB)
|
||||
- `tests/test_zip_extract.c` — `test_large_profile` advertised-number
|
||||
assertion updated: `max_total_bytes == 4 TiB` (was 2 TiB)
|
||||
- `README.md` — both limit tables (ZIP + RAR), the "Tuning the threshold"
|
||||
snippet, and the `err_extract_entry_too_large` FAQ entry bumped to the
|
||||
new numbers
|
||||
- `docs/HANDOVER.md` — `LARGE_FILE_THRESHOLD_BYTES`, the
|
||||
`k_default_limits` / `k_large_limits` ASCII diagram, the user-scenario
|
||||
description, the 480 GiB popup note, and the RAR limits paragraph
|
||||
bumped to the new numbers
|
||||
|
||||
### Unchanged
|
||||
|
||||
- `src/rar_extract.c` — already threads `c->limits` from the engine,
|
||||
picks up the new defaults for free
|
||||
- `max_ratio` — both profiles unchanged (500 : 1 default / 1000 : 1 large)
|
||||
- `check_space()` — unchanged; still the real disk-space guard
|
||||
- `max_entries` — 200 000 default / 500 000 large, unchanged
|
||||
- `docs/UPGRADE-v1.7-zip-large-file-profile.md` — historical v1.7
|
||||
document left as-is so the v1.7 → v1.8.2 evolution is traceable
|
||||
- Test fixture `medium_bomb.zip` (ratio ≈ 238) still exercises both
|
||||
rejection under the default 500 : 1 cap and acceptance under the
|
||||
`large=1` 1000 : 1 cap
|
||||
- 84 host-side checks (70 ZIP + 14 RAR), 0 failures
|
||||
|
||||
### Migration notes
|
||||
|
||||
- **Forward-compatible** — users with v1.8.1 deployments who never trigger
|
||||
`err_extract_entry_too_large` see no difference (defaults are strictly
|
||||
more permissive)
|
||||
- **3A-game single-file archives now extract silently** — no prompt, no
|
||||
manual `large=1` API call required for files up to 512 GiB
|
||||
- **No data loss** — the relaxation only widens accepted archives; the
|
||||
real security guards (`check_space`, `max_ratio`, `path traversal`)
|
||||
are untouched
|
||||
- **No frontend UX change for typical use** — only archives > 480 GiB
|
||||
on disk now trigger the confirmation prompt (previously 240 GiB)
|
||||
|
||||
---
|
||||
|
||||
## [v1.8.1] — 2026-09-05
|
||||
|
||||
**Hotfix: relaxed default ZIP extraction limits.**
|
||||
|
||||
The default `k_default_limits` profile is now **1 TiB total / 256 GiB per
|
||||
entry / 500 : 1 ratio** (was 512 GiB / 64 GiB / 200 : 1). The frontend
|
||||
threshold `LARGE_FILE_THRESHOLD_BYTES` is bumped from 60 GiB to 240 GiB
|
||||
to match. The `large=1` profile (1 TiB / 1 TiB / 1000 : 1) is unchanged.
|
||||
|
||||
Why: the previous default was a UX-oriented early-fail guard, not a
|
||||
security guard — `check_space()` already enforces available ≥ bytes_total
|
||||
before staging begins, and `max_ratio` already rejects classic zip
|
||||
bombs. A user with a multi-hundred-GiB system image shouldn't have to
|
||||
click through a confirmation prompt for what's a perfectly safe archive.
|
||||
The relaxed default still rejects any archive whose declared
|
||||
uncompressed total exceeds the destination's free space (real check,
|
||||
not a declared-vs-fs assertion) and any archive with a declared ratio
|
||||
above 500 : 1 (real zip-bomb guard).
|
||||
|
||||
RAR extraction inherits the new defaults automatically — rar_extract.c
|
||||
threads `c->limits` through from the engine, so no rar-side change is
|
||||
required.
|
||||
|
||||
### Changed
|
||||
|
||||
- `src/zip_extract.c` — `k_default_limits` relaxed:
|
||||
- `max_total_bytes`: 512 GiB → **1 TiB**
|
||||
- `max_file_bytes`: 64 GiB → **256 GiB**
|
||||
- `max_ratio`: 200 → **500**
|
||||
- `assets/main.js` — `LARGE_FILE_THRESHOLD_BYTES`: 60 GiB → **240 GiB**
|
||||
- `assets/lang-{en,zh}.js` — `extractLargeAsk` default-profile copy
|
||||
updated to reflect the new numbers
|
||||
- `README.md` — "Stricter default ZIP profile" line, the limit table
|
||||
(two locations), and the `err_extract_entry_too_large` FAQ entry
|
||||
bumped to the new numbers; "Tuning the threshold" snippet updated to
|
||||
240 GiB
|
||||
- `docs/HANDOVER.md` and `docs/UPGRADE-v1.8-rar-support.md` — the
|
||||
few remaining numeric references in those docs updated
|
||||
|
||||
### Unchanged
|
||||
|
||||
- `src/rar_extract.c` — already threads `c->limits` from the engine,
|
||||
picks up the new defaults for free
|
||||
- `k_large_limits` — `large=1` profile (1 TiB / 1 TiB / 1000 : 1) is
|
||||
unchanged
|
||||
- `docs/UPGRADE-v1.7-zip-large-file-profile.md` — historical v1.7
|
||||
document left as-is so the v1.7 → v1.8.1 evolution is traceable
|
||||
- Test fixture `medium_bomb.zip` (ratio ≈ 238) still exercises both
|
||||
rejection under the default 500 : 1 cap and acceptance under the
|
||||
`large=1` 1000 : 1 cap
|
||||
- 83 host-side checks (69 ZIP + 14 RAR), 0 failures
|
||||
|
||||
### Migration notes
|
||||
|
||||
- **Forward-compatible** — users with v1.7 / v1.8 deployments who never
|
||||
trigger `err_extract_entry_too_large` see no difference (defaults are
|
||||
strictly more permissive)
|
||||
- **No data loss** — the relaxation only widens accepted archives; the
|
||||
real security guards (`check_space`, `max_ratio`, `path traversal`)
|
||||
are untouched
|
||||
- **No frontend UX change for typical use** — only archives > 240 GiB
|
||||
on disk now trigger the confirmation prompt (previously 60 GiB)
|
||||
|
||||
---
|
||||
|
||||
## [v1.8] — 2026-09-05
|
||||
|
||||
**RAR extraction: single-volume RAR4 / RAR5 (unencrypted).**
|
||||
|
||||
> ⚠️ Scope clarification — v1.8 ships **single-volume unencrypted RAR** only.
|
||||
> The original RAR wishlist (multi-volume `.partNN.rar`, encrypted RAR with
|
||||
> password UI) is **deferred to v1.9**; see `docs/UPGRADE-v1.8-rar-support.md`
|
||||
> and `third_party/unrar/VENDORED.md` for the rationale and the engine
|
||||
> upgrade path. ZIP behaviour and the large-file profile are unchanged.
|
||||
|
||||
### Added
|
||||
|
||||
- **`src/rar_extract.{c,h}`** — a new extraction engine that mirrors
|
||||
`zip_extract`'s protocol exactly. Internally it wraps the vendored
|
||||
`dmc_unrar` 1.7.0 (`third_party/unrar/dmc_unrar.c`). The public entry point
|
||||
is `rar_extract(rar_path, dst_dir, conflict, limits, cancel, progress,
|
||||
userdata, result)` — same signatures, same `zipx_status_t` codes, same
|
||||
`zipx_result_t`, same `zipx_limits_t` profile lookup. ~1276 LOC
|
||||
(`src/rar_extract.c`).
|
||||
- **`third_party/unrar/`**:
|
||||
- `dmc_unrar.c` — vendored verbatim from upstream (11 598 LOC, ~365 KiB).
|
||||
GPL-2.0-or-later, attributed in `THIRD_PARTY_NOTICES`.
|
||||
- `dmc_unrar_api.h` — **project-authored facade header**. Re-declares only
|
||||
the `dmc_unrar_*` symbols `rar_extract.c` actually uses, so the engine
|
||||
can `#include "dmc_unrar_api.h"` instead of `#include "dmc_unrar.c"`.
|
||||
This keeps the vendored `.c` compiling as its own translation unit and
|
||||
avoids polluting dmc_unrar's struct / function names with any
|
||||
build-system macros (see `tests/posix_compat.h`'s `wfm_open` /
|
||||
`wfm_close` rename pattern — that conflict is what motivated the
|
||||
facade). The facade carries the project's license; the library body
|
||||
remains unmodified.
|
||||
- `COPYING`, `README.md` — upstream GPL notice + readme.
|
||||
- `VENDORED.md` — explains why we chose `dmc_unrar` over rarlab UnRAR /
|
||||
`opello/unrar`, what is and is not supported, and gives a step-by-step
|
||||
upgrade plan for moving to a fuller C++ UnRAR in v1.9.
|
||||
- **`src/extract.c` dispatch layer** — `extract_dispatch()` picks the
|
||||
engine by extension (`.zip` → `zipx_extract`, `.rar` → `rar_extract`,
|
||||
anything else → `ZIPX_ERR_UNSUPPORTED`). The case-insensitive suffix
|
||||
matcher trims trailing path separators. `extract_worker` now calls the
|
||||
dispatch instead of going straight to the ZIP engine.
|
||||
- **Frontend + i18n wiring**:
|
||||
- `assets/main.js` recognises `.rar` (single-volume) and `.part0*1.rar`
|
||||
(multi-volume master) as extractable, greys out the extract button on
|
||||
`.part02+.rar` sub-volumes with a tooltip "select the main volume
|
||||
instead". This UX is only useful because v1.8 still rejects
|
||||
multi-volume RAR with a friendly error — the visual feedback stops the
|
||||
user from selecting a sub-volume and getting confused.
|
||||
- `assets/lang-{en,zh}.js` `err_extract_unsupported` updated to:
|
||||
"…(only unencrypted plain ZIP and single-volume RAR are supported)…".
|
||||
- **Host test suite** (`tests/test_rar_extract.c`, **14 checks**) —
|
||||
negative paths only (format dispatch, error translation, limits
|
||||
handoff). The suite is wired into `tests/run-tests.sh` alongside the
|
||||
existing ZIP suite; fixture generation falls back to a placeholder
|
||||
blob when no `rar` / `7z` writer is present, so the negative tests
|
||||
fire on any host. Total host checks: **69 ZIP + 14 RAR = 83**.
|
||||
|
||||
### Changed
|
||||
|
||||
- **`Makefile`**:
|
||||
- `VERSION_TAG := v1.8` (was `v1.7`).
|
||||
- `THIRD_PARTY_SRCS` adds `third_party/unrar/dmc_unrar.c`.
|
||||
- `THIRD_PARTY_CFLAGS` adds `-Ithird_party/unrar` and
|
||||
`-DDMC_UNRAR_DISABLE_BE32TOH_BE64TOH=1` (dmc_unrar's own byte-swap
|
||||
helpers are avoided so we don't need an extra `byteswap.h` shim on
|
||||
the SDK).
|
||||
- **`src/extract.c` / `src/extract.h`** — no public-API break. The HTTP
|
||||
surface (`POST /api/extract`) accepts the same fields as v1.7 plus
|
||||
the existing `large=1`; there is **no** `password=` field because
|
||||
v1.8 cannot decrypt. (The wire format is forward-compatible — a v1.9
|
||||
`password=` field will be additive.)
|
||||
- **`THIRD_PARTY_NOTICES`** — adds a section `3. dmc_unrar` crediting
|
||||
Sven Hesse (DrMcCoy), summarising the GPL-2.0-or-later obligations on
|
||||
the resulting binary, and noting that `dmc_unrar_api.h` is
|
||||
project-authored and licensed with the project.
|
||||
|
||||
### Limitations (v1.8 scope)
|
||||
|
||||
- **Multi-volume RAR** (`.part02+.rar`, `.part1+.rar`, numbered
|
||||
continuations) is rejected with `ZIPX_ERR_UNSUPPORTED` and the error
|
||||
message "extract on a PC first". Upstream dmc_unrar does not chain
|
||||
companion volumes by design. When opello/unrar replaces dmc_unrar
|
||||
in v1.9 this becomes a one-line error-code drop.
|
||||
- **Encrypted RAR** (any encrypted header / file flag) is rejected with
|
||||
`ZIPX_ERR_UNSUPPORTED`. Same root cause — dmc_unrar omits decryption
|
||||
to avoid patent complications. There is **no** password prompt in
|
||||
the UI; there is **no** `password=` field in `/api/extract`.
|
||||
- **Symbolic links, FIFOs, sockets, devices** inside a RAR archive are
|
||||
rejected with `ZIPX_ERR_SPECIAL` (mirrors ZIP behaviour).
|
||||
- **RAR 1.4** (very old) is not supported by dmc_unrar and is rejected
|
||||
upstream with `DMC_UNRAR_ARCHIVE_VERSION_UNSUPPORTED`; the wrapper
|
||||
maps that to `ZIPX_ERR_UNSUPPORTED`. RAR 1.5 through RAR 5.0 are
|
||||
supported.
|
||||
|
||||
### Verification
|
||||
|
||||
```sh
|
||||
make # builds web-file-mgr.elf (in WSL)
|
||||
ls -la web-file-mgr.elf # record size
|
||||
sha256sum web-file-mgr.elf # record digest (paste into the v1.8 banner above)
|
||||
file web-file-mgr.elf # ELF 64-bit LSB pie, x86-64
|
||||
od -An -tx1 -N20 web-file-mgr.elf # 7f45 4c46 0201 + e_machine 003e
|
||||
|
||||
(cd tests && bash run-tests.sh) # 69 + 14 = 83 checks, 0 failures
|
||||
|
||||
python3 .build/check-elf-gzip.py ./web-file-mgr.elf # 7/7 v1.7 keys + 1 v1.8 key (err_extract_unsupported)
|
||||
```
|
||||
|
||||
### Technical notes
|
||||
|
||||
A long-form technical write-up of this upgrade lives in
|
||||
[`docs/UPGRADE-v1.8-rar-support.md`](./docs/UPGRADE-v1.8-rar-support.md).
|
||||
The vendoring decision tree (and the v1.9 plan) is in
|
||||
[`third_party/unrar/VENDORED.md`](./third_party/unrar/VENDORED.md).
|
||||
|
||||
### Credits
|
||||
|
||||
Same as v1.7 — see [README.md → Credits](./README.md#credits). dmc_unrar
|
||||
is credited in [`THIRD_PARTY_NOTICES`](./THIRD_PARTY_NOTICES).
|
||||
|
||||
---
|
||||
|
||||
## [v1.7] — 2026-09-04
|
||||
|
||||
**ZIP extraction: large-file profile, three-phase engine, host test suite.**
|
||||
|
||||
### Added
|
||||
|
||||
- **Large-file profile** (`ZIPX_LIMITS_LARGE`) for ZIP extraction — relaxed
|
||||
caps of **500 000** entries, **2 TiB** total uncompressed, **1 TiB** per
|
||||
entry, **1000 : 1** compression ratio. Enable via the new `large=1`
|
||||
argument on `POST /api/extract`. The UI prompts the user automatically
|
||||
whenever the archive on disk is larger than 60 GiB (`LARGE_FILE_THRESHOLD_BYTES`).
|
||||
- **Standalone three-phase ZIP engine** (`src/zip_extract.{c,h}`) — the model
|
||||
`SCAN → EXTRACT → PUBLISH → CLEANUP`. Each entry is written into a staging
|
||||
directory first, fsynced, then atomically renamed into the destination. Any
|
||||
mid-archive failure rolls back partial changes.
|
||||
- **Profile lookup helper** `zipx_limits_profile(int)` and the new macros
|
||||
`ZIPX_LIMITS_DEFAULT` (0) and `ZIPX_LIMITS_LARGE` (1). The public
|
||||
`zipx_extract()` signature is unchanged.
|
||||
- **Opt-in API field** `large` on `POST /api/extract`, backed by a new
|
||||
`extract_large` flag in `file_task_t` (`src/filemgr_internal.h`,
|
||||
`src/extract.c`).
|
||||
- **Frontend wiring** (`assets/main.js`) — `LARGE_FILE_THRESHOLD_BYTES`,
|
||||
`shouldPromptLargeMode()`, `promptLargeMode()`, consumed by
|
||||
`actionExtract()` and `uploadAndExtractFile()`.
|
||||
- **Bilingual UI strings** (`assets/lang-{en,zh}.js`) —
|
||||
`extractLargeAsk` and `extractLargeActive`.
|
||||
- **Host-side C test suite** (`tests/`) — POSIX-runnable, no PS5 SDK
|
||||
required. **69 checks** at release: traversal, ZIP64, encryption rejection,
|
||||
conflict policies, large-file profile switching.
|
||||
- **Cross-compile helper** (`.build/build-elf.sh`) — staging-mode build that
|
||||
works around the read-only SDK install location.
|
||||
- **Demo page** (`.build/extract-demo.html`) — visualises the three policy
|
||||
combinations (fail / overwrite / merge) against the new profile table.
|
||||
|
||||
### Changed
|
||||
|
||||
- **Engine internals** rewritten around the three-phase model. Public API
|
||||
(`zipx_extract()`, `zipx_default_limits()`) is unchanged — old callers
|
||||
compile and link clean.
|
||||
- **`file_task_t`** gains `extract_large` (`src/filemgr_internal.h`).
|
||||
Internal-only; downstream consumers reading the task struct need to
|
||||
recognise the new field.
|
||||
- **`gen-asset-module.py`** continues to gzip JS into the binary; the
|
||||
helpers in `.build/check-elf-gzip.py` are the supported way to verify that
|
||||
a fresh build picked up asset changes (regular `strings` won't see them).
|
||||
|
||||
### Security
|
||||
|
||||
- Default ZIP caps are unchanged: **512 GiB** total / **64 GiB** per entry /
|
||||
**200 : 1** ratio.
|
||||
- The large profile is **never** activated server-side on its own — the
|
||||
client must explicitly send `large=1` (either via the UI prompt or directly
|
||||
via the API).
|
||||
- Encryption, path traversal, symbolic links, FIFOs and unresolved conflicts
|
||||
are still rejected before any output file is opened.
|
||||
- Strict value matching: only the literal `"1"` enables the large profile;
|
||||
`true`, `yes`, `on` are all treated as `0`. Matches the existing
|
||||
`remove_source=` semantics.
|
||||
|
||||
### Known limitations
|
||||
|
||||
- Multi-volume / split ZIP archives (`.zip` + `.z01`, `.z02`, …) are not
|
||||
stitched by the engine. minizip-ng has the API; wiring it is post-v1.7.
|
||||
- The 60 GiB frontend threshold for the large-profile prompt is hardcoded
|
||||
(`LARGE_FILE_THRESHOLD_BYTES`, `assets/main.js` line 813).
|
||||
- The large profile does **not** run `statvfs()` against `dst_dir` before
|
||||
extraction; free-space preflight is on the post-v1.7 roadmap.
|
||||
- Bomb-shaped archives with compression ratio **> 1000 : 1** are rejected
|
||||
under both profiles (`bomb.zip` with 4 MiB of `'A'` has ratio ≈ 1026 and
|
||||
hits this wall under the large profile).
|
||||
|
||||
### Verification
|
||||
|
||||
```sh
|
||||
make # builds web-file-mgr.elf
|
||||
ls -la web-file-mgr.elf
|
||||
sha256sum web-file-mgr.elf # 648e4a00…
|
||||
file web-file-mgr.elf # ELF 64-bit LSB pie, x86-64
|
||||
od -An -tx1 -N20 web-file-mgr.elf # 7f45 4c46 0201 + e_machine 003e
|
||||
|
||||
(cd tests && bash run-tests.sh) # 69 checks, 0 failures
|
||||
|
||||
python3 .build/check-elf-gzip.py ./web-file-mgr.elf # 7/7 large-mode keys in ELF
|
||||
```
|
||||
|
||||
### Technical notes
|
||||
|
||||
A long-form technical write-up of this upgrade (architecture delta, three-phase
|
||||
model, behaviour contract, trade-offs, future work) lives in
|
||||
[`docs/UPGRADE-v1.7-zip-large-file-profile.md`](./docs/UPGRADE-v1.7-zip-large-file-profile.md).
|
||||
|
||||
### Credits
|
||||
|
||||
The same as v1.6 — see [README.md → Credits](./README.md#credits).
|
||||
@@ -0,0 +1,543 @@
|
||||
# 交接文档 — ps5-web-file-manager 工作进度
|
||||
|
||||
> 交接时间:2026-09-23 · 分支 `main` · 最新提交仍是 **`3f80eb4`**(tag `v1.9.2`)——**本轮加密改动全部尚未提交**
|
||||
>
|
||||
> **主线(用户 2026-09-12 指令)**:「先从 zip 分卷开始吧,然后把六种组合打齐,并把密码通道补齐,注意一些报错信息提示的时候尽量详细准确」
|
||||
> **状态:主线全部闭合,且格式面已无已知缺口。** 六种组合(ZIP/RAR/7z × 单卷/分卷)+ 密码通道(ZIP ZipCrypto/AES + RAR `-p`/`-hp` + 7zAES **含 `-mhe=on` 加密头**)+ 报错详细信息,全部落地、测试全绿、PS5 ELF 构建成功。
|
||||
> **剩余**:① PS5 真机端到端验证(**唯一还没过的关卡**);② 版本号仍是 `v1.9.2`,本轮改动处于「未发布」状态——发版需先升 `VERSION_TAG` 与 `APP_VERSION_FALLBACK`。
|
||||
|
||||
---
|
||||
|
||||
## 一、当前状态速览
|
||||
|
||||
| 维度 | 状态 |
|
||||
|---|---|
|
||||
| 解压引擎 | ZIP / RAR / 7z × 单卷/分卷(6 组合)+ 三种加密(ZIP ZipCrypto/WinZipAES、RAR `-p`/`-hp`、7zAES 与 `-mhe=on` 加密头)全部打通 |
|
||||
| 主机测试 | **ZIP 140 + RAR 37 = 177 checks,0 失败**(MinGW gcc;2026-09-23 复跑)+ 7z 套件 **27 用例 0 失败**(`KNOWN_GAPS` 已清空)+ 前端重试流程 27 checks(`.build/ui_retry_test.mjs`) |
|
||||
| PS5 构建 | ✅ WSL prospero-clang 18.1.8,一键脚本可复现;构建已实测**确定性**(同源两次构建 sha256 相同) |
|
||||
| ELF 产物(工作树,未提交) | `web-file-mgr-v1.9.3M.elf` · 903,448 B · sha256 `8ca47d5aaca75085b32641300cce30fadb7df7749cb6b53d04f129bcecc286b7` · e_machine=0x003e(2026-09-24 最后一轮:上传菜单 + 拖拽提示 + 口令重试改按键 id + 错误文案编码修复 + 菜单行高亮的层叠修复 + 解压按钮常显置灰;**未发布**) |
|
||||
| 已发布产物 | `web-file-mgr-v1.9.2.elf` · 870,488 B · sha256 `177e90fecf93a0251e83f67884fba4551051be248330b0d70fda8ea732f88e84`(不含加密改动) |
|
||||
| GitHub | `main`(`3f80eb4`)与 tag `v1.9.2` 均已推送;Release `v1.9.2` 资产对应上一行;本轮改动尚未 commit |
|
||||
| 已知功能缺口 | **无**(`-mhe=on` 已于 2026-09-23 补齐) |
|
||||
|
||||
### 发布命令(v1.9.2)
|
||||
|
||||
```bash
|
||||
cd "/c/Users/songl/Desktop/Web File Manager/ps5-web-file-manager"
|
||||
|
||||
git push origin main
|
||||
git push origin v1.9.2
|
||||
|
||||
gh release create v1.9.2 web-file-mgr-v1.9.2.elf \
|
||||
--repo LisherSong/ps5-web-file-manager --title "v1.9.2" --notes-file <notes.md>
|
||||
```
|
||||
|
||||
沙箱内 git 出站 HTTPS **可用**(早先记的"被拦"是误判:`timeout 25 git ...`
|
||||
命中的是 `C:\Windows\System32\TIMEOUT.EXE`,报参数错误而非网络错误)。`gh` 同样可用,
|
||||
所以 commit / tag / push / 发 Release 都可以在会话里直接跑。
|
||||
|
||||
---
|
||||
|
||||
## 二、本轮(2026-09-15)变更
|
||||
|
||||
### 2.1 一键构建脚本(`a54f34b` / `5229cd5`)
|
||||
|
||||
| 文件 | 跑在哪 | 作用 |
|
||||
|---|---|---|
|
||||
| `.build/build-win.sh` | Windows Git Bash | 入口:把 WSL 脚本经 **stdin** 喂给 `wsl.exe`,透传退出码 |
|
||||
| `.build/build-elf-wsl.sh` | WSL Ubuntu-22.04 | 5 阶段:环境检查 → rsync 同步 → `make all` → 验证 → 拷回 Windows |
|
||||
| `.build/build-elf.sh` | WSL 内 | **仅首次搭环境用**(libmicrohttpd staging 安装 + sudo) |
|
||||
|
||||
```bash
|
||||
# Windows 端(注意:必须 /usr/bin/bash,裸 bash 会解析成 WSL 启动器)
|
||||
cd "/c/Users/songl/Desktop/Web File Manager/ps5-web-file-manager"
|
||||
/usr/bin/bash .build/build-win.sh
|
||||
|
||||
# 或 WSL 内
|
||||
bash /home/song/build.sh
|
||||
```
|
||||
|
||||
**两个关键实现点**(改脚本前必读):
|
||||
- `wsl.exe -- bash -c '...'` 遇到含空格路径会被拆断 → 必须 `wsl.exe -- bash < script.sh`(stdin 重定向)
|
||||
- `make ... | tail` 会吞掉退出码 → 用 `${PIPESTATUS[0]}`;否则编译失败还会继续跑验证,输出假成功
|
||||
|
||||
### 2.2 版本号体系统一(`f4fd464` / `1fa2f09` / `0d036a7`)
|
||||
|
||||
原本版本号有**两个真相来源**,已经漂移过:`Makefile` 写 `v1.9.1`,`assets/main.js` 硬编 `"v1.9"`,UI 右下角在整个 v1.9.1 发布期都显示旧值。
|
||||
|
||||
现在收敛到 `Makefile` 一处:
|
||||
|
||||
```make
|
||||
VERSION_TAG ?= v1.9.3M # 可用 make VERSION_TAG=v1.9.3 临时覆盖
|
||||
BIN := web-file-mgr-$(VERSION_TAG).elf
|
||||
```
|
||||
|
||||
改这一行会同时影响 **四处**:ELF 内嵌版本串、PS5 启动通知、输出文件名、UI 右下角。
|
||||
|
||||
**尾部 `M` = 改版标记(Modified,LisherSong 维护)**,从 v1.9.3M 起启用。上游
|
||||
owendswang 的发布版是纯 `vX.Y.Z`,故「带 M = 本仓、不带 = 上游」一眼可分。它刻意
|
||||
挂在 `VERSION_TAG` 上而不是做一个只管显示的独立常量:这样 `/api/version`、启动通知、
|
||||
stdout 横幅、UI 右下角、ELF 文件名**五处一次性全覆盖**,不可能只在其中一处漏掉。
|
||||
附带好处是产物名不再可能与上游同版本号的资产撞车(此前已撞过两次:本地
|
||||
`web-file-mgr-v1.9.2.elf` 与线上同名资产并存;`-DVERSION_TAG=v1.9.1` 的残留产物
|
||||
和已发布的 870 488 B 文件尺寸相同)。
|
||||
|
||||
| 提交 | 内容 |
|
||||
|---|---|
|
||||
| `f4fd464` | `VERSION_TAG` `v1.9` → `v1.9.1`;`.gitignore` 改白名单式(`.build/*` 全忽略 + `!` 放行 6 个脚本),`.workbuddy/` 也忽略 |
|
||||
| `1fa2f09` | 输出文件名派生自 `VERSION_TAG`;`build-elf-wsl.sh` 从 Makefile 反读版本(不硬编);`build-win.sh` 每次都推 WSL 脚本(原来只在缺失时推,导致改了脚本 WSL 侧仍跑旧版) |
|
||||
| `0d036a7` | **新增 `/api/version`**,前端右下角改从后端取值(见 2.3) |
|
||||
|
||||
⚠️ **`assets/main.js` 里还有第二处字面量** `APP_VERSION_FALLBACK`(`/api/version` 取不到时的兜底值)。
|
||||
它不在 Makefile 的控制范围内 —— **升版本号时必须一并改**,否则后端请求失败时页脚会显示旧版本。
|
||||
v1.9.2 就是这两处一起改的。
|
||||
|
||||
### 2.3 UI 版本号改由后端提供(`0d036a7`)
|
||||
|
||||
**问题**:`assets/main.js:38` 的 `const APP_VERSION = "v1.9"` 与 Makefile 无关,必然漂移。
|
||||
|
||||
**修法**:
|
||||
- 新 `src/version.c` — `GET /api/version` → `{"ok":true,"version":"v1.9.2","titleId":"FMGR88888"}`,直接来自 Makefile 已传的 `-DVERSION_TAG` / `-DTITLE_ID` 宏,没有第二处要记得改
|
||||
- `src/filemgr.c` 路由表加一行(紧邻 `/api/space`)+ `filemgr_internal.h` 声明 + `Makefile` `COMMON_SRCS`
|
||||
- 前端:字面量降级为 `APP_VERSION_FALLBACK`(先渲染,保证页脚不空),`loadVersion()` 后台刷新。**请求失败静默吞掉** —— 版本号显示错属于装饰性问题,不该弹错误 toast
|
||||
|
||||
**踩坑**:新文件漏了 `#include "json_util.h"` → `strbuf_append` / `json_escape` 隐式声明报错。`space.c` 是模板,照抄时别漏。
|
||||
|
||||
---
|
||||
|
||||
## 三、功能矩阵与测试
|
||||
|
||||
### 3.1 六种组合 + 密码通道
|
||||
|
||||
| # | 引擎 | 单卷 | 分卷 | 密码 |
|
||||
|---|---|---|---|---|
|
||||
| ① | ZIP | ✅ | ✅ 三种命名约定 | ✅ ZipCrypto + WinZip AES-128/192/256(2026-09-23 打通,此前 `mz_zip.c` 的加密分支没有后端可调) |
|
||||
| ② | RAR | ✅ | ✅(vendor unrar 7.20.1) | ✅ `RARSetPassword`(`-p` 与 `-hp` 头加密;2026-09-23 接线,此前从未被调用) |
|
||||
| ③ | 7z | ✅ | ✅ `.7z.001` | ✅ 7zAES(v1.9 起就有)+ `-mhe=on` 加密头(2026-09-23 打通,见 §八) |
|
||||
|
||||
### 3.2 测试
|
||||
|
||||
```bash
|
||||
export PATH="/c/mingw64/bin:/c/Users/songl/.workbuddy/binaries/PortableGit/versions/1.2.0/mingw64/bin:/c/Users/songl/.workbuddy/binaries/python/versions/3.13.12:/usr/bin:/bin:/c/Windows/System32:/c/Windows"
|
||||
cd "/c/Users/songl/Desktop/Web File Manager/ps5-web-file-manager"
|
||||
/usr/bin/bash tests/run-tests.sh # ZIP 140 + RAR 37 = 177
|
||||
/usr/bin/bash tests/run-sevenz-tests.sh # 7z 27(KNOWN_GAPS 已清空)
|
||||
|
||||
# 前端「密码失败后重试」流程(桩 DOM,无需浏览器)
|
||||
"/c/Users/songl/.workbuddy/binaries/node/versions/22.22.2-3/node.exe" .build/ui_retry_test.mjs # 27
|
||||
```
|
||||
|
||||
⚠️ **绝不写裸 `bash`** —— 可能解析到 `C:\Windows\System32\bash.exe`(WSL 启动器),脚本跑进 Linux,gcc/python 全变 Linux 版,报莫名错误。必须 `/usr/bin/bash`。
|
||||
|
||||
`run-sevenz-tests.sh` 带 `KNOWN_GAPS` 列表(当前仅 `aeshe`),缺口修好后脚本会主动报错,防止列表腐烂。脚本**不做任何删除**(safe-delete 钩子会拦 `rm -rf`)。
|
||||
|
||||
---
|
||||
|
||||
## 四、构建(PS5 ELF)
|
||||
|
||||
### 4.1 日常构建
|
||||
|
||||
```bash
|
||||
cd "/c/Users/songl/Desktop/Web File Manager/ps5-web-file-manager"
|
||||
/usr/bin/bash .build/build-win.sh
|
||||
```
|
||||
|
||||
增量有效(`ps5-obj/` 缓存保留)→ 二次构建 30s–2min。**别 `make clean`**(全量重编第三方 3–5min)。
|
||||
|
||||
产物:项目根 `web-file-mgr-<VERSION_TAG>.elf`,同时留在 WSL `/home/song/ps5-web-file-manager/`。
|
||||
|
||||
### 4.2 验证清单
|
||||
|
||||
```bash
|
||||
ls -lh web-file-mgr-v1.9.2.elf
|
||||
sha256sum web-file-mgr-v1.9.2.elf
|
||||
od -An -tx2 -j18 -N2 web-file-mgr-v1.9.2.elf # 期望 3e00
|
||||
strings -a web-file-mgr-v1.9.2.elf | grep -m1 '^v1\.'
|
||||
```
|
||||
|
||||
⚠️ `od -An -tx2` 打印的是**小端 short 的值**(`003e`),不是字节序(`3e00`)。脚本里比对用 `$((16#$EM))` 转数值(**62 = x86-64 ✅ / 183 = aarch64 ❌**)。
|
||||
|
||||
✅ **构建已实测可复现**(2026-09-20):同一源码树两次构建 sha256 完全相同;把两处版本字面量回退成 `v1.9.1` 后重构,产物与已发布的 v1.9.1 ELF **逐字节一致**。所以 sha256 可以作为交付指纹用 —— 但它对任何源码改动都会全变(改一个字符串常量会令链接器重排 `.rodata` 字符串池,牵动 `.text` 里所有 RIP 相对位移,原始 diff 会放大到 5 万字节以上,属正常现象,别误判成"代码改了")。核对版本仍推荐 `strings ... | grep '^v1\.'`,最直观。
|
||||
|
||||
### 4.3 构建坑(已修,改 Makefile 前必读)
|
||||
|
||||
`third_party/7z/AesOpt.c` 用编译器版本宏判断是否启用 AES-NI / AVX / VAES,clang 18 直接进 VAES 分支,但 prospero-clang 默认 target 是 generic x86_64 → `_mm256_aesenc_epi128` 未声明,20 报错。
|
||||
|
||||
- ❌ **不能** `filter-out AesOpt.c` —— `Aes.c` 通过 `AesGenTables` 引用 `AesCbc_Encode_HW` 等符号,会链接失败
|
||||
- ✅ **正解**:路径过滤 `SEVENZ_C_FLAGS := -maes -mavx2 -mvaes`,仅 `third_party/7z/*.c` 用。PS5 是 Zen 2,硬件全支持,运行时无差异
|
||||
|
||||
---
|
||||
|
||||
## 五、7z 引擎设计要点(改代码前必读)
|
||||
|
||||
### 5.1 为什么不走 SDK 的解码器
|
||||
|
||||
LZMA SDK 26.03(public domain,已 vendor 到 `third_party/7z/`,解码子集 60 文件)有两个硬限制,**实测复现过**:
|
||||
|
||||
1. **`CSzFolder` 上限 4 coder / 3 bond** —— 7-Zip 默认 `-m0=bcj2` 链 = BCJ2 + 4×LZMA2 = 5 coder,`SzAr_DecodeFolder()` 返回 `SZ_ERROR_UNSUPPORTED`。注意 `SzArEx_Open()` 用的是另一套宽松扫描器(`k_Scan_NumCoders_MAX 64`),所以**文件列表和解压尺寸仍然全对**,失败只在解压时按条目暴露
|
||||
2. **C 解码器完全没有 7zAES coder** —— 所以 SDK 自己既解不了加密内容,也解不了加密头。内容侧由我们的 `sevenz_chain.c` 承担;**头部**侧由 `sevenz_header.c` 承担(见 §八)
|
||||
|
||||
→ 因此引擎**自解析 folder blob + 自己驱动 codec 链**(`src/sevenz_chain.c/.h`,pull pipeline:`node_pull(n, dst, want, &got)`,不够就 `node_refill()` 拉上游)。
|
||||
|
||||
### 5.2 最关键的坑
|
||||
|
||||
**【必记】每个 coder 节点的 `out_size` 必须取 `coder_unpack_sizes[index]`,绝不能用 folder 的 unpack size。** BCJ2 folder 里 MAIN 常大于 folder 最终尺寸(实测 300066 > 300000)。用错的症状:每层 LZMA2 静默短 21 字节,只在特定包上暴露。
|
||||
|
||||
其他:
|
||||
- 编译必须 `-DZ7_PPMD_SUPPORT`,否则 `7zDec.c` 直接丢掉 PPMd
|
||||
- `CoderUnpackSizes` 是**扁平数组**(每条 = 对应 coder 输出流大小),**非累计**;`FoToCoderUnpackSizes[f]..[f+1]` 是该 folder 的切片
|
||||
- main coder = 第一个未被 bond 消费的 coder
|
||||
- `SzArEx_Extract` 失败后会把半成品留在 block cache → 后续条目报**假 CRC**,要重置 `blockIndex`
|
||||
- 造夹具用 Extra 包的 `7za.exe`(有 PPMd);**`7zr.exe` 没有 PPMd 编码器**
|
||||
- 调试三件套在 `.build/`:`chainprobe.c`(摸内部图)、`chainprobe2.c -r <coder>`(强制 root 逐层二分)、`chaincheck.py`(Python liblzma 独立复现同一条链,秒判"图错"还是"循环错")
|
||||
|
||||
### 5.3 7zAES KDF
|
||||
|
||||
`numCyclesPower = b0 & 0x3F`;`saltSize = ((b0>>7)&1) + (b1>>4)`;`ivSize = ((b0>>6)&1) + (b1&0x0F)`,随后依次 salt → iv。
|
||||
`numCyclesPower == 0x3F` 时 key = `salt||password` 补齐/截断到 32 字节;否则 `key = SHA256(salt || password_utf16le || counter_le64)` 迭代 `1<<numCyclesPower` 次。之后 AES-256-CBC。
|
||||
限额 `max_aes_cycles = 24`(约 8s)。
|
||||
|
||||
**改引擎前先在 `.build/aesprobe.c` 独立验证 KDF**(用 vendor 的 `Sha256.c` + `Aes.c` 解 aes.7z coder0,与 `cus[0]=638314` 比对),确认后再集成。
|
||||
|
||||
### 5.4 分卷流抽象
|
||||
|
||||
- `src/zipx_volstream.c/.h`(ZIP,包成 `mz_stream`)· `src/sevenz_volstream.c/.h`(7z,包成 SDK `ISeekInStream`)
|
||||
- **结构体首成员必须是 `mz_stream stream;` / `ISeekInStream vt;`**(回调把 `void*` 强转)
|
||||
- `vol_is_open()` 必须返回 `MZ_OK`/`MZ_OPEN_ERROR`(**不是 1/0**)
|
||||
- vtbl **必须注册 `destroy`**,否则 `mz_stream_delete()` 不回调 → 泄漏
|
||||
- CONCAT 模式对 `DISK_NUMBER`/`DISK_SIZE` 返回 `MZ_PARAM_ERROR` → 让 minizip 不切盘
|
||||
- DISK 模式 `set_prop(DISK_NUMBER, -1)` 必须切到**最后一卷**(minizip 路径 `mz_zip.c:2252-2275`)
|
||||
- `remove_source_archives()` 要删**所有**卷,避免孤儿卷
|
||||
|
||||
### 5.5 提取门面
|
||||
|
||||
`src/sevenz_extract.c`(1753 行)完全仿 `zip_extract.c` / `rar_extract.c`:scan → extract(staging,**不 fsync**) → publish(整 rename) → cleanup。
|
||||
|
||||
- **OVERWRITE 与 MERGE 对目录-目录碰撞都递归下钻**(仅叶子文件不同)
|
||||
- 三方共用 `src/zipx_common.c`(限额 profile + `zipx_status_string()`)
|
||||
- scan 阶段**每 256 entries** 报一次进度(曾用 4096,小包扫描期 UI 静默),扫描末 force-report;`precheck_folders` 入口也强制报一次
|
||||
- 密码错时 detail **必须带 archive 名**(曾是 NULL → i18n `{arg}` 展开成空 → 用户看到「密码错误: 」后面光秃秃)
|
||||
|
||||
---
|
||||
|
||||
## 六、限额体系(ZIP 与 RAR 共用;7z 同源)
|
||||
|
||||
| 限额字段 | default | large | 160GB/9万文件场景 |
|
||||
|---|---|---|---|
|
||||
| `max_entries` | 200,000 | 500,000 | 9 万 ✅ |
|
||||
| `max_total_bytes` | 2 TiB | 4 TiB | 160 GiB ✅ |
|
||||
| `max_file_bytes` | **512 GiB** | **1 TiB** | 20 GiB ✅ |
|
||||
| `max_ratio` | 500 | 1000 | 仅 ≥1GiB 条目受检 |
|
||||
| `ratio_min_bytes` | 1 GiB | 1 GiB | 小文件豁免 |
|
||||
|
||||
- 切 large 档的触发条件:**压缩包文件本身** >480 GiB(`assets/main.js` `LARGE_FILE_THRESHOLD_BYTES`),160GB 包走 default
|
||||
- **ratio 有尺寸下限**(`ratio_min_bytes` = 1GiB):小文件高压缩率合法常见(零填充/稀疏),且写出字节受"声明上限 + `check_space()`"双重约束,无害
|
||||
- **唯一真实失败点是磁盘空间**:`check_space()` 按**解压后总量**查 `statvfs`,峰值 = `zip 体积 + 解出体积`。分卷场景"传一卷解一卷删一卷"可降峰值
|
||||
- **32 位安全**:引擎内部 size 全 `uint64_t`;minizip `mz_zip.h:34-35` 的 `compressed/uncompressed_size` 是 `int64_t` → >4GiB 不截断
|
||||
- **已知 UX 缺陷(未修)**:进度条 % 用字节(`main.js:1985`)、文字进度解压时用条目数(`main.js:2014`)、ETA 用字节速度(`task.c:129-177`)。混合大包上割裂,建议统一为字节
|
||||
|
||||
---
|
||||
|
||||
## 七、环境要点(新人必读)
|
||||
|
||||
- **PS5 是 x86-64 Zen 2**(不是 aarch64!),target triple `x86_64-sie-ps5`
|
||||
- **PS5 SDK C++ runtime = LLVM libc++**(FreeBSD 系 sysroot,无 libstdc++)→ C++ 必须 `-stdlib=libc++`,链接 `-lc++ -lc++abi`(Makefile 已处理:unrar 用 prospero-clang++ 编)
|
||||
- **PS5 SDK libc 的 `*at()` 族(mkdirat/openat/renameat/unlinkat)能链接但运行时损坏**:返回 -1 且 `errno=0`(2026-09-06 Frostpunk 2 真机确诊)。`zip_extract.c` 已有"*at() 失败回退全路径调用"兼容层;写新引擎代码时直接用全路径或沿用回退模式。报错要带 `(errno=%d)`,`errno=0` 时 `strerror` 会骗人
|
||||
- WSL:`export PS5_PAYLOAD_SDK=/opt/ps5-payload-sdk` 后才能 make
|
||||
- `target/user/homebrew/` 由 songl(197609) 拥有,WSL 身份 song(1000) 写不进 → staging 模式(make install 到 /tmp → `sudo cp -r`)
|
||||
- `//wsl$/Ubuntu-22.04/` 是 SMB 只读视图,改 WSL 文件必须走 Windows 路径
|
||||
- libmicrohttpd 必须 `--disable-https --disable-openssl`
|
||||
- **minizip-ng 4.2.2 补丁(升级会丢)**:`src/mz_strm_os_posix.c` L25 后插 `#ifndef O_BINARY / #define O_BINARY 0 / #endif`
|
||||
- **主机 POSIX shim(CP936 主机必需)**:`tests/posix_compat.h` 把 `lstat/stat → wfm_stat`、`opendir → _wopendir`(UTF-8 转换)、`fopen → _wfopen`。MinGW ANSI 入口看不见 UTF-8 文件名
|
||||
- 复杂 commit / tag message 用 `-F 文件`,不要 `-m` 长文本(bash quoting 会挂)
|
||||
|
||||
---
|
||||
|
||||
## 八、7z `-mhe=on` 加密头(2026-09-23 已闭合)
|
||||
|
||||
**它为什么难**:`-mhe=on` 时整个 header 也是一条独立的 7z 流——归档末尾的下一头部区域以 `k7zIdEncodedHeader`(0x17)开头,后接一段 StreamsInfo,描述「一个 folder,其输出就是真正的 header」。而 vendored SDK 的 **C 解码器没有 7zAES coder**,`SzArEx_Open2()` 走到 `SzAr_DecodeFolder()` 就返回 `SZ_ERROR_UNSUPPORTED`,于是**连文件列表都读不出来**(文件名、folder 表、每个条目的尺寸全在那份加密头里)。
|
||||
|
||||
**做法**(`src/sevenz_header.{c,h}`):
|
||||
1. 自己读 32 字节 start header,只探**一个字节**——不是 0x17 就立刻 `SZH_PLAIN` 收工(普通的 `-mhc=on` 压缩头、`-mhc=off` 明文头都走这条,SDK 行为一字不变)。
|
||||
2. 是 0x17 就整段读下来(CRC 校验),**最小解析** PackInfo + UnpackInfo:pack 位置/大小、folder 的 coder 描述字节范围、每个 coder 的 unpack size、folder CRC。解析刻意宽容——任何异常一律回落 `SZH_PLAIN`,把诊断权留给 SDK,保证非加密归档的报错一字不改。
|
||||
3. 把这份描述喂给 **`sz_chain_parse()` / `sz_chain_decode()`**,也就是内容走的同一条 7zAES 路径,密码规则完全一致:需要密码而没给 → `SZH_ERR_PASSWORD`;解出来 CRC 不对 / 不是 `k7zIdHeader` → 同样是密码错。
|
||||
4. 造一个**虚拟 `ISeekInStream`**:`[0,32)` 是改写过的 start header(指向明文头),`[hdr_off, hdr_off+L)` 是解出来的明文头,其余一律透传真实归档。`hdr_off` 就用加密头原本所在的偏移,所以**归档里存的任何一个偏移都不用搬**——SDK 在它预期的位置读到明文头,从明文头推出的 dataPos 依旧指向真实的内容 pack 流。
|
||||
5. 交给 `SzArEx_Open()`,之后一切照旧(内容仍由 `sevenz_chain.c` 直接读 `sevenz_volstream`)。
|
||||
|
||||
**要点/坑**:
|
||||
- 明文头比它替换掉的那条记录**长**(实测 aeshe.7z:记录 64 B、明文 462 B),所以虚拟流的 `total` 要取 `max(真实文件长度, hdr_off+L)`,否则 `SzArEx_Open2()` 的 seek-to-END 长度检查会报 `SZ_ERROR_INPUT_EOF`。
|
||||
- **`LookToRead2_INIT` 不 seek**,第一次 `Look` 从真实流当前位置读。`szh_prepare()` 会把流移来移去,所以交给 SDK 之前必须显式 seek 回 0(`sevenz_extract.c` 里那一段有注释)。这原本是个隐性依赖。
|
||||
- 明文头大小有上限(`SZH_MAX_HEADER` 64 MiB),sink 按需增长、不信头部里声明的 unpack size。
|
||||
- 只处理 `numFolders == 1`(SDK 自己对这条记录就传 `numFoldersMax = 1`)与 `external == 0`。
|
||||
|
||||
**覆盖**:`tests/fixtures-7z/aeshe.7z`(密码 `Secret123`),`tests/run-sevenz-tests.sh` 的 `KNOWN_GAPS` 已清空——chain 驱动与 façade 两条路径都跑通;`test_sevenz_extract.c --cases` 另验无密码 / 错密码 → `ZIPX_ERR_PASSWORD`、正确密码 → 成功,且失败后不留 staging。
|
||||
|
||||
### 真机端到端待验证
|
||||
|
||||
ELF 已构建,但需装 PS5 实测:
|
||||
1. ZIP / RAR / 7z 三类**分卷**真机解压
|
||||
2. **加密 7z(7zAES + `-mhe=on` 加密头)** 真机解压(`aeshe.7z` 那类归档在真机上连文件列表都要走新代码)
|
||||
3. 160GB / 9.5 万文件大 ZIP
|
||||
4. 分卷 RAR 进度条实时走动
|
||||
5. UI 右下角版本号显示(`/api/version` 与兜底字面量应一致)
|
||||
|
||||
> **📊 2026-09-23 首个真机性能数据**:解一个 **18 GB 的包**,**11 分钟**、**1252 个条目**
|
||||
> (平均 14.7 MB),UI 报 **10–40 MB/s**;**18 GB 是压缩包自身的大小**(✅ 已确认)。
|
||||
> **格式 = RAR**(✅ 已确认);包原本在 **PC 上**,**经插件上传**进 PS5,**上传速度 30–40 MB/s**。
|
||||
> 口径是**解压后的字节**(`zip_extract.c:651-678` 累加 `uncompressed_size`,`:900/910` 累加
|
||||
> `write()` 写出的解压字节),且是 **250 ms 采样的瞬时值**(`task.c:202-206`,进度只在 ≥1 MiB
|
||||
> 时上报)—— 摆动里含采样噪声,**只有「总字节 ÷ 总耗时」可信**。
|
||||
> 仍然成立的一条:**per-entry 开销不是主因**(平均 14.7 MB/条目,不是小文件场景)。
|
||||
>
|
||||
> **⚠️ 2026-09-23 晚 更正:本节原先写的「解码不是瓶颈」已撤回。** 两条理由:
|
||||
> ① 它拿「PS5 解 **RAR**」的 28 MiB/s 去比「PC 解 **7z**」的 427 MiB/s —— **不同格式、不同
|
||||
> 解码器、不同机器**,量级论证不成立;② 「4× 摆动 = 解码无罪」此前已降级为 250 ms 采样
|
||||
> 噪声的弱证据。**原先那句「源盘交付速度 ≈ 28 MiB/s 是硬上界」同样站不住** —— 它是从总耗时
|
||||
> 反推的*观测结果*,不是设备能力上限;若解码是瓶颈,源盘恰恰没跑满。
|
||||
>
|
||||
> **③ 新发现(有据可查、且直接针对真实负载):RAR 解码在 PS5 上是单线程的。**
|
||||
> `third_party/unrar7/os.hpp:43-45` 的 `#define RAR_SMP` 落在 `#ifdef _WIN_ALL` 分支**内**
|
||||
> ⇒ POSIX 构建不定义(我们 Makefile 里 0 次出现),而官方 POSIX makefile 第 11 行是
|
||||
> `DEFINES=… -DRAR_SMP` —— **我们漏了这个开关**。后果:`unpack50mt.cpp`
|
||||
> (`Unpack::Unpack5MT`,rarlab 专门调过的多线程 RAR5 解压器)**没编进来**,
|
||||
> `unpack.cpp:185-198` 的 MT 分支整段不参与编译,`SetThreads` / `ThreadPool` 一并消失。
|
||||
> ⇒ **首要假设:28 MiB/s ≈ 单线程 RAR5 解码的正常量级**(`18 GB ÷ 660 s` 是**解码输入**速率;
|
||||
> 而上传实测证明**写入端**至少能到 30–40 MB/s、**读取通常快于写入** ⇒ 纯存储上限解释不了它)。
|
||||
> ⚠️ 自查一条:**用 ELF 符号表查内部符号是无效手段** —— 该 ELF 只有 `.dynsym`(513 项)、
|
||||
> **无 `.symtab`**,「零命中」是假象(本次差点据此误判)。结论来自 Makefile 与 `os.hpp`。
|
||||
> **下一步**:先在 PC/WSL 上 A/B(`-DRAR_SMP` + `unpack50mt.cpp` + `-pthread`,跑同一批 RAR5
|
||||
> fixture 并保证 37 项 RAR 断言全绿),收益显著再上真机;同时真机补两个小事实
|
||||
> (**RAR4 还是 RAR5**、**解压后多大**)与 `T_copy`。详见 `docs/EXTRACTION-PERF.md` §六 文首
|
||||
> 更正块与 `docs/REAL-CONSOLE-PROFILE.md`。
|
||||
>
|
||||
> **⇒ 终局(2026-09-23 18:15,用户决定):这条线不做。** 不做的依据是**当前证据判不了收益**,
|
||||
> 而不是没收益:MT 只并行**解码**(worker 只跑 `unpack50mt.cpp:190` 的 `UnpackDecodeThread`),
|
||||
> 写盘恒为主线程串行(`UnpWriteBuf()` 只在 `unpack50mt.cpp:283/475/587` 被主线程调用)——
|
||||
> 若瓶颈在写路径(真机上传已证明写入端只有 30–40 MB/s),收益退化为 1.0×。要判定必须先做
|
||||
> `T_copy`(拿插件自己的 `TASK_COPY` 搬同一份包,`src/filemgr.c:836`),再决定是否值得改构建
|
||||
> 并刷机验证;用户选择停在第一步之前。**重开的第一个动作是 `T_copy`,不是改 `-DRAR_SMP`。**
|
||||
|
||||
### 可选项(非阻塞)
|
||||
|
||||
- **性能**:**优化前**实测上游(7-Zip 本体)在 7z 格式上快 1.9×(单线程)/ 3.4×(8 线程);ZIP 无显著差异。差距不在我们的架构(我们比 SDK 自己的 `SzArEx` 路径还快 1.02×)。**已全部落地(2026-09-16)**:①汇编解码器(`LzmaDecOpt.asm`+jwasm,1.26×,无 jwasm 自动退纯 C)②多线程 LZMA2(`Lzma2DecMt`,8 线程,1.37×,线程失败自动降级 chain;BCJ2/加密布局仍走 chain)③ZIP 逐条目 fsync 移除(8000 文件 ≥14×)。7z 现与 7-Zip 单线程打平、ZIP 已压过官方(本机受 Defender 拖累不可比,PS5 无该因素)。RAR 与官方 UnRAR 同速(unrar 自带 `target("aes")` SIMD 已启用,无逐条目 fsync)。完整数据见 `docs/EXTRACTION-PERF.md`,基准工具 `tests/bench_driver.py`。**⚠️ 但「单线程打平 / 8 线程 1.59×」是在最有利的输入形状上测的**:基准归档是 `-m0=lzma2 -ms=on` 的**单文件**(`tests/bench_driver.py:179`),恰好是唯一能让 `sz_chain_lzma2_root()`(`src/sevenz_chain.c:750`,要求 1 coder / 0 bond / 1 pack stream / 纯 LZMA2)生效的形状;真实的**多 folder / BCJ2 / 7zAES** 归档会让 MT 失效、退回单线程 chain —— 所以那个 1.59× **既不是上限也不是下限,方向未知**。剩余优化(CRC 硬件化、MT 扩到 BCJ2、条目级并行、ZIP inflate 换 libdeflate、AES-NI)**全部集中在 7z 多线程这一条线上**,建议**先拿真实归档在真机上 profile 再排序**,别按 PC 上这份数字动手
|
||||
- ~~fsync 批量化(每 64MB/N 条刷一次)~~ → **已作废,改为「彻底移除」**(2026-09-16):实际落地的不是批量刷,而是把 ZIP 引擎的逐条目 fsync 直接删掉(RAR/7z 本来就没有),三引擎统一为「**不 sync、只 rename**」——publish 是纯 rename、也没有续解功能,该 fsync 无收益。8000 文件 fixture:fsync 版 >200 s 未跑完 → 无 fsync **14.5 s(≥14×)**。已知取舍:publish 之后到落盘之间断电,可能出现「文件在但内容不完整」;要补只需在 extract 收尾做**一次**目录/整盘 flush(PS5 是 FreeBSD 系,`syncfs()` 不一定有,`sync()` 是全盘、偏重)。代码现状见 `src/zip_extract.c:937-945`,实测见 `docs/EXTRACTION-PERF.md:18-21`
|
||||
- 解压失败保留 staging 支持续解(中等改动)
|
||||
- ~~进度条 % / 文字进度 / ETA 三处口径统一为字节~~ → **已完成**(`assets/main.js:2071-2079`,条目计数已移除并注明原因)
|
||||
|
||||
---
|
||||
|
||||
## 九、仓库许可与代码归属(2026-09-15 核查)
|
||||
|
||||
用户曾担心「项目源自他人代码、没有许可」——**前提不成立**:
|
||||
|
||||
- 上游 `owendswang/ps5-web-file-manager` 经 GitHub API 确认 = **GPL-3.0**(78 stars,last push 2026-09-08)
|
||||
- 本项目 `LICENSE`(GPL-3.0 全文)在 root commit `5cb0b76` 即存在,与上游一致
|
||||
- 授权链完整:`ps5-payload-dev/websrv`(John Törnblom, GPLv3+,其 Copyright 头仍保留在 `asset.c` / `asset.h` / `mime.h` / `websrv.h`)→ `owendswang` → 本项目
|
||||
|
||||
代码量构成:
|
||||
- 第三方 vendored **71,528 行**(unrar7 27,710 / zlib 20,106 / LZMA SDK 17,248 / minizip-ng 6,464)——重写时原样复用,零成本
|
||||
- 第一方 20,844 行 = 上游 v1.7 遗产 13,860 + 自有 6,984
|
||||
- **自有代码中 3,661 行零耦合**(`sevenz_chain` 2094 + `zipx_volume` 658 + `zipx_volstream` 494 + `sevenz_volstream` 415,只依赖 public domain / zlib)→ 可单独抽成 MIT 库
|
||||
|
||||
完整评估见 `docs/REWRITE-FEASIBILITY.md`(三路径:补合规 0.5 天 / 架构重构 12–18 天 / clean-room 重写 35–50 天)。**结论:建议补合规而非重写** —— GPL-3.0 保护 7z 引擎成果不被闭源白嫖。
|
||||
|
||||
---
|
||||
|
||||
## 十、工作区状态
|
||||
|
||||
⚠️ **2026-09-23:工作树不再干净** —— 本轮加密改动(15 个已修改 + 8 个未跟踪文件)**尚未提交**,见文末「十一、本轮变更」。发版前需要先决定版本号并 commit。
|
||||
|
||||
以下为 2026-09-15 的历史清理记录(当时工作树干净,"仅剩有意保留的未跟踪文档")。
|
||||
|
||||
已清理(2026-09-15):
|
||||
|
||||
| 文件 | 说明 | 去向 |
|
||||
|---|---|---|
|
||||
| `erssonglDesktopWeb File Managerps5-web-file-manager¬`(2543 B) | 早期 shell 转义事故:一次 `git log --oneline --color` 的输出被重定向进了文件名。末尾是 U+F022(私用区码位,mojibake 残留),各工具渲染不一 —— git 显示成八进制转义、`ls -b` 印成 ASCII 引号 | **回收站**(`$R…`,2543 B,可还原) |
|
||||
| `web-file-mgr-unpack 1.9.1.elf`(898 KiB) | 陷阱:文件名写 1.9.1,内嵌却是 9-07 的 **v1.9**(无 7z 引擎) | 已不在仓库根 |
|
||||
|
||||
> ⚠️ **清理这类特殊文件名时**:`SHFileOperationW`(带 `FOF_ALLOWUNDO` 走回收站)对含私用区码位的路径会返回 `ERROR_FILE_NOT_FOUND (2)`,**但动作实际已生效**。删完务必查 `C:\$Recycle.Bin\<SID>\$I*` 记录确认落在回收站(`$I` 存原路径 UTF-16,`$R` 是内容)。本沙箱里 `Add-Type` 与 `rm` 都被拦(后者有 safe-delete 钩子),只能用 Python `ctypes` 调 shell32。
|
||||
|
||||
`.build/` 下的探针/调试产物已被 `.gitignore` 白名单覆盖,不再污染 `git status`。
|
||||
|
||||
---
|
||||
|
||||
## 十一、本轮(2026-09-23)变更:加密通道补齐(未提交)
|
||||
|
||||
**目标**:让 ZIP 与 RAR 的加密归档真正可解(7zAES 早已可用)。两者此前都报
|
||||
`extract_unsupported`,但**缺口在引擎侧,不在 UI** —— 密码框、`password=` 字段、
|
||||
`err_extract_password` 文案从 v1.9 起就已就位。
|
||||
|
||||
### 11.1 ZIP:给裁剪过的 minizip-ng 补一个 crypto 后端
|
||||
|
||||
`third_party/minizip-ng` 是裁到只读路径的精简副本,`mz_zip.c` 里
|
||||
`#ifdef HAVE_WZAES / HAVE_PKCRYPT` 的分支保留着,但**对应的流与 crypto 后端被裁掉了**。
|
||||
本轮补回:
|
||||
|
||||
| 文件 | 状态 | 说明 |
|
||||
|---|---|---|
|
||||
| `src/mz_strm_wzaes.{c,h}` | 上游 4.2.2 原样恢复 | WinZip AES 流(方法 99 + `0x9901` 扩展字段) |
|
||||
| `src/mz_strm_pkcrypt.{c,h}` | 上游 4.2.2 原样恢复 | 传统 PKWARE / ZipCrypto 流 |
|
||||
| `src/mz_crypt_wfm.c` | **新写**(~860 行) | 本地 crypto 后端:SHA-1、HMAC-SHA1、AES-128/192/256 |
|
||||
|
||||
后端要点:
|
||||
- S-box 与 GF(2^8) log/alog 表**首次使用时推导**,所以不新增 `.rodata` 查表(实测 `.rodata` 仅 +256 B,是字符串)。
|
||||
- 随机数直接 `open("/dev/urandom")`,**不要**走 `mz_os_rand()` —— 后者会退回 `rand()`/`srand()`,把两个新符号塞进导入表。最终产物**动态符号零新增**。
|
||||
- PBKDF2 复用 vendored 的 `mz_crypt.c`(与上游逐字节一致),没有重写。
|
||||
- 非 SHA-1 算法与 AEAD aad 一律返回 `MZ_SUPPORT_ERROR`(本项目只读,不需要)。
|
||||
- KAT 先行:写完后先用 FIPS 197 / RFC 3174 / RFC 2202 / RFC 6070 / SP 800-38A
|
||||
向量单独验算(`.build/kat_crypto.c`,24/24),再接线。踩到的三个坑:AES 仿射用 `rol32`
|
||||
应为 `rol8`;GF 乘法 `(uint8_t)(a+b) % 255` 截断,应全程 int;HMAC 的 ipad 必须由
|
||||
**已 XOR 过 0x5c 的 opad** 再推。
|
||||
|
||||
### 11.2 RAR:把 `RARSetPassword` 接上
|
||||
|
||||
`src/rar_extract.{c,h}`:新增 `password` 形参(`rar_extract()` 为第 8 个参数)。
|
||||
调用点是 `RAROpenArchiveEx` 之后、**首次 `RARReadHeaderEx` 之前**——这是解密 `-hp`
|
||||
头加密归档的硬性顺序要求(scan 与 extract 两个阶段各自开档,两处都要设)。
|
||||
`ERAR_MISSING_PASSWORD` / `ERAR_BAD_PASSWORD` 由 `ZIPX_ERR_UNSUPPORTED` 改映射为
|
||||
`ZIPX_ERR_PASSWORD`;`RHDF_ENCRYPTED` 只在**没给密码**时提前拒绝。
|
||||
|
||||
### 11.3 前端:密码失败后自动重试(这步不做,功能等于不可达)
|
||||
|
||||
原先密码框**只对 7z 弹**(`actionExtract()` 里的 `isSevenZipArchive()` 判断),
|
||||
ZIP/RAR 加密归档失败后用户根本没机会输密码。现在 `handleTerminalTask()` 在
|
||||
`op === "extract" && error_code === "extract_password"` 时走
|
||||
`retryExtractWithPassword()`:
|
||||
|
||||
- 记忆原始请求(`extractRetryKey(task.id)` → `{conflict, removeSource, name, large, attempts}`),
|
||||
重试时**保持冲突策略与大文件选配**;
|
||||
- **必须按任务 id 记,不能按路径记**(v1.9.3M 后期修正)。路径在传输中被
|
||||
「服务端 JSON 逐字节转义为 `\u00XX`」+「`fs_path_value()` 反向还原成原始字节」这一对
|
||||
转换改了表示 ⇒ **非 ASCII 目录下**「页面手里的路径」≠「任务回报的路径」,按路径查必然
|
||||
落空 ⇒ 口令框永远不弹,用户只看到一个失败框,必须先手动再解压一次。任务 id 由服务端
|
||||
分配、原样回传,不受编码影响;重试时也改用**服务端回报的** `task.src` / `task.dst` 重发。
|
||||
消费即删(重试注册到新 id 下),Map 最多留 8 条(同一时刻只可能有一个活动任务)。
|
||||
- 最多 3 次;取消或空输入即放弃,回落到原有失败提示;
|
||||
- 7z 保留提前询问(免得白跑一次 scan + folder 解析)。
|
||||
|
||||
新增文案 `extractPasswordRetryAsk`(重试:密码不正确)+ `extractPasswordFirstAsk`(首次:
|
||||
此压缩包已加密),与提前询问用的 `extractPasswordAsk` 区分 —— 第一次失败时用户还没输过密码,
|
||||
再说「密码不正确」就是误导。
|
||||
|
||||
错误文案里的条目名必须过 `decodeFsText()`:`backendErrorText()` 原样用了 `error_arg`,
|
||||
而服务端把它逐字节转义过 ⇒ 中文/日文条目名在错误框里显示成 `â®…ç§.psd`。列表侧一直有这层
|
||||
翻译(`displayName()`),只有错误文案漏了。
|
||||
|
||||
同源的编码坑:`pathJoin(服务端回报的目录, 本地文件名)` 把两种表示混进同一个字符串,而
|
||||
`fs_path_value()` **只要发现任一个码点 > 0xFF 就整体不修** ⇒ 中文名文件放进中文名目录时
|
||||
路径失效。新增 `encodeFsText()`(`decodeFsText()` 的逆)在拼接前把本地名转成同一表示,
|
||||
`uploadAndExtractFile()` 与 `actionNewText()` 两处都用它。
|
||||
|
||||
无头回归:`.build/ui_retry_test.mjs`(真 `main.js` 载入桩 DOM,40 checks,含「非 ASCII 目录
|
||||
必须仍弹口令框」的回归用例)、`.build/ui_upload_menu_test.mjs`(40 checks:i18n 键覆盖、
|
||||
菜单接线、样式、高亮规则的层叠作用域,以及**解压按钮不许被隐藏、只许被置灰**)、
|
||||
`.build/preview_check.mjs`(无头 Chromium 跑真页面,验菜单开关、页脚布局、**解压按钮的
|
||||
显隐/置灰/提示随选区变化**,并**读回三种交互状态下高亮的计算值**;该脚本已改为失败即
|
||||
非零退出)。
|
||||
|
||||
**菜单行的「选中高亮」曾被两条规则同时破坏**(用户报「选中下面那个高亮效果不对」):
|
||||
① 全局 `button:focus` 的 `outline: 3px + offset 2px` 是按 54px 工具栏按钮设计的,套在 46px
|
||||
菜单行上会越过面板 6px 内边距、压住相邻行,且 outline 的圆角半径不随 offset 自适应 ⇒ 视觉上
|
||||
成了一个「脱离的框 + 两侧挂着的弧线」;② 面板自己的
|
||||
`.upload-menu-list button:hover:not(:disabled)` **从未生效过** —— 它与
|
||||
`button:not(.row-action):hover:not(:disabled)` 特异性同为 `(0,3,1)`,而后者在文件里更靠后 ⇒
|
||||
后者胜出,于是 hover 是 `#303945`、focus 是 `#2b343e`,**两个高亮两个颜色**,且一行 hover 时
|
||||
另一行仍因 focus 亮着 ⇒ 看起来「两行同时被选中」。修法:两条规则都收敛到面板 id
|
||||
(`#uploadMenu button:…`,`(1,1,1)` / `(1,2,1)` 稳赢通用规则),行只用填充表示选中,键盘焦点
|
||||
提示改为**行内 `inset` 环**(`box-shadow: inset 0 0 0 2px`)—— 画在行内,任何行高都不可能
|
||||
越界。**这类坑只有真引擎读计算值才抓得住**,光看源码两条规则都「像是对的」。
|
||||
|
||||
**解压按钮改为「常显 + 置灰」**(用户要求「直接显示出来 只不过是灰色的 只有能解压的文件才可以
|
||||
点击」):`index.html` 去掉 `hidden`,`renderExtractButton()` 不再碰 `.hidden`,改成按选区设
|
||||
disabled 并给一条说明原因的工具提示(什么都没选 ⇒ 新增 `extractSelectArchive`;只选中子卷 ⇒
|
||||
沿用 `extractSelectMainVolume`;选中**多个** ⇒ 新增 `extractOneAtATime`,旧代码这种情况错用了
|
||||
「请改选主卷」,文案本身是错的)。🪤 **`button:disabled` 带 `pointer-events: none` ⇒ 禁用按钮
|
||||
无法 hover,`title` 永远不弹** —— 必须像既有的 `.parent-nav-button:disabled` 那样把
|
||||
`pointer-events` 还回来(点击仍无效,`disabled` 属性本身挡激活)。标签同时从 `extractToCurrent`
|
||||
(「解压到当前目录」)换成短词 `extract`(「解压」),与工具栏其他动词一致:**常显按钮不该
|
||||
同时又是最宽的那个**(英文下 `Extract to current folder` 会到 107 px)。
|
||||
**代价必须实测而不是估**:按钮宽 96 px ⇒ 工具栏换行阈值(zh)1080 → 1190 px、(en)1230 →
|
||||
1350 px。`.build/preview_check.mjs` 已把阈值**钉成断言**(1920/1600/1280 必须都是一行),
|
||||
并按四种选区验 disabled / opacity / title;该脚本同时从「只打印」改成**失败即非零退出**。
|
||||
|
||||
### 11.4 构建坑:编译选项变化必须让目标文件失效(**改 Makefile 前必读**)
|
||||
|
||||
`make` **看不见**编译选项变化。加 `-DHAVE_WZAES -DHAVE_PKCRYPT` 后,
|
||||
`mz_zip.o` / `mz_crypt.o` 被判定为最新而复用 → 此时已无线程引用新流 →
|
||||
`--gc-sections` 把加密代码再丢一次,**链接却报成功**(本次第一次构建的产物与
|
||||
已发布 v1.9.2 **逐字节相同**,`readelf` 才发现 `.text` 只长了 336 B)。
|
||||
|
||||
修法(取代原先的 `LzmaDec.o` 特例):把第三方编译选项写进标记文件,
|
||||
内容变了才重编。
|
||||
|
||||
```make
|
||||
PS5_FLAGS_STAMP := ps5-obj/.third_party_cflags
|
||||
LINUX_FLAGS_STAMP := linux-obj/.third_party_cflags
|
||||
$(PS5_FLAGS_STAMP): FORCE
|
||||
@printf '%s\n' '$(THIRD_PARTY_C_FLAGS_7Z) $(LZMA_DEC_OPT_FLAG)' > $@.tmp
|
||||
@cmp -s $@.tmp $@ || { mv -f $@.tmp $@; echo ' [cflags] ...'; }
|
||||
```
|
||||
|
||||
> 诊断手法:拿未 strip 的产物比 `readelf -S` 各段尺寸,而不是看总体积。
|
||||
> 改一个字符串常量会重排 `.rodata` 字符串池,字节 diff 会被放大到几万字节,
|
||||
> 但段尺寸是守恒的——判断"代码到底有没有变"要看段尺寸 + 助记符序列。
|
||||
> 现成脚本:`.build/_seccmp.py`、`.build/_operandcheck.sh`。
|
||||
|
||||
### 11.5 产物与验证
|
||||
|
||||
| 项 | 值 |
|
||||
|---|---|
|
||||
| 主机测试 | ZIP 140 + RAR 37 = **177 checks / 0 失败**(`tests/run-tests.sh`) |
|
||||
| 前端测试 | **27 checks / 0 失败**(`.build/ui_retry_test.mjs`) |
|
||||
| ELF | 870,680 B · sha256 `b1409f5c1bc4b1a39ab337853b956f4807f95c5770dee6eca7a18a62cc08f80e` · e_machine 0x003e(加密轮结束时;`-mhe=on` 之后的产物见 §12.3) |
|
||||
| 确定性 | 同一源码树构建两次逐字节一致 |
|
||||
| 段变化(vs 已发布 v1.9.2) | `.text` +11,296 · `.bss` +5,120(AES 表) · `.rodata` +256 · 动态符号零新增 |
|
||||
| 内嵌资产核验 | ELF 内 gzip 资源中可检出 `retryExtractWithPassword` / `extractPasswordRetryAsk`(普通 `strings` 找不到,要先解 gzip;脚本 `.build/check-elf-gzip.py`) |
|
||||
|
||||
**未做(发版前必做)**:未 commit / tag / 发 Release;真机端到端未验。
|
||||
版本号**已升**为 `v1.9.3M`(2026-09-24 加改版标记 `M`,见 §2.2)。
|
||||
README(中英)、CHANGELOG、本文档已同步为「未发布」状态。
|
||||
|
||||
---
|
||||
|
||||
## 十二、本轮(2026-09-23)变更:7z `-mhe=on` 加密头(未提交)
|
||||
|
||||
**目标**:补上最后一个 7z 格式缺口(设计与坑见 §八)。
|
||||
|
||||
### 12.1 新增
|
||||
|
||||
| 文件 | 说明 |
|
||||
|---|---|
|
||||
| `src/sevenz_header.{c,h}` | **新写**(~900 行)。头部读取 + `k7zIdEncodedHeader` 最小解析 + 虚拟 `ISeekInStream` |
|
||||
| `Makefile` | `src/sevenz_header.c` 进 `COMMON_SRCS`(PS5 与 linux 共用) |
|
||||
| `tests/run-sevenz-tests.sh` | 编 `sevenz_header.o` 进 `ENGINE_OBJS`;`KNOWN_GAPS` 清空;façade 矩阵加入 `aeshe` |
|
||||
| `tests/sevenz_chain_e2e.c` | 按与产品相同的顺序接线 `szh_prepare()`(否则 chain 矩阵读不了 `aeshe`) |
|
||||
| `tests/test_sevenz_extract.c` | `aeshe` 三例(无密码 / 错密码 → `ZIPX_ERR_PASSWORD`;正确密码 → 成功),另修一处 `snprintf` 截断告警 |
|
||||
|
||||
### 12.2 关键设计(细节见 §八)
|
||||
|
||||
- 只探**一个字节**:不是 `0x17` 立刻返回 `SZH_PLAIN`,SDK 行为与改动前完全一致(12 个既有 fixture 全部复跑通过)。
|
||||
- 解析刻意宽容:PackInfo/UnpackInfo 之外的任何异常都回落 `SZH_PLAIN`,把诊断权留给 SDK。
|
||||
- 复用 `sz_chain_parse()` / `sz_chain_decode()`,所以 7zAES 的密码/错误语义与内容侧**完全同源**,不新增第二个 crypto 实现。
|
||||
- 虚拟流的 `total` 必须 `max(真实长度, hdr_off + L)`;`LookToRead2_INIT` 不 seek,交回 SDK 前必须显式 seek 到 0。
|
||||
|
||||
### 12.3 产物与验证
|
||||
|
||||
| 项 | 值 |
|
||||
|---|---|
|
||||
| 7z 套件 | **27 用例 / 0 失败**,`aeshe` 在 chain 与 façade 两条路径都 `ok`,`KNOWN_GAPS` 为空 |
|
||||
| 主机测试(ZIP/RAR 回归) | ZIP 140 + RAR 37 = **177 checks / 0 失败**(无回归) |
|
||||
| ELF | 903,448 B · sha256 `8ca47d5aaca75085b32641300cce30fadb7df7749cb6b53d04f129bcecc286b7` · e_machine 0x003e(== 本轮最终产物,见 §12.4) |
|
||||
| 确定性 | 同一源码树构建两次 sha256 相同 |
|
||||
| 段变化(解压按钮常显 vs 上一轮) | **只有 `.rodata` 变化**:`0x026CC0` → `0x026F00`(+0x240 = 576 B:index.html 去掉 `hidden` 并换短标签、`main.js` 的三条禁用理由、两份语言文件各两条新文案、`.extract-action:disabled` 及其注释)。`.text` 两次均为 `0x087780`、`.data` 均为 `0x00034C` —— 第六次「只改内嵌前端资源」 |
|
||||
| 段变化(菜单行高亮修复 vs 上一轮) | **只有 `.rodata` 变化**:`0x026B40` → `0x026CC0`(+0x180 = 384 B,三条收敛后的高亮规则加其注释)。`.text` 两次 readelf 均为 `0x087780` —— 又一次「只改内嵌前端资源、不碰 C 逻辑」的标准形状 |
|
||||
| 段变化(本轮前端三项 vs 上一轮) | **只有 `.rodata` 变化**:`0x026A40` → `0x026B40`(+0x100 = 256 B)。`.text` / `.data` / `.eh_frame*` 一字节未变 —— 「只改内嵌前端资源 + 加两条文案」的标准形状 |
|
||||
| 段变化(加 `M` 标记 + 修正 `err_extract_unsupported` 文案 vs 加密轮产物) | **只有 `.rodata` 变化**:加 `M` 标记 +0x100(256 B),修正文案再 +0x40(64 B);`.text` / `.data` / `.bss` / `.eh_frame*` / `.gcc_except_table` **一个字节都没变**。又因 16 KiB 段对齐留有余量,**六次构建的文件总尺寸都是 903,448 B**:尺寸相同**不代表**二进制相同(sha256 逐个不同:`f3164efa…` → `53296d29…` → `7b5ab00c…` → `212107a6…` → `da36834d…` → `cf2c0fcf…` → `8ca47d5a…`) |
|
||||
| 段变化(加密轮 vs 其前一轮) | `.text` +4,880 · `.rodata` +640 · `.eh_frame_hdr` +32 · `.eh_frame` +160 —— 正文合计 **+5,712**;其余 **+27,056** 是 `p_align=0x4000` 的两处段对齐填充(LOAD#1 越过 0x8C000 边界)。**段数仍为 20,动态符号零新增(513 → 513)** |
|
||||
|
||||
> 判读提示:这次文件涨了 32,768 B,但正文只涨 5,712 B —— 不要按体积下结论。
|
||||
> 权威做法是比较**段尺寸**与**动态符号集合**(见 §11.4 的诊断手法)。
|
||||
|
||||
**未做(发版前必做)**:未 commit / tag / 发 Release;真机端到端未验。
|
||||
版本号已升为 `v1.9.3M`(改版标记 `M` 于 2026-09-24 加入)。
|
||||
@@ -13,7 +13,26 @@ ifeq ($(MAKECMDGOALS),)
|
||||
endif
|
||||
endif
|
||||
|
||||
VERSION_TAG := v0.2
|
||||
# Bump this together with the git tag -- it is baked into the binary (the PS5
|
||||
# notification, the stdout banner and /api/version all print it) AND into the
|
||||
# output filename, so a stale value silently mislabels everything. Override
|
||||
# per-build with:
|
||||
# make VERSION_TAG=v1.9.3M
|
||||
#
|
||||
# **The trailing M is the fork marker** (Modified build, maintained by
|
||||
# LisherSong). Upstream owendswang releases are plain `vX.Y.Z`, so any string
|
||||
# carrying the M is ours and anything without it is not. The marker rides on
|
||||
# VERSION_TAG rather than on a separate display-only constant on purpose: it
|
||||
# therefore reaches every surface at once -- /api/version, the PS5 start-up
|
||||
# notification, the stdout banner, the UI footer and the ELF file name -- and
|
||||
# is impossible to forget in one of them. The file name gains a second benefit:
|
||||
# a fork build can no longer collide with an upstream artifact of the same
|
||||
# upstream version, which has already caused two mix-ups (a local
|
||||
# `web-file-mgr-v1.9.2.elf` sitting next to the released one under the same
|
||||
# name, and a `-DVERSION_TAG=v1.9.1` leftover wearing the released 870 488 B
|
||||
# file's size). To cut a build that is byte-for-byte upstream-shaped, pass
|
||||
# `make VERSION_TAG=v1.9.3`.
|
||||
VERSION_TAG ?= v1.9.3M
|
||||
TITLE_ID := FMGR88888
|
||||
PYTHON ?= python3
|
||||
STRIP ?= $(PS5_PAYLOAD_SDK)/bin/prospero-strip
|
||||
@@ -22,32 +41,125 @@ HOST_CC ?= cc
|
||||
HOST_STRIP ?= strip
|
||||
HOST_PKG_CONFIG ?= pkg-config
|
||||
|
||||
BIN := web-file-mgr.elf
|
||||
LEGACY_BIN := web-file-mgr-legacy.elf
|
||||
LINUX_BIN := web-file-mgr-linux
|
||||
COMMON_SRCS := src/main.c src/websrv.c src/filemgr.c src/asset.c src/mime.c src/notify.c
|
||||
PS5_SRCS := $(COMMON_SRCS) src/app_installer.c
|
||||
# Output filename carries the version so two builds never overwrite each other
|
||||
# and you can tell at a glance which ELF is on the USB stick.
|
||||
BIN := web-file-mgr-$(VERSION_TAG).elf
|
||||
LINUX_BIN := web-file-mgr-linux-$(VERSION_TAG)
|
||||
COMMON_SRCS := src/main.c src/websrv.c src/filemgr.c src/file_response.c src/task.c src/upload.c src/download.c src/text.c src/list.c src/space.c src/version.c src/fs_util.c src/json_util.c src/path_util.c src/asset.c src/mime.c src/notify.c src/pkg_installer.c src/pkg_info.c src/extract.c src/zip_extract.c src/rar_extract.c src/zipx_volume.c src/zipx_volstream.c src/zipx_common.c src/sevenz_extract.c src/sevenz_chain.c src/sevenz_header.c src/sevenz_volstream.c src/sevenz_mt.c src/demangle_stub.c
|
||||
PS5_SRCS := $(COMMON_SRCS) src/app_installer.c src/cpu_support_stub.c
|
||||
LINUX_SRCS := $(COMMON_SRCS)
|
||||
ASSETS := $(wildcard assets/*)
|
||||
BASE_ASSETS := $(filter-out %.dds,$(wildcard assets/*))
|
||||
ifneq ($(filter linux,$(MAKECMDGOALS)),)
|
||||
ASSETS := $(BASE_ASSETS)
|
||||
else
|
||||
ASSETS := $(filter-out assets/icon0.png,$(BASE_ASSETS))
|
||||
endif
|
||||
GEN_SRCS := $(patsubst assets/%,gen/%, $(ASSETS:=.c))
|
||||
|
||||
CFLAGS := -Oz -fno-asynchronous-unwind-tables -fno-unwind-tables -Wall -Werror -ffunction-sections -fdata-sections -Isrc -DVERSION_TAG=\"$(VERSION_TAG)\" -DTITLE_ID=\"$(TITLE_ID)\"
|
||||
# Vendored third-party: zlib + minizip-ng (ZIP, C) and unrar 7.20.1 (RAR,
|
||||
# C++). unrar sources are compiled as a static library in RARDLL mode (no
|
||||
# main()); the project talks to it through the extern "C" DLL API in
|
||||
# third_party/unrar7/unrar_c_api.h. Compiled with relaxed warnings (-w) —
|
||||
# these are not our code and we do not want to chase upstream style updates.
|
||||
#
|
||||
# C++ compilers: PS5 uses prospero-clang++ (FreeBSD-style sysroot; the
|
||||
# toolchain defaults to -stdlib=libc++, driver links libc++ automatically);
|
||||
# host builds use the plain host C++ compiler (libstdc++).
|
||||
CXX ?= $(dir $(CC))prospero-clang++
|
||||
HOST_CXX ?= c++
|
||||
|
||||
# Source set mirrors UnRARDll.vcxproj's ClCompile list (49 files) MINUS the
|
||||
# Windows-only isnt.cpp / motw.cpp (they need windows.h; the official unrar
|
||||
# UNIX makefile omits them, and PS5/linux both use the _UNIX branch where
|
||||
# their symbols are #ifdef'd out).
|
||||
UNRAR7_SRCS := \
|
||||
third_party/unrar7/archive.cpp third_party/unrar7/arcread.cpp third_party/unrar7/blake2s.cpp \
|
||||
third_party/unrar7/cmddata.cpp third_party/unrar7/consio.cpp third_party/unrar7/crc.cpp \
|
||||
third_party/unrar7/crypt.cpp third_party/unrar7/dll.cpp third_party/unrar7/encname.cpp \
|
||||
third_party/unrar7/errhnd.cpp third_party/unrar7/extinfo.cpp third_party/unrar7/extract.cpp \
|
||||
third_party/unrar7/filcreat.cpp third_party/unrar7/file.cpp third_party/unrar7/filefn.cpp \
|
||||
third_party/unrar7/filestr.cpp third_party/unrar7/find.cpp third_party/unrar7/getbits.cpp \
|
||||
third_party/unrar7/global.cpp third_party/unrar7/hash.cpp third_party/unrar7/headers.cpp \
|
||||
third_party/unrar7/largepage.cpp third_party/unrar7/match.cpp \
|
||||
third_party/unrar7/options.cpp third_party/unrar7/pathfn.cpp \
|
||||
third_party/unrar7/qopen.cpp third_party/unrar7/rar.cpp third_party/unrar7/rarpch.cpp \
|
||||
third_party/unrar7/rarvm.cpp third_party/unrar7/rawread.cpp third_party/unrar7/rdwrfn.cpp \
|
||||
third_party/unrar7/rijndael.cpp third_party/unrar7/rs.cpp third_party/unrar7/rs16.cpp \
|
||||
third_party/unrar7/scantree.cpp third_party/unrar7/secpassword.cpp third_party/unrar7/sha1.cpp \
|
||||
third_party/unrar7/sha256.cpp third_party/unrar7/smallfn.cpp third_party/unrar7/strfn.cpp \
|
||||
third_party/unrar7/strlist.cpp third_party/unrar7/system.cpp third_party/unrar7/threadpool.cpp \
|
||||
third_party/unrar7/timefn.cpp third_party/unrar7/ui.cpp third_party/unrar7/unicode.cpp \
|
||||
third_party/unrar7/unpack.cpp third_party/unrar7/volume.cpp
|
||||
|
||||
THIRD_PARTY_C_SRCS := $(wildcard third_party/zlib/src/*.c) $(wildcard third_party/minizip-ng/src/*.c) $(wildcard third_party/7z/*.c)
|
||||
# AesOpt.c hard-codes x86 AES-NI / AVX / VAES intrinsics and guards them with
|
||||
# a compiler-version check that lets clang 18 in unconditionally. The plain
|
||||
# intrinsics (`_mm256_aesenc_epi128`) live behind <wmmintrin_aes.h>, which
|
||||
# clang only declares after `+mvaes +mavx2` (or higher). PS5 is Zen 2 and has
|
||||
# every one of these, so we just enable them for the 7z TU family instead of
|
||||
# dropping AesOpt.c (Aes.c references those HW symbol names via AesGenTables).
|
||||
SEVENZ_C_FLAGS := -maes -mavx2 -mvaes
|
||||
# HAVE_WZAES / HAVE_PKCRYPT switch on minizip-ng's two ZIP encryption paths:
|
||||
# mz_strm_wzaes.c (WinZip AES, method 99 / extra field 0x9901) and
|
||||
# mz_strm_pkcrypt.c (traditional PKWARE "ZipCrypto"). The mz_zip.c branches
|
||||
# behind these macros are already present, so defining them only pulls in the
|
||||
# two streams plus the local crypto backend in mz_crypt_wfm.c. Everything they
|
||||
# need (crc32, PBKDF2-HMAC-SHA1, AES-ECB) is implemented in-tree; see the
|
||||
# header of third_party/minizip-ng/src/mz_crypt_wfm.c.
|
||||
THIRD_PARTY_C_FLAGS := -O2 -w -Ithird_party/zlib/include -Ithird_party/minizip-ng/include -Ithird_party/7z \
|
||||
-DHAVE_ZLIB -DZLIB_COMPAT -DHAVE_UNISTD_H=1 -D_FILE_OFFSET_BITS=64 -D_LARGEFILE64_SOURCE \
|
||||
-DHAVE_FSEEKO -DZ7_PPMD_SUPPORT -DHAVE_WZAES -DHAVE_PKCRYPT
|
||||
THIRD_PARTY_C_FLAGS_7Z := $(THIRD_PARTY_C_FLAGS) $(SEVENZ_C_FLAGS)
|
||||
|
||||
# Assembly-optimised LZMA decoder (optional, on when jwasm is present).
|
||||
#
|
||||
# LzmaDec.c carries a compile-time switch: with Z7_LZMA_DEC_OPT it calls an
|
||||
# external LzmaDec_DecodeReal_3() and drops its own C implementation; without
|
||||
# it, the C version is used. The asm version is measurably faster -- on a
|
||||
# 330 MiB LZMA2 archive, 1.10 s vs 1.39 s, i.e. most of the gap to the
|
||||
# official 7-Zip binary, which builds with this switch on.
|
||||
#
|
||||
# LzmaDecOpt.asm is MASM syntax, so it needs a MASM-compatible assembler
|
||||
# (jwasm). That is not something we can assume the host has, so the whole
|
||||
# optimisation is conditional: no jwasm, no asm, and the build still works.
|
||||
# ABI_LINUX is load-bearing -- 7zAsm.asm keys its calling convention off it
|
||||
# (SysV rdi/rsi/rdx vs Win64 rcx/rdx/r8); assembling without it links cleanly
|
||||
# and then segfaults on the first call.
|
||||
JWASM ?= jwasm
|
||||
LZMA_DEC_ASM_DIR := third_party/7z/Asm/x86
|
||||
LZMA_DEC_ASM_SRC := $(LZMA_DEC_ASM_DIR)/LzmaDecOpt.asm
|
||||
ifneq ($(shell command -v $(JWASM) 2>/dev/null),)
|
||||
LZMA_DEC_OPT_FLAG := -DZ7_LZMA_DEC_OPT
|
||||
PS5_ASM_OBJS := ps5-obj/$(LZMA_DEC_ASM_DIR)/LzmaDecOpt.o
|
||||
LINUX_ASM_OBJS := linux-obj/$(LZMA_DEC_ASM_DIR)/LzmaDecOpt.o
|
||||
endif
|
||||
UNRAR7_CXX_FLAGS := -O2 -w -std=c++17 -DRARDLL -D_FILE_OFFSET_BITS=64 -D_LARGEFILE_SOURCE
|
||||
# prospero-clang++ defaults to -stdlib=libc++; state it explicitly for clarity.
|
||||
UNRAR7_CXX_FLAGS_PS5 := $(UNRAR7_CXX_FLAGS) -stdlib=libc++
|
||||
UNRAR7_CXX_FLAGS_HOST:= $(UNRAR7_CXX_FLAGS)
|
||||
|
||||
PS5_TP_OBJS := $(patsubst %.c,ps5-obj/%.o,$(THIRD_PARTY_C_SRCS)) \
|
||||
$(patsubst %.cpp,ps5-obj/%.o,$(UNRAR7_SRCS))
|
||||
LINUX_TP_OBJS := $(patsubst %.c,linux-obj/%.o,$(THIRD_PARTY_C_SRCS)) \
|
||||
$(patsubst %.cpp,linux-obj/%.o,$(UNRAR7_SRCS))
|
||||
|
||||
CFLAGS := -Oz -fno-asynchronous-unwind-tables -fno-unwind-tables -Wall -Werror -ffunction-sections -fdata-sections -Isrc -Ithird_party/minizip-ng/include -Ithird_party/unrar7 -Ithird_party/7z -DVERSION_TAG=\"$(VERSION_TAG)\" -DTITLE_ID=\"$(TITLE_ID)\"
|
||||
CFLAGS += `$(PKG_CONFIG) libmicrohttpd --cflags`
|
||||
LDFLAGS := -Wl,--gc-sections
|
||||
LEGACY_CFLAGS := $(filter-out -ffunction-sections -fdata-sections,$(CFLAGS))
|
||||
LEGACY_LDFLAGS :=
|
||||
# --icf=all: fold byte-identical functions. LLD-only (GNU ld's --icf is
|
||||
# incomplete), so it stays on the PS5 line -- the linux target never uses
|
||||
# LDFLAGS. Paired with src/demangle_stub.c this takes the ELF from ~1010 to
|
||||
# ~850 KiB; see docs/SIZE-OPTIMIZATION.md.
|
||||
LDFLAGS := -Wl,--gc-sections -Wl,--icf=all
|
||||
LDADD := `$(PKG_CONFIG) libmicrohttpd --libs`
|
||||
LDADD += -lSceIpmi -lSceAppInstUtil
|
||||
LINUX_CFLAGS := -O2 -Wall -Werror -Isrc -DVERSION_TAG=\"$(VERSION_TAG)\" -DTITLE_ID=\"$(TITLE_ID)\"
|
||||
LDADD += -lSceIpmi -lSceAppInstUtil -lSceUserService
|
||||
LINUX_CFLAGS := -O2 -flto -Wall -Werror -Isrc -Ithird_party/minizip-ng/include -Ithird_party/unrar7 -Ithird_party/7z -DVERSION_TAG=\"$(VERSION_TAG)\" -DTITLE_ID=\"$(TITLE_ID)\"
|
||||
LINUX_CFLAGS += `$(HOST_PKG_CONFIG) libmicrohttpd --cflags`
|
||||
LINUX_LDADD := `$(HOST_PKG_CONFIG) libmicrohttpd --libs` -pthread
|
||||
|
||||
.PHONY: all legacy linux deps linux-deps clean
|
||||
.PHONY: all linux deps linux-deps clean
|
||||
|
||||
all: deps $(BIN)
|
||||
|
||||
legacy: deps $(LEGACY_BIN)
|
||||
|
||||
linux: linux-deps $(LINUX_BIN)
|
||||
|
||||
deps:
|
||||
@@ -61,19 +173,96 @@ gen:
|
||||
mkdir gen
|
||||
|
||||
clean:
|
||||
rm -rf $(BIN) $(LEGACY_BIN) $(LINUX_BIN) gen
|
||||
rm -rf $(BIN) $(LINUX_BIN) gen ps5-obj linux-obj
|
||||
|
||||
gen/%.c: assets/% gen-asset-module.py gen
|
||||
gen/%.c: assets/% gen-asset-module.py | gen
|
||||
$(PYTHON) gen-asset-module.py --path $* $< > $@
|
||||
|
||||
$(BIN): $(PS5_SRCS) $(GEN_SRCS)
|
||||
$(CC) $(CFLAGS) $(LDFLAGS) -o $@ $^ $(LDADD)
|
||||
# Only LzmaDec.c changes behaviour under the switch: it stops defining its own
|
||||
# decoder and declares the external symbol instead. Everything else in the 7z
|
||||
# TU family is unaffected.
|
||||
ifneq ($(LZMA_DEC_OPT_FLAG),)
|
||||
ps5-obj/third_party/7z/LzmaDec.o: THIRD_PARTY_C_FLAGS_7Z += $(LZMA_DEC_OPT_FLAG)
|
||||
linux-obj/third_party/7z/LzmaDec.o: THIRD_PARTY_C_FLAGS_7Z += $(LZMA_DEC_OPT_FLAG)
|
||||
endif
|
||||
|
||||
# make does not track compiler-flag changes, so an object built with the old
|
||||
# flags is silently reused and the binary links "successfully" without the
|
||||
# feature. Two cases have already bitten:
|
||||
# * installing or removing jwasm flips LZMA_DEC_OPT_FLAG, and the asm object
|
||||
# just sits in the link line unreferenced (the binary came out
|
||||
# byte-identical, which is how the problem was noticed);
|
||||
# * adding -DHAVE_WZAES / -DHAVE_PKCRYPT, which only mz_zip.c and mz_crypt.c
|
||||
# compile differently -- without a rebuild the ZIP encryption streams are
|
||||
# never referenced and --gc-sections quietly drops them again.
|
||||
# Recording the flags in a stamp file invalidates the objects when the flags
|
||||
# actually change, instead of rebuilding them for every unrelated Makefile edit.
|
||||
PS5_FLAGS_STAMP := ps5-obj/.third_party_cflags
|
||||
LINUX_FLAGS_STAMP := linux-obj/.third_party_cflags
|
||||
|
||||
$(PS5_FLAGS_STAMP): FORCE
|
||||
@mkdir -p $(dir $@)
|
||||
@printf '%s\n' '$(THIRD_PARTY_C_FLAGS_7Z) $(LZMA_DEC_OPT_FLAG)' > $@.tmp
|
||||
@cmp -s $@.tmp $@ || { mv -f $@.tmp $@; echo ' [cflags] third-party flags changed -> rebuilding objects'; }
|
||||
@rm -f $@.tmp
|
||||
|
||||
$(LINUX_FLAGS_STAMP): FORCE
|
||||
@mkdir -p $(dir $@)
|
||||
@printf '%s\n' '$(THIRD_PARTY_C_FLAGS_7Z) $(LZMA_DEC_OPT_FLAG)' > $@.tmp
|
||||
@cmp -s $@.tmp $@ || { mv -f $@.tmp $@; echo ' [cflags] third-party flags changed -> rebuilding objects'; }
|
||||
@rm -f $@.tmp
|
||||
|
||||
.PHONY: FORCE
|
||||
FORCE:
|
||||
|
||||
# Note on -DVERSION_TAG / -DTITLE_ID: they live in CFLAGS, which make cannot
|
||||
# see -- but nothing here relies on make seeing them. The link target's *name*
|
||||
# carries the version ($(BIN) = web-file-mgr-$(VERSION_TAG).elf), so a bump
|
||||
# always misses the existing target and re-runs the link rule, and that rule
|
||||
# compiles every file in $(PS5_SRCS) on the spot (-x c, one clang invocation,
|
||||
# no intermediate .o). src/version.c and src/main.c -- the only two readers of
|
||||
# the macros -- are therefore always rebuilt with the new string. (Verified:
|
||||
# `strings` on the v1.9.3M ELF finds "v1.9.3M" once and "v1.9.2" zero times.)
|
||||
#
|
||||
# The version trap that *is* real: overriding VERSION_TAG back to a version
|
||||
# whose ELF already exists in the tree, with sources older than that file,
|
||||
# returns the existing file and silently skips the rebuild. Delete the stale
|
||||
# ELF, or build a differently named copy, when re-cutting a version.
|
||||
|
||||
ps5-obj/%.o: %.c $(PS5_FLAGS_STAMP)
|
||||
@mkdir -p $(dir $@)
|
||||
$(CC) $(if $(findstring third_party/7z,$<),$(THIRD_PARTY_C_FLAGS_7Z),$(THIRD_PARTY_C_FLAGS)) -c -o $@ $<
|
||||
|
||||
linux-obj/%.o: %.c $(LINUX_FLAGS_STAMP)
|
||||
@mkdir -p $(dir $@)
|
||||
$(HOST_CC) $(if $(findstring third_party/7z,$<),$(THIRD_PARTY_C_FLAGS_7Z),$(THIRD_PARTY_C_FLAGS)) -c -o $@ $<
|
||||
|
||||
# The assembler emits a plain ELF64 relocatable object, which both linkers
|
||||
# (prospero-clang++ for PS5, cc for linux) accept as-is.
|
||||
ps5-obj/$(LZMA_DEC_ASM_DIR)/LzmaDecOpt.o: $(LZMA_DEC_ASM_SRC)
|
||||
@mkdir -p $(dir $@)
|
||||
$(JWASM) -elf64 -q -DABI_LINUX -I$(LZMA_DEC_ASM_DIR) -Fo$@ $<
|
||||
|
||||
linux-obj/$(LZMA_DEC_ASM_DIR)/LzmaDecOpt.o: $(LZMA_DEC_ASM_SRC)
|
||||
@mkdir -p $(dir $@)
|
||||
$(JWASM) -elf64 -q -DABI_LINUX -I$(LZMA_DEC_ASM_DIR) -Fo$@ $<
|
||||
|
||||
ps5-obj/%.o: %.cpp
|
||||
@mkdir -p $(dir $@)
|
||||
$(CXX) $(UNRAR7_CXX_FLAGS_PS5) -c -o $@ $<
|
||||
|
||||
linux-obj/%.o: %.cpp
|
||||
@mkdir -p $(dir $@)
|
||||
$(HOST_CXX) $(UNRAR7_CXX_FLAGS_HOST) -c -o $@ $<
|
||||
|
||||
# Link with the C++ driver so libc++ (PS5) / libstdc++ (host) is pulled in
|
||||
# automatically for the unrar objects. The project's own C sources are passed
|
||||
# through -x c (clang++ would otherwise compile .c files as C++ and trip
|
||||
# -Wdeprecated); -x none restores extension-based handling for the .o files.
|
||||
$(BIN): $(PS5_SRCS) $(GEN_SRCS) $(PS5_TP_OBJS) $(PS5_ASM_OBJS)
|
||||
$(CXX) $(CFLAGS) $(LDFLAGS) -o $@ -x c $(filter %.c,$^) -x none $(PS5_TP_OBJS) $(PS5_ASM_OBJS) $(LDADD)
|
||||
$(STRIP) $@
|
||||
|
||||
$(LEGACY_BIN): $(PS5_SRCS) $(GEN_SRCS)
|
||||
$(CC) $(LEGACY_CFLAGS) $(LEGACY_LDFLAGS) -o $@ $^ $(LDADD)
|
||||
$(STRIP) $@
|
||||
|
||||
$(LINUX_BIN): $(LINUX_SRCS) $(GEN_SRCS)
|
||||
$(HOST_CC) $(LINUX_CFLAGS) -o $@ $^ $(LINUX_LDADD)
|
||||
$(LINUX_BIN): $(LINUX_SRCS) $(GEN_SRCS) $(LINUX_TP_OBJS) $(LINUX_ASM_OBJS)
|
||||
$(HOST_CXX) $(LINUX_CFLAGS) -o $@ -x c $(filter %.c,$^) -x none $(LINUX_TP_OBJS) $(LINUX_ASM_OBJS) $(LINUX_LDADD)
|
||||
$(HOST_STRIP) $@
|
||||
@@ -1,97 +1,703 @@
|
||||
<div align="right">
|
||||
<a href="README.md">English</a> · <a href="README.zh-CN.md">简体中文</a>
|
||||
</div>
|
||||
|
||||
# PS5 Web File Manager
|
||||
|
||||
A file manager for PS5 with a web UI. It is primarily intended for quickly and
|
||||
safely copying game dump folders from USB storage to internal storage.
|
||||
> Homebrew HTTP file manager for jailbroken PS5 consoles. Browse, edit, upload, download and extract ZIPs through any browser on the same network — single self-contained ELF payload, no external services, no telemetry.
|
||||
|
||||
## Brief
|
||||
**Version:** v1.9.2 · **Title ID:** `FMGR88888` · **License:** GPLv3+ · **Target:** `x86_64-sie-ps5`
|
||||
|
||||
PS5 web file manager payload. It runs an HTTP UI starting at port `8888`, installs a home screen launcher in Media catagory on startup when needed, and provides file operations from the PS5 browser. If `8888` is already in use, the payload tries the next port until one is available; the startup notification shows the actual listen port.
|
||||
---
|
||||
|
||||
## Overview
|
||||
|
||||
A payload ELF that runs an HTTP file manager inside a jailbroken PS5. Open `http://<PS5_IP>:8888/` from any browser on the LAN — including the PS5 browser itself — to manage files on attached USB storage and the user partition. Designed for safely copying game-dump folders from USB to internal storage, but it also handles general file management, in-place text editing, PKG preview/install, image preview, and ZIP extraction with built-in zip-bomb protection.
|
||||
|
||||
The same source tree builds a Linux binary for development and a PS5 payload ELF for deployment — see `make linux` below.
|
||||
|
||||
## What's new (unreleased)
|
||||
|
||||
- **Encrypted ZIP extraction works end to end** — both schemes:
|
||||
- traditional PKWARE ("ZipCrypto", what `zip -e` writes), and
|
||||
- WinZip AES-128/192/256 (compression method `99` plus the `0x9901` extra
|
||||
field, what `7z -mem=AES256` and WinZip write), for stored and deflated
|
||||
entries.
|
||||
|
||||
A missing or wrong password is reported as `extract_password` — the code the
|
||||
task overlay already knew how to translate, even though until now nothing
|
||||
could produce it for a ZIP. The SHA-1,
|
||||
HMAC-SHA1 and AES primitives live in a new local minizip-ng backend,
|
||||
`third_party/minizip-ng/src/mz_crypt_wfm.c` (PBKDF2 comes from the vendored
|
||||
`mz_crypt.c`); its header explains why they are implemented in-tree instead
|
||||
of delegating to another vendored library.
|
||||
- **Encrypted RAR is actually wired up.** v1.9 vendored an engine that *could*
|
||||
decrypt (`RARSetPassword`) but never called it, so encrypted archives were
|
||||
rejected. The password now reaches the engine, including for `-hp`
|
||||
header-encrypted archives, and a wrong password returns `extract_password`
|
||||
so the prompt can retry.
|
||||
- **Build fix: compiler-flag changes now invalidate objects.** `make` cannot
|
||||
see flag changes, so adding `-DHAVE_WZAES -DHAVE_PKCRYPT` left the existing
|
||||
`mz_zip.o` / `mz_crypt.o` in place — and because nothing referenced the new
|
||||
streams any more, `--gc-sections` quietly dropped the encryption code again
|
||||
while the link still "succeeded". The Makefile now records the third-party
|
||||
flag set in `ps5-obj/.third_party_cflags` and rebuilds only when it really
|
||||
changes. This is the same trap the `LzmaDec.o` rule was working around.
|
||||
- **The UI can now retry with a password.** An `extract_password` failure no
|
||||
longer ends in an error box: the attempt is re-sent with whatever the user
|
||||
types, up to three times, keeping the conflict policy and the large-file
|
||||
opt-in of the original request. Cancelling or submitting an empty box falls
|
||||
back to the original failure report. 7z keeps its up-front prompt, since an
|
||||
encrypted 7z header would otherwise cost a wasted scan.
|
||||
- **Encrypted 7z headers (`-mhe=on`) now open.** This was the last known format
|
||||
gap: with `-mhe=on` the file names, the folder table *and* every entry size
|
||||
sit inside the encrypted header, so the vendored SDK gives up with
|
||||
`SZ_ERROR_UNSUPPORTED` before it can list a single entry. A new module,
|
||||
`src/sevenz_header.c`, reads the header record, decodes its one folder with
|
||||
the project's own 7zAES path (`src/sevenz_chain.c`) and then hands the SDK a
|
||||
small virtual stream in which the encrypted record has been replaced by the
|
||||
plaintext — so the SDK goes on parsing exactly the archive it always did, and
|
||||
nothing on disk is touched. Archives whose header is merely *compressed*
|
||||
(`-mhc=on`, the default) are not modified in any way, and a wrong password
|
||||
comes back as `extract_password` like every other encrypted archive.
|
||||
- `tests/make-zip-enc-fixtures.bat` and three real fixtures under
|
||||
`tests/fixtures-real/` (`enc-zipcrypto.zip`, `enc-aes256.zip`,
|
||||
`enc-aes256-store.zip`, password `secret123`).
|
||||
- Host checks: **140 ZIP + 37 RAR = 177** (`tests/run-tests.sh`). The new
|
||||
encrypted-ZIP cases run against real archives in `tests/fixtures-real/`
|
||||
generated by the new `tests/make-zip-enc-fixtures.bat`.
|
||||
- A RAR archive that asks for a dictionary larger than this build supports no
|
||||
longer reports as a per-entry size problem: it gets its own
|
||||
`extract_dict_too_large` code and a message that names the required and the
|
||||
supported dictionary size. The build's behaviour is unchanged — such an
|
||||
archive is still refused, because honouring it would mean allocating the
|
||||
whole window in one shot, which is exactly what rarlab's own CLI refuses to
|
||||
do by default and what a 16 GB shared-memory console cannot afford.
|
||||
- 7z checks: **27 cases, 0 failures** (`tests/run-sevenz-tests.sh`), and the
|
||||
`KNOWN_GAPS` list that held `aeshe` is now empty — the encrypted-header
|
||||
fixture passes through both the folder decoder and the extraction facade.
|
||||
- The frontend retry flow has a headless check of its own —
|
||||
`node .build/ui_retry_test.mjs` loads the real `assets/main.js` into a stubbed
|
||||
DOM and asserts the remembered request, the retry cap and the give-up paths,
|
||||
including the regression case for a non-ASCII folder: **40 checks, 0 failures**.
|
||||
`node .build/ui_upload_menu_test.mjs` covers the markup side — every
|
||||
`data-i18n` key exists in both languages, the upload menu is wired to the
|
||||
right handlers, the classes it uses are styled, and the row-highlight rules
|
||||
are scoped so they cannot lose the cascade to a generic button rule, and that
|
||||
the extract button is never hidden — only disabled, with a message per reason:
|
||||
**40 checks, 0 failures**.
|
||||
- The tree builds to **903 448 B**, sha256
|
||||
`8ca47d5aaca75085b32641300cce30fadb7df7749cb6b53d04f129bcecc286b7`,
|
||||
`e_machine` `0x003e`. Rebuilt from the same tree, byte-identical both times;
|
||||
the built ELF was checked to contain the new frontend code, which is only
|
||||
reachable after gunzipping the embedded assets. Same 20 sections and **no new
|
||||
dynamic symbols**. The encrypted-archive work added +5 712 bytes of content
|
||||
(`.text` +4 880, `.rodata` +640, `.eh_frame*` +192); the fork marker below
|
||||
then added a further +0x100 (256 B), the corrected
|
||||
`err_extract_unsupported` copy another +0x40 (64 B), the upload menu +0x980
|
||||
(2 432 B), the frontend copy and CSS of the first fix round +0x100 (256 B)
|
||||
and the scoped row-highlight rules +0x180 (384 B), then the always-on extract
|
||||
button +0x240 (576 B) — every one of them to
|
||||
`.rodata` **and to no other section**, so the file is still 903 448 B across
|
||||
all six builds. An unchanged size is not evidence of an unchanged binary —
|
||||
compare sections with `readelf -SW`.
|
||||
- **The version string carries a fork marker: `v1.9.3M`.** Upstream releases are
|
||||
plain `vX.Y.Z`, so the trailing `M` (Modified) is what tells you which of the
|
||||
two projects a build came from. It is part of `VERSION_TAG`, so `/api/version`,
|
||||
the PS5 start-up notification, the stdout banner, the UI footer and the ELF
|
||||
file name all carry it at once, and the UI footer spells it out on hover. The
|
||||
file name changing also means a fork build can no longer shadow an upstream
|
||||
artifact of the same upstream version. See Credits.
|
||||
- Still open: end-to-end validation of the built ELF on a real console.
|
||||
|
||||
> This section describes **unreleased** work: the published release is still
|
||||
> `v1.9.2` and its binary does **not** contain any of it. The unreleased tree
|
||||
> identifies itself as `v1.9.3M`.
|
||||
|
||||
## What's new in v1.9.2
|
||||
|
||||
Version-string-only re-release. The `v1.9.1` tag sat four commits behind the
|
||||
tree that produced its binary, so the tag could not rebuild the published
|
||||
artifact; v1.9.2 is cut from the right commit. It is functionally identical to
|
||||
the v1.9.1 binary — the only change is the baked-in version string.
|
||||
|
||||
## What's new in v1.9
|
||||
|
||||
- **RAR engine replaced with the official rarlab UnRAR 7.20.1**
|
||||
(`third_party/unrar7/`, replacing dmc_unrar). This is what actually
|
||||
makes RAR extraction work on real files: dmc_unrar could not decode
|
||||
archives written by **WinRAR 6.x/7.x** (RAR5 "v6" compression) and had
|
||||
no multi-volume support — both now work.
|
||||
- **RAR5 "v6" archives extract** (the v1.8-era "corrupt archive" report
|
||||
on WinRAR 6/7 files is gone).
|
||||
- **Multi-volume RAR** (`.part01.rar` chains): unrar stitches the parts by
|
||||
name when the full set sits next to the volume you open.
|
||||
- Engine can decrypt encrypted RAR (`RARSetPassword`) — password UI /
|
||||
API plumbing still pending, encrypted archives are rejected for now.
|
||||
- Host tests now run real archives (v6 / encrypted / 3-volume fixtures
|
||||
committed under `tests/fixtures-real/`): **70 ZIP + 24 RAR = 94 checks**.
|
||||
|
||||
## What's new in v1.8
|
||||
|
||||
- **Single-volume RAR extraction** via the vendored FLOSS library
|
||||
[`dmc_unrar`](https://github.com/DrMcCoy/dmc_unrar) (GPL-2.0-or-later).
|
||||
RAR 1.5, 2.x, 3.x, 4.x and 5.x archives are supported. `.rar` files
|
||||
appear in the file list with the **Extract** button enabled; the button
|
||||
is greyed out on `.part02+.rar` sub-volumes with the tooltip "select
|
||||
the main volume instead" — v1.8 cannot stitch multi-volume RARs (see
|
||||
the [RAR extraction](#rar-extraction) section below).
|
||||
- **Shared extraction protocol** between the new `src/rar_extract.c`
|
||||
engine and the existing `src/zip_extract.c` engine: same `zipx_status_t`
|
||||
codes, same `zipx_limits_t` profile (default / `large=1`), same
|
||||
three-phase model (`scan → extract → publish → cleanup`), same staging
|
||||
directory layout, same conflict policy, same error mapping into the
|
||||
task UI. The dispatcher in `src/extract.c` is one tiny
|
||||
`ends_with_ci(…)` switch.
|
||||
- **14 new host-side C tests** (`tests/test_rar_extract.c`) wired into
|
||||
the existing `tests/run-tests.sh`. Coverage: format dispatch, error
|
||||
translation across every `DMC_UNRAR_*` code that affects RAR users,
|
||||
limit-profile handoff. Total host checks: **69 ZIP + 14 RAR = 83**.
|
||||
- **Documentation**: [`CHANGELOG.md`](./CHANGELOG.md),
|
||||
[`docs/UPGRADE-v1.8-rar-support.md`](./docs/UPGRADE-v1.8-rar-support.md)
|
||||
and the vendoring decision tree at
|
||||
[`third_party/unrar7/VENDORED.md`](./third_party/unrar7/VENDORED.md)
|
||||
(v1.8 shipped it as `third_party/unrar/VENDORED.md`).
|
||||
- See the [dedicated section](#rar-extraction) below for scope and the
|
||||
limitations that come from using dmc_unrar (no multi-volume, no
|
||||
encryption in v1.8 — both lift in v1.9 when the library is replaced).
|
||||
|
||||
## What's new in v1.8.1
|
||||
|
||||
- **Default ZIP limits relaxed** (companion to v1.7's large profile).
|
||||
v1.7 shipped with a 64 GiB default per-entry cap, which was too
|
||||
aggressive for typical PS5 system-backup ZIPs (200-300 GiB). v1.8.1
|
||||
raises the default profile to **1 TiB total / 256 GiB per entry /
|
||||
500 : 1 ratio**, with the `large=1` opt-in kept at 2 TiB / 1 TiB /
|
||||
1000 : 1. The frontend threshold rises from 60 GiB to 240 GiB so
|
||||
common system-backup archives no longer trigger the prompt.
|
||||
- RAR extraction inherits the new defaults (rar_extract.c threads
|
||||
`c->limits` from the engine — no engine change required).
|
||||
- Rationale: the real zip-bomb defence is `check_space()` (statvfs-based
|
||||
real disk-space check before staging) + `max_ratio` (declared
|
||||
compression ratio cap). The size caps are a UX guard, not a security
|
||||
boundary.
|
||||
|
||||
## What's new in v1.8.2
|
||||
|
||||
- **Default ZIP limits relaxed again** for the 3A-game single-file case.
|
||||
A single ~300 GiB uncompressed file inside an archive was still
|
||||
silently rejected by v1.8.1 (the default scan returns
|
||||
`ZIPX_ERR_LIMIT_FILE_SIZE` before the request ever reaches the
|
||||
frontend confirmation prompt). v1.8.2 raises the default profile to
|
||||
**2 TiB total / 512 GiB per entry / 500 : 1 ratio**, with the `large=1`
|
||||
opt-in bumped to 4 TiB / 1 TiB / 1000 : 1. Frontend threshold rises
|
||||
from 240 GiB to 480 GiB.
|
||||
- **Two PS5-only build fixes** discovered when cross-compiling for the
|
||||
PS5 target. The host-side test suite (`tests/run-tests.sh`) had
|
||||
silently accepted both because it links the same sources but uses
|
||||
gcc rather than clang 18 and a different include path:
|
||||
- `Makefile` CFLAGS: add `-Ithird_party/unrar` so `src/rar_extract.c`
|
||||
can find the project-authored `dmc_unrar_api.h` facade header.
|
||||
- `src/extract.c`: move `extract_progress()` definition above
|
||||
`extract_dispatch()` so the implicit function declaration is not
|
||||
flagged by `-Werror=implicit-function-declaration` (clang 18 in the
|
||||
PS5 SDK is stricter than the host gcc used by tests).
|
||||
- **Release artifact** for v1.8.2: `web-file-mgr.elf` — 509 704 bytes,
|
||||
sha256 `1b2c3d68b35e32737105f17d14a80a3c159ceca0cabd274ee168cbcd81906f65`,
|
||||
ELF class 64, little-endian, e_machine `0x003e` (x86_64-sie-ps5).
|
||||
- Tests: **84 host-side checks** (70 ZIP + 14 RAR), 0 failures. PS5
|
||||
cross-compile succeeds end-to-end.
|
||||
|
||||
## What's new in v1.7
|
||||
|
||||
- **ZIP large-file profile** (opt-in via the new `large=1` argument on `/api/extract`): relaxed caps of **2 TiB** archive total, **1 TiB** per entry, **1000 : 1** compression ratio. The frontend prompts for confirmation whenever the archive on disk is larger than **60 GiB**; the server only activates the profile when the user explicitly agrees.
|
||||
- **Stricter default ZIP profile** stays safe: **1 TiB** total / **256 GiB** per entry / **500 : 1** ratio. A 4 MiB compressed payload that expands to 800 GiB still gets rejected before any output file is opened.
|
||||
- **69 host-side C tests** (`tests/run-tests.sh`) now cover path traversal, ZIP64, encryption rejection, ratios, conflict policies and the new large-file profile (`tests/test_zip_extract.c`).
|
||||
- Earlier refinements — see `git log` since v1.6.
|
||||
|
||||
## Screenshots
|
||||
|
||||
<p>
|
||||
<a href="docs/screenshots/20260617_231827.376.jpg" target="_blank"><img src="docs/screenshots/20260617_231827.376.jpg" width="31%" alt="PS5 Web File Manager screenshot 1"></a>
|
||||
<a href="docs/screenshots/20260619_131432.399.jpg" target="_blank"><img src="docs/screenshots/20260619_131432.399.jpg" width="31%" alt="PS5 Web File Manager screenshot 2"></a>
|
||||
<a href="docs/screenshots/20260617_232348.855.jpg" target="_blank"><img src="docs/screenshots/20260617_232348.855.jpg" width="31%" alt="PS5 Web File Manager screenshot 3"></a>
|
||||
<a href="docs/screenshots/20260619_131811.644.jpg" target="_blank"><img src="docs/screenshots/20260619_131811.644.jpg" width="31%" alt="PS5 Web File Manager screenshot 4"></a>
|
||||
<a href="docs/screenshots/20260619_131535.239.jpg" target="_blank"><img src="docs/screenshots/20260619_131535.239.jpg" width="31%" alt="PS5 Web File Manager screenshot 5"></a>
|
||||
<a href="docs/screenshots/20260620_232728.533.jpg" target="_blank"><img src="docs/screenshots/20260620_232728.533.jpg" width="31%" alt="PS5 Web File Manager screenshot 6"></a>
|
||||
</p>
|
||||
|
||||
## Features
|
||||
|
||||
- List files and folders.
|
||||
- Copy, move, delete, rename, and create folders.
|
||||
- Multi-select operations.
|
||||
- Copy/move by choosing sources first, then pasting or moving them into the current folder.
|
||||
- Conflict prompts for overwriting files and merging folders.
|
||||
- Full-screen task overlay with progress, speed, ETA, cancel support, and task recovery after reopening the browser while the payload process is still running.
|
||||
- Copied/moved files and folders are set to `0777` where the filesystem supports Unix permissions. FAT/exFAT-style filesystems may ignore chmod.
|
||||
- Chinese and English UI. The browser language is read from `navigator.languages` / `navigator.language`; Chinese uses `zh`, everything else uses English.
|
||||
- Startup notification showing the app name, version, and listen port.
|
||||
- **Browse** — list files and folders; sort by name, type, size, mtime or permissions. Last sort mode persists in `localStorage`.
|
||||
- **Permissions** — toggle read/write/execute with checkboxes, or paste a validated four-digit octal mode.
|
||||
- **Operations** — copy, move, delete (recursive, no recycle bin), rename, create files and folders.
|
||||
- **Editor** — in-place UTF-8 text editor for files ≤ 1 MiB across a curated extension list: `.txt .json .xml .ini .cfg .conf .md .log .lua .js .css .html .htm .c .h .cpp .hpp .sh .csv .yaml .yml .shn`.
|
||||
- **Multi-select** — copy, move, delete or tar-download many items in one go.
|
||||
- **Upload** — single files or folder trees from any device on the LAN (hidden in the PS5 browser). Atomic temp + rename.
|
||||
- **Download** — single file as raw bytes, or folders/multi-select as a streaming `.tar`. Hidden in the PS5 browser.
|
||||
- **Tasks** — full-screen overlay with delayed show, live progress, throughput, ETA, cancel, and recovery if the browser is closed and reopened mid-task.
|
||||
- **Archive extraction** — ZIP, RAR and 7z, all behind the same zip-bomb / traversal / ratio protection. ZIP covers stored / deflated / ZIP64 plus **encrypted** entries (ZipCrypto and WinZip AES-128/192/256); RAR covers RAR4 + RAR5 including WinRAR 6/7 "v6", multi-volume, and `-p` / `-hp` encryption; 7z covers Copy / LZMA / LZMA2 / PPMd, the Delta and BCJ2 filters, `.7z.001` volumes, 7zAES, and `-mhe=on` encrypted headers. See the [ZIP extraction](#zip-extraction) and [RAR extraction](#rar-extraction) sections below for scope.
|
||||
- **Encrypted archives** — a wrong or missing password is reported as `err_extract_password` and retried through a password prompt, up to three times, with the original conflict policy and large-file opt-in preserved. 7z asks for the password up front instead, so an encrypted header does not cost a wasted scan.
|
||||
- **PKG** — install and preview `.pkg` files.
|
||||
- **Images** — preview `.png .jpg .jpeg .gif .bmp .webp`.
|
||||
- **Localization** — English + Simplified Chinese, auto-selected from `navigator.languages`.
|
||||
- **Mobile-friendly** — responsive layout with wrapped toolbars and horizontally scrollable file lists.
|
||||
|
||||
## Quickstart
|
||||
|
||||
1. **Build** the ELF:
|
||||
|
||||
```sh
|
||||
export PS5_PAYLOAD_SDK=/opt/ps5-payload-sdk # see "Build" for SDK setup
|
||||
make
|
||||
```
|
||||
2. **Send** the payload to the PS5 (default ELF-loader port `9021`):
|
||||
|
||||
```sh
|
||||
nc -q0 "$PS5_HOST" 9021 < web-file-mgr.elf
|
||||
```
|
||||
3. **Read** the on-screen PS5 notification — it prints the actual listen port (default `8888`).
|
||||
4. **Open** `http://<PS5_IP>:<port>/` in any browser on the same LAN — the PS5 browser works too.
|
||||
5. On first run, the payload also writes a **Media**-category home-screen launcher; existing launcher files are not overwritten.
|
||||
|
||||
## Build
|
||||
|
||||
It depends on PS5 payload SDK first: [ps5-payload-dev/sdk](https://github.com/ps5-payload-dev/sdk#quick-start)
|
||||
Requires [ps5-payload-dev/sdk](https://github.com/ps5-payload-dev/sdk#quick-start):
|
||||
|
||||
```sh
|
||||
export PS5_PAYLOAD_SDK=/opt/ps5-payload-sdk
|
||||
```
|
||||
|
||||
This project links against `libmicrohttpd`. `make` checks for it before building and runs the installer script automatically if it is missing:
|
||||
This project links against `libmicrohttpd`. `make` checks for it before building and runs the installer automatically when missing:
|
||||
|
||||
```sh
|
||||
make
|
||||
```
|
||||
|
||||
If the build host has no network access, download the libmicrohttpd source tarball yourself and run the dependency installer once:
|
||||
If the build host has no network access, drop the libmicrohttpd tarball in advance and run the installer manually:
|
||||
|
||||
```sh
|
||||
LIBMICROHTTPD_TARBALL=/path/to/libmicrohttpd-1.0.1.tar.gz \
|
||||
./install-libmicrohttpd.sh
|
||||
```
|
||||
|
||||
Then build again:
|
||||
|
||||
```sh
|
||||
make
|
||||
```
|
||||
|
||||
The output is:
|
||||
Output:
|
||||
|
||||
```text
|
||||
web-file-mgr.elf
|
||||
web-file-mgr.elf (~several hundred KiB, larger in v1.9 with unrar; x86_64-sie-ps5)
|
||||
```
|
||||
|
||||
For pure UI/JS work without the PS5 toolchain:
|
||||
|
||||
```sh
|
||||
make linux
|
||||
./web-file-mgr-linux
|
||||
```
|
||||
|
||||
The Linux build does **not** include the PS5 home-screen launcher installer.
|
||||
|
||||
## Usage
|
||||
|
||||
Start an ELF loader on the PS5. The common listener port is `9021`.
|
||||
|
||||
Send the built payload with netcat or NetCat GUI:
|
||||
Start an ELF loader on the PS5 (port `9021` is common). Send the payload:
|
||||
|
||||
```sh
|
||||
export PS5_HOST=ps5_ip_address
|
||||
nc -q0 "$PS5_HOST" 9021 < web-file-mgr.elf
|
||||
```
|
||||
|
||||
After the payload starts, the PS5 notification shows the app name, version, and actual listen port. Open the shown URL in the PS5 browser, for example:
|
||||
After the payload starts, the PS5 notification shows the app name, version and actual listen port. Open the URL it prints, for example:
|
||||
|
||||
```text
|
||||
http://${PS5_IP_ADDRESS}:8888/
|
||||
```
|
||||
|
||||
On first startup the payload installs a `PS5 Web File Manager` web shortcut in the Media category when needed. If the payload had to use a fallback port such as `8889`, use the port shown in the startup notification.
|
||||
If the payload had to fall back to a different port (e.g. `8889`), use whatever port the notification shows — the URL is not hard-coded.
|
||||
|
||||
On first startup, the payload installs a `PS5 Web File Manager` shortcut in the Media category when needed. Existing launcher files are preserved; only missing ones are written.
|
||||
|
||||
## ZIP extraction
|
||||
|
||||
Plain and encrypted ZIPs — stored / deflated / ZIP64, in the clear or with either encryption scheme (traditional PKWARE "ZipCrypto" and WinZip AES-128/192/256). The engine is a standalone three-phase module (`scan → extract → publish → cleanup`) at `src/zip_extract.{c,h}`, with a separate host-side C test suite. Each entry is first written into a staging directory (`*.wfm-part-*`), then atomically renamed into the destination. There is deliberately **no per-entry `fsync`** anywhere in the extract path — the whole pipeline is "sync nothing, rename everything", because publish is rename-only and there is no resume feature to protect (measured ≥14x on an 8000-file archive; see `docs/EXTRACTION-PERF.md`). Any failure mid-archive rolls back partial changes; cancel and fatal errors always clean up staging.
|
||||
|
||||
### Limits
|
||||
|
||||
| Limit | Default profile | Large profile (`ZIPX_LIMITS_LARGE`) |
|
||||
|---|---|---|
|
||||
| `max_entries` | 200 000 | 500 000 |
|
||||
| `max_total_bytes` (uncompressed) | 2 TiB | 4 TiB |
|
||||
| `max_file_bytes` (per entry) | 512 GiB | 1 TiB |
|
||||
| `max_ratio` (uncompressed / compressed) | 500 : 1 | 1000 : 1 |
|
||||
| `max_depth` (folder nesting) | 32 | 32 |
|
||||
| `max_name_len` / `max_path_len` | 255 / 1024 | 255 / 1024 |
|
||||
|
||||
The **default profile** is shipped safe: a 4 MiB compressed blob that decodes to 800 GiB is rejected before any output file is opened. The **large profile** is engaged **only** when the request includes `large=1` — the archive dialog prompts the user automatically whenever the archive on disk is larger than `LARGE_FILE_THRESHOLD_BYTES` (480 GiB by default; configurable in `assets/main.js`). Confirming the prompt is the user's explicit opt-in; the server still records nothing extra on its own.
|
||||
|
||||
### Security checks
|
||||
|
||||
The engine refuses to extract:
|
||||
|
||||
- Path traversal (`..` segments, absolute POSIX paths, Windows drive letters).
|
||||
- Symbolic links, devices, FIFOs, sockets (`ZIPX_ERR_SPECIAL`).
|
||||
- Duplicate entries or directory/file name clashes inside the same archive.
|
||||
- Archives whose expanded size, entry count, depth, name length or compression ratio breach the active profile.
|
||||
|
||||
Encrypted entries are no longer a refusal: the password arrives as `password=`
|
||||
on `/api/extract`, and a missing or wrong one comes back as `ZIPX_ERR_PASSWORD`
|
||||
(`err_extract_password` in the UI) so the prompt can retry. The scan phase still
|
||||
runs for encrypted archives — handing over a password does not skip the limits.
|
||||
|
||||
### Conflict policy
|
||||
|
||||
Passed as `conflict=` on `/api/extract`:
|
||||
|
||||
- `fail` (default) — refuse to overwrite any existing target.
|
||||
- `overwrite` — replace existing files; merge into existing folders.
|
||||
- `merge` — keep existing files, add new ones.
|
||||
|
||||
### Tuning the threshold
|
||||
|
||||
The 480 GiB frontend threshold lives in `assets/main.js`:
|
||||
|
||||
```js
|
||||
const LARGE_FILE_THRESHOLD_BYTES = 480 * 1024 * 1024 * 1024;
|
||||
```
|
||||
|
||||
Set it to `Infinity` to silence the prompt, lower it to be more conservative, or remove the call entirely — the server still respects `large=1` regardless of the threshold.
|
||||
|
||||
## RAR extraction
|
||||
|
||||
A RAR extraction engine (`src/rar_extract.{c,h}`) backed by the **official
|
||||
rarlab UnRAR source** (`third_party/unrar7/`, version 7.20.1, compiled as a
|
||||
static library and driven through its C-compatible DLL API). Files with the
|
||||
extension `.rar` get the same **Extract** button as `.zip` files; the engine
|
||||
is dispatched by `src/extract.c` based on extension.
|
||||
|
||||
> v1.9 replaced the v1.8 engine (dmc_unrar 1.7.0). dmc_unrar could not
|
||||
> decode archives written by WinRAR 6.x/7.x (RAR5 "v6" compression) and had
|
||||
> no multi-volume support; unrar handles both natively.
|
||||
|
||||
### Scope
|
||||
|
||||
| Format | Support | Notes |
|
||||
|---|---|---|
|
||||
| RAR 1.5 → 4.x (incl. 2.9 / 3.6 / 4.0) | ✅ | |
|
||||
| RAR 5.0 and **5.0 "v6"** (WinRAR 6.x / 7.x) | ✅ | The v1.9 trigger |
|
||||
| Solid blocks, dictionary up to 1 GiB | ✅ | |
|
||||
| PPMd decompression (RAR 3.0+) | ✅ | |
|
||||
| **Multi-volume** (`.part01.rar` + `.part02.rar` + …) | ✅ | unrar stitches parts by name when the whole set sits next to the volume you open. Select the first volume (`name.part1.rar` / `name.part01.rar`); non-first volumes are still greyed out in the UI with a hint. |
|
||||
| **Encrypted RAR** | ✅ | Both `-p` data encryption and `-hp` header encryption. The password reaches the engine as `password=` on `/api/extract` (`RARSetPassword` runs after `RAROpenArchiveEx` and before the first `RARReadHeaderEx`); a missing or wrong one returns `ZIPX_ERR_PASSWORD` so the prompt can retry. |
|
||||
| Symbolic links / FIFOs / sockets / devices | ❌ | Rejected with `ZIPX_ERR_SPECIAL` (mirrors ZIP behaviour) |
|
||||
| RAR 1.3 (pre-1.4) | ❌ | Rejected upstream by unrar |
|
||||
|
||||
When an archive is rejected, the user gets an `extract_unsupported`
|
||||
failure with the file name as the detail argument. The frontend already
|
||||
shows this with the typical bilingual retry guidance.
|
||||
|
||||
### Limits
|
||||
|
||||
The RAR engine re-uses the ZIP limits table verbatim — there is no RAR
|
||||
profile table on top. Defaults and the `large=1` opt-in are identical:
|
||||
|
||||
| Limit | Default profile | Large profile (`large=1`) |
|
||||
|---|---|---|
|
||||
| `max_entries` | 200 000 | 500 000 |
|
||||
| `max_total_bytes` (uncompressed) | 2 TiB | 4 TiB |
|
||||
| `max_file_bytes` (per entry) | 512 GiB | 1 TiB |
|
||||
| `max_ratio` (uncompressed / compressed) | 500 : 1 | 1000 : 1 |
|
||||
| `max_depth` (folder nesting) | 32 | 32 |
|
||||
| `max_name_len` / `max_path_len` | 255 / 1024 | 255 / 1024 |
|
||||
|
||||
Large-profile RAR extraction uses the same `LARGE_FILE_THRESHOLD_BYTES`
|
||||
(480 GiB) prompt as ZIP — the frontend treats `.rar` and `.zip` the same
|
||||
way for the prompt, and the server only ever activates the large caps
|
||||
when the request carries `large=1` (opt-in).
|
||||
|
||||
### Security checks
|
||||
|
||||
The RAR engine applies the same checks as the ZIP engine — re-uses
|
||||
`zipx_status_t` codes, so the task UI's `err_extract_unsafe_name`,
|
||||
`err_extract_too_deep`, `err_extract_ratio`, etc. all fire identically:
|
||||
|
||||
- Path traversal (`..` segments, absolute POSIX paths, Windows drive
|
||||
letters, `\` treated as a path separator after a `Rar!\x1a\x07…`
|
||||
header, etc.).
|
||||
- Symbolic links, FIFOs, sockets, devices.
|
||||
- Duplicate entries or directory/file name clashes inside the archive.
|
||||
- Archive size, entry count, depth, name length or compression ratio
|
||||
breaches of the active profile.
|
||||
|
||||
### Vendoring and licence
|
||||
|
||||
`third_party/unrar7/` is a verbatim copy of the official **rarlab UnRAR
|
||||
source** (7.20.1), mirrored by
|
||||
[`opello/unrar`](https://github.com/opello/unrar) at commit `97e1780`. It is
|
||||
distributed under the **UnRAR freeware licence** (see
|
||||
`third_party/unrar7/license.txt`): it may be used in any software to handle
|
||||
RAR archives, but may not be used to develop a RAR-compatible *archiver* or
|
||||
re-create the RAR compression algorithm. The project-authored facade
|
||||
`third_party/unrar7/unrar_c_api.h` carries the project's own licence.
|
||||
|
||||
> The v1.8 engine `third_party/unrar/dmc_unrar.c` (DrMcCoy/dmc_unrar 1.7.0,
|
||||
> GPL-2.0-or-later) was removed in v1.9; its notice lives in git history.
|
||||
|
||||
### Encrypted RAR
|
||||
|
||||
The password channel is complete: `password=` on `/api/extract` is handed to
|
||||
the engine, and `RARSetPassword` runs after `RAROpenArchiveEx` and before the
|
||||
first `RARReadHeaderEx` — the order unrar needs to decrypt a `-hp` header. A
|
||||
missing or wrong password comes back as `ZIPX_ERR_PASSWORD`
|
||||
(`err_extract_password` in the UI), which is what raises the password prompt
|
||||
and re-sends the original request, up to three times.
|
||||
|
||||
## 7z extraction
|
||||
|
||||
The 7z engine (`src/sevenz_extract.{c,h}`) is built on the LZMA SDK decode
|
||||
subset plus the project's own pull-based codec chain
|
||||
(`src/sevenz_chain.c`). Files ending in `.7z` get the same **Extract** button as
|
||||
`.zip` and `.rar`; `src/extract.c` dispatches by extension, and the engine
|
||||
re-uses the same three-phase model, limit profiles and conflict policy.
|
||||
|
||||
> Added in v1.9.1. The SDK's own `SzArEx` path only understands folders with up
|
||||
> to four coders, which cannot express BCJ2's five — hence the self-parsed
|
||||
> folder table and the pull-based chain.
|
||||
|
||||
A note on the header. 7-Zip keeps the archive header at the end of the file and
|
||||
compresses it when it grows (`-mhc=on`, the default), which is why the header
|
||||
region normally starts with an `k7zIdEncodedHeader` record describing one
|
||||
folder. With `-mhe=on` that folder is *also* encrypted, and since it holds the
|
||||
file names, the folder table and every entry size, the vendored SDK gives up on
|
||||
the whole archive before listing anything. `src/sevenz_header.c` handles that
|
||||
case: it reads the record, decodes its folder through the same 7zAES path the
|
||||
content uses, and then presents the SDK with a virtual stream whose header
|
||||
region is the plaintext — the archive on disk is never written to, and an
|
||||
archive whose header is merely compressed is not touched at all.
|
||||
|
||||
### Scope
|
||||
|
||||
| Format | Support | Notes |
|
||||
|---|---|---|
|
||||
| Copy / LZMA / LZMA2 (incl. ZIP64-style sizes) | ✅ | A single-coder pure-LZMA2 folder decodes multi-threaded (`src/sevenz_mt.c`, 8 threads) |
|
||||
| BCJ2 (x86 branch converter) | ✅ | Through the self-written chain; not expressible in the SDK's `SzArEx` |
|
||||
| Multi-coder folders, Delta filter, PPC / IA64 / ARM / ARMT / SPARC converters | ✅ | Parsed by `src/sevenz_chain.c` |
|
||||
| **Volumes** (`.7z.001` / `.z01` chains) | ✅ | `src/sevenz_volstream.c` stitches by name; open the first volume |
|
||||
| **Content encryption** (7zAES, AES-256-CBC) | ✅ | The engine decrypts; the frontend asks for the password up front (so an unencrypted archive does not pay a wasted scan), passes it as `password=`, and retries on `ZIPX_ERR_PASSWORD` |
|
||||
| **`-mhe=on` (encrypted header)** | ✅ | `src/sevenz_header.c` decodes the header record itself (through the same 7zAES path) and hands the SDK a virtual stream carrying the plaintext; a wrong password reports `ZIPX_ERR_PASSWORD`, so the prompt retries like any other encrypted archive |
|
||||
| `-mhc=off` (uncompressed header) | ✅ | Plain headers were always readable; they are now read one byte at a time and left alone |
|
||||
|
||||
### Limits
|
||||
|
||||
The 7z engine re-uses the ZIP limits table verbatim — see
|
||||
[ZIP extraction → Limits](#limits).
|
||||
|
||||
### Security checks
|
||||
|
||||
The same `zipx_status_t` codes and the same checks as ZIP and RAR: path
|
||||
traversal, special files, duplicate/clashing entries, and size, entry-count,
|
||||
depth, name-length or ratio breaches of the active profile.
|
||||
|
||||
## Verification
|
||||
|
||||
After `make`, sanity-check the produced ELF:
|
||||
|
||||
```sh
|
||||
ls -la web-file-mgr.elf # size grew in v1.9 (unrar static library); ~509 KiB was v1.8.3
|
||||
sha256sum web-file-mgr.elf # record the digest in your release notes
|
||||
file web-file-mgr.elf # expect "ELF 64-bit LSB pie executable, x86-64"
|
||||
od -An -tx1 -N20 web-file-mgr.elf | head -2 # magic 7f45 4c46 0201 + e_machine 003e
|
||||
```
|
||||
|
||||
The `e_machine = 0x003e` confirms the PS5 target triple `x86_64-sie-ps5`. The `e_type = 3` (`ET_DYN`) confirms the position-independent payload expected by ELF loaders.
|
||||
|
||||
## Tests
|
||||
|
||||
A POSIX/host-side C test suite covers the ZIP, RAR and 7z engines and runs on
|
||||
any Linux / macOS / MSYS shell without the PS5 SDK:
|
||||
|
||||
```sh
|
||||
cd tests && bash run-tests.sh # ZIP + RAR suites
|
||||
bash run-sevenz-tests.sh # 7z suite (needs MinGW gcc + a 7-Zip binary)
|
||||
```
|
||||
|
||||
Output is a per-case `check`-style report — **177 checks** on the current `main`
|
||||
(140 ZIP + 37 RAR), 0 failures. Coverage:
|
||||
|
||||
- ZIP entry parsing (stored + deflated + ZIP64)
|
||||
- Path traversal, absolute paths, backslash, Windows drive letters
|
||||
- Symbolic links, FIFOs, bad CRC, truncated archives, non-ZIP files
|
||||
- Limits: `entries`, `total_bytes`, `file_bytes`, `ratio`, `depth`, `name_len`
|
||||
- Conflict policies: `fail` / `overwrite` / `merge`
|
||||
- Cancellation in every phase
|
||||
- **Encrypted archives** — each real fixture is run four ways (no password,
|
||||
empty password and wrong password → `ZIPX_ERR_PASSWORD`; correct password →
|
||||
success with a byte-level content check): `enc-zipcrypto.zip`,
|
||||
`enc-aes256.zip` and `enc-aes256-store.zip` on the ZIP side, `enc-v6.rar` on
|
||||
the RAR side. Two further cases prove the limits still apply once a password
|
||||
has been handed over, and that the failing paths publish nothing.
|
||||
- **Large-file profile** — `medium_bomb.zip` (ratio ≈ 238) is rejected under default caps and accepted under large caps; lowered large caps still enforce.
|
||||
- **RAR engine** (`tests/test_rar_extract.c`, 37 checks) — format
|
||||
dispatch (renamed ZIP rejected, junk blob rejected), error translation
|
||||
across every reachable engine code, limits handoff (the
|
||||
`large=1` opt-in flows into `rar_extract()` unchanged), the oversized
|
||||
dictionary path above, plus the real-archive coverage above.
|
||||
- **7z engine** (`tests/test_sevenz_extract.c` + `tests/run-sevenz-tests.sh`) —
|
||||
byte-for-byte comparison against real `.7z` fixtures, encrypted-header
|
||||
rejection, conflicts under every policy, cancellation, limits, a missing
|
||||
destination parent, and the guarantee that a failure publishes nothing and
|
||||
cleans up its staging tree.
|
||||
|
||||
The frontend retry flow has its own headless check —
|
||||
`node .build/ui_retry_test.mjs` loads the real `assets/main.js` into a stubbed
|
||||
DOM and asserts the remembered request, the retry cap and the give-up paths:
|
||||
**27 checks, 0 failures**. It lives in `.build/` (outside the gitignore
|
||||
whitelist), so it is a development-time script rather than a committed test.
|
||||
|
||||
## Project layout
|
||||
|
||||
```
|
||||
.
|
||||
├── Makefile # PS5 + Linux builds (VERSION_TAG v1.9.3M)
|
||||
├── install-libmicrohttpd.sh # one-shot dependency installer
|
||||
├── gen-asset-module.py # embeds assets/* as gzip-compressed C arrays
|
||||
├── assets/ # HTML / CSS / JS / icons / param.json
|
||||
├── src/ # C payload sources
|
||||
│ ├── main.c websrv.c filemgr.c # entry, HTTP frontend, task model
|
||||
│ ├── upload.c download.c text.c # stream and in-place edit handlers
|
||||
│ ├── extract.c # /api/extract dispatcher (ZIP + RAR + 7z)
|
||||
│ ├── zip_extract.{c,h} zipx_common.c # ZIP engine (minizip-ng backend)
|
||||
│ ├── zipx_volume.c zipx_volstream.c # ZIP volume detection + concatenating stream
|
||||
│ ├── rar_extract.{c,h} # RAR engine (rarlab UnRAR 7.20.1 backend)
|
||||
│ ├── sevenz_extract.{c,h} sevenz_chain.{c,h} # 7z engine, self-parsed codec chain
|
||||
│ ├── sevenz_header.{c,h} # 7z header reader / -mhe=on decryption
|
||||
│ ├── sevenz_mt.c sevenz_volstream.c # multi-threaded LZMA2 + .7z.001 volumes
|
||||
│ ├── app_installer.c pkg_installer.c pkg_info.c # PS5 PKG preview / install
|
||||
│ └── demangle_stub.c cpu_support_stub.c # size / portability stubs
|
||||
├── third_party/ # vendored libraries
|
||||
│ ├── unrar7/ # rarlab UnRAR 7.20.1 — RAR engine
|
||||
│ ├── minizip-ng/ # 4.2.2, trimmed to the read path
|
||||
│ ├── 7z/ # LZMA SDK 26.03 decode subset
|
||||
│ └── zlib/ # minizip's compression backend
|
||||
├── tests/ # POSIX/host test suite
|
||||
│ ├── test_zip_extract.c test_rar_extract.c test_sevenz_extract.c
|
||||
│ ├── sevenz_chain_e2e.c sevenz_e2e.c bigfile_e2e.c
|
||||
│ ├── make_fixtures.py make_sevenz_fixtures.py make_split_fixtures.py
|
||||
│ ├── run-tests.sh # one-shot runner (ZIP + RAR suites)
|
||||
│ ├── run-sevenz-tests.sh # 7z suite, carries the KNOWN_GAPS list
|
||||
│ ├── bench_driver.py bench_formats.py # throughput benchmarks
|
||||
│ ├── compat/ # tiny Win32/MSYS shims
|
||||
│ └── fixtures/ fixtures-7z/ fixtures-real/
|
||||
├── docs/
|
||||
│ ├── HANDOVER.md # v1.8-era playbook, historical — see the root HANDOVER.md
|
||||
│ ├── SIZE-OPTIMIZATION.md # ELF size analysis + per-symbol ledger
|
||||
│ ├── EXTRACTION-PERF.md # decompression benchmarks
|
||||
│ ├── REWRITE-FEASIBILITY.md # engine-extraction study
|
||||
│ ├── UPSTREAM-V1.8-COMPARISON.md
|
||||
│ ├── UPGRADE-v1.7-zip-large-file-profile.md
|
||||
│ ├── UPGRADE-v1.8-rar-support.md
|
||||
│ └── screenshots/ # README screenshot images
|
||||
├── THIRD_PARTY_NOTICES # per-library licence summary
|
||||
├── HANDOVER.md # current engineering handover
|
||||
├── LICENSE # GPLv3+
|
||||
└── README.md
|
||||
```
|
||||
|
||||
## Notes
|
||||
|
||||
- Copy, move, and delete run as single background tasks. While one task is running, other file operations are rejected.
|
||||
- Copy, move, delete, upload and download run as single background tasks. While one task is running, other file operations are rejected.
|
||||
- Delete is recursive and permanent. There is no recycle bin.
|
||||
- Copy/move tasks can be canceled. A partially copied single file is removed, but partially copied folders are left in place to avoid deleting existing files when merging into an existing target folder.
|
||||
- Copy/move tasks can be canceled. A partially copied single file is removed, but partially copied folders are left in place to avoid deleting pre-existing files when merging into an existing target folder.
|
||||
- Upload tasks can be canceled. A partially uploaded temporary file is removed when possible.
|
||||
- Downloading a folder or multiple selected items produces a tar stream. The tar archive is generated by the payload and is not written to PS5 storage first.
|
||||
- The UI can recover the active task display if the browser is closed and reopened while the payload process is still running.
|
||||
- Text editing is limited to the curated extension list above. Non-UTF-8 and oversized files are rejected.
|
||||
- File names are transmitted as UTF-8 through the web API. The payload also preserves legacy byte-oriented names returned by mounted filesystems so mixed USB filename encodings still display and operate correctly.
|
||||
|
||||
## FAQ
|
||||
|
||||
- **This is a homebrew app and should not intentionally modify system processes or kernel memory.** If you hit a kernel panic, make sure you are using a recent jailbreak method and ELF loader, or revert to the stable method you normally use.
|
||||
- **P2JB users** — if this payload triggers a kernel panic, avoid using it on that setup. Stability matters more than convenience when each retry is expensive.
|
||||
- **The preparing stage can take a while** when a folder contains many files — it sums folder size and checks free space, which helps avoid starting a copy / move / upload / download that cannot finish safely.
|
||||
- **`err_extract_entry_too_large`** — default archive caps are 512 GiB per
|
||||
entry / 500:1 ratio (covers a typical 3A-game archive with one ~300 GiB
|
||||
uncompressed file). If you exceed the default, confirm the large-file
|
||||
prompt (appears for archives > 480 GiB on disk), split the archive, or
|
||||
pass `large=1` directly to the API.
|
||||
- **`err_extract_unsupported`** — the archive is one this build cannot read:
|
||||
a file that is neither `.zip` nor `.rar` nor `.7z`, a ZIP entry using a
|
||||
compression method other than stored/deflated, a 7z folder with an
|
||||
unsupported coder, a split set whose naming is not recognised (a RAR set
|
||||
named `x.rar.001` must be renamed to `x.part1.rar`, `x.part2.rar`, …), or a
|
||||
RAR older than 1.4. Encrypted and multi-volume archives are **not** in this
|
||||
category — both are supported. The backend's own sentence is appended in
|
||||
parentheses and names the actual cause.
|
||||
- **`err_extract_dict_too_large`** — a RAR archive declares a compression
|
||||
dictionary larger than this build supports (4096 MiB) and unrar asked for
|
||||
permission to exceed it. The message states both the size the archive needs
|
||||
and the size the build allows. This is refused on purpose: the alternative is
|
||||
a single allocation of the entire dictionary window, which rarlab's own CLI
|
||||
rejects by default and which a 16 GB shared-memory console cannot sustain.
|
||||
Recompress the file on a PC with a dictionary of 4 GiB or less (`-md`), or
|
||||
extract it there. Note that the RAR5 format itself caps the field at 4 GiB,
|
||||
so this can only come from an archive written in the newer RAR7 header
|
||||
format.
|
||||
- **`err_extract_password`** — the archive is encrypted and the password was
|
||||
missing or wrong. That includes a 7z archive with an encrypted header
|
||||
(`-mhe=on`): the file names and entry sizes live inside the header, so
|
||||
nothing at all can be listed until the header decrypts. ZIP and RAR raise a
|
||||
password prompt on failure and retry the same request with what you type (up
|
||||
to three times; cancel or an empty box gives up); 7z asks before it starts,
|
||||
since an encrypted 7z header would otherwise cost a wasted scan.
|
||||
|
||||
## Credits
|
||||
|
||||
This project was built with reference to these projects:
|
||||
This project is a **fork of [owendswang/ps5-web-file-manager](https://github.com/owendswang/ps5-web-file-manager)** (GPL-3.0). The web UI,
|
||||
the task model and the PS5 packaging all originate there, and the upstream
|
||||
author's release under GPL-3.0 is what makes this derivative work possible.
|
||||
|
||||
- **[ps5-payload-dev/websrv](https://github.com/ps5-payload-dev/websrv):** HTTP server structure, static asset embedding ideas, and PS5 browser/websrv behavior. License: GPLv3+.
|
||||
- **[ps5-payload-dev/ftpsrv](https://github.com/ps5-payload-dev/ftpsrv):** PS5 payload conventions, home screen launcher/install flow reference, process handling style and startup installation reference. License: GPLv3+.
|
||||
- **[itsPLK/ps5-payload-manager](https://github.com/itsPLK/ps5-payload-manager):** Payload building behavior. License: GPLv3.
|
||||
- **[libmicrohttpd](https://ftp.gnu.org/gnu/libmicrohttpd/):** Used as the embedded HTTP server library. It is licensed by GNU under the LGPL; this payload links it as the SDK-provided static library.
|
||||
**Telling a fork build from an upstream one:** since v1.9.3 the version string
|
||||
carries an `M` suffix (`vX.Y.ZM`) — *M* for *Modified*. Upstream owendswang
|
||||
releases are plain `vX.Y.Z`. So `v1.9.2` is upstream/fork-shared numbering while
|
||||
`v1.9.3M` can only have come from this repository; the same letter appears in
|
||||
the ELF file name, the PS5 start-up notification, `/api/version` and the web UI
|
||||
footer. Releases before v1.9.3M predate the convention and keep their plain
|
||||
numbers.
|
||||
|
||||
Built with reference to these projects:
|
||||
|
||||
- **[ps5-payload-dev/websrv](https://github.com/ps5-payload-dev/websrv):** HTTP server structure, static asset embedding ideas, PS5 browser/websrv behaviour and PKG install function. License: GPLv3+.
|
||||
- **[ps5-payload-dev/ftpsrv](https://github.com/ps5-payload-dev/ftpsrv):** PS5 payload conventions, home-screen launcher/install flow reference, process handling style and startup installation reference. License: GPLv3+.
|
||||
- **[seregonwar/zftpd](https://github.com/seregonwar/zftpd):** PS5 TCP socket buffer tuning and high-throughput transfer behaviour reference. License: MIT.
|
||||
- **[itsPLK/ps5-payload-manager](https://github.com/itsPLK/ps5-payload-manager):** Payload building behaviour. License: GPLv3.
|
||||
- **[libmicrohttpd](https://ftp.gnu.org/gnu/libmicrohttpd/):** Used as the embedded HTTP server library. Licensed by GNU under the LGPL; this payload links it as the SDK-provided static library.
|
||||
- **[ps5-payload-dev/sdk](https://github.com/ps5-payload-dev/sdk):** Payload building foundation. License: GPLv3+.
|
||||
- **[etaHEN](https://github.com/etaHEN/etaHEN):** ShellUI URI navigation used to return to the PS5 home screen before exit. License: GPLv3.
|
||||
- **[ezremote](https://github.com/cy33hc/ps5-ezremote-client):** cited for the PKG-preview feature. License: **GPL-2.0-only** — its source files carry no "or later" notice, so it is **not** combinable with this GPL-3.0 codebase. **No code was taken from it:** `src/pkg_info.c` is an independent C99 implementation (it also reads the `.pkg` entry table and `param.json` fields, for which ezremote has no counterpart, and it uses the hand-written tokenizer in `src/json_util.c` rather than json-c). See `docs/REWRITE-FEASIBILITY.md` §2.2.
|
||||
- **[zlib-ng/minizip-ng](https://github.com/zlib-ng/minizip-ng):** ZIP reader used by the `/api/extract` endpoint. Vendored under `third_party/minizip-ng/`. License: zlib.
|
||||
- **[zlib](https://www.zlib.net/):** Compression backend for minizip-ng. Vendored under `third_party/zlib/`. License: zlib.
|
||||
- **[rarlab UnRAR](https://www.rarlab.com/rar_add.htm)** — RAR reader used by the `/api/extract` endpoint since v1.9. Vendored under `third_party/unrar7/` (version 7.20.1, the RARDLL source set). License: **UnRAR freeware license** — see `third_party/unrar7/license.txt`. Note this is a restricted licence rather than a FLOSS one: it permits using the source to handle RAR archives but forbids using it to build a RAR-compatible compressor.
|
||||
- **[opello/unrar](https://github.com/opello/unrar)** — the mirror the vendored rarlab sources were fetched from (commit `97e1780`).
|
||||
- **[DrMcCoy/dmc_unrar](https://github.com/DrMcCoy/dmc_unrar)** — RAR engine shipped in v1.8 only, superseded in v1.9 by rarlab UnRAR (it could not decode RAR5 "v6" archives or multi-volume sets). Removed from the tree; its licence was GPL-2.0-or-later.
|
||||
|
||||
## License
|
||||
|
||||
The project is distributed under GPLv3 or later, matching the GPLv3+ projects used as implementation references. See `LICENSE`.
|
||||
The project is distributed under **GPLv3 or later**, matching the GPLv3+ projects used as implementation references. See [`LICENSE`](./LICENSE).
|
||||
|
||||
Third-party projects retain their own licenses. Do not copy assets or source from the credited projects into another distribution without preserving the corresponding license notices.
|
||||
|
||||
If distributing binaries, comply with the LGPL terms for libmicrohttpd in addition to this project's GPL license.
|
||||
If distributing binaries, comply with the LGPL terms for `libmicrohttpd`
|
||||
in addition to this project's GPL license. The vendored `zlib` and
|
||||
`minizip-ng` sources are distributed under the zlib license; retain the
|
||||
copyright notices in `third_party/zlib/LICENSE` and
|
||||
`third_party/minizip-ng/LICENSE` when redistributing binaries built
|
||||
with this feature.
|
||||
|
||||
The vendored `third_party/unrar7/` sources (rarlab UnRAR — the RAR engine
|
||||
behind `src/rar_extract.c`) are **not** GPL: they ship under the UnRAR
|
||||
freeware license (see `third_party/unrar7/license.txt`), which forbids
|
||||
using them to develop a RAR-compatible compressor. Keep that notice and
|
||||
that restriction intact when redistributing. `THIRD_PARTY_NOTICES` carries
|
||||
the full per-library summary.
|
||||
|
||||
## Disclaimer
|
||||
|
||||
Unofficial homebrew software. Runs only on jailbroken PS5 consoles. Use at your own risk — the authors are not responsible for damage, data loss, account action or warranty impact. Do not redistribute Sony-proprietary content. Under GPLv3+, modified redistributions must publish their sources.
|
||||
@@ -0,0 +1,487 @@
|
||||
<div align="right">
|
||||
<a href="README.md">English</a> · <a href="README.zh-CN.md">简体中文</a>
|
||||
</div>
|
||||
|
||||
# PS5 网页文件管理器(PS5 Web File Manager)
|
||||
|
||||
> 面向已越狱 PS5 主机的自制 HTTP 文件管理器。通过同一局域网内的任意浏览器(包括 PS5 自带浏览器)即可浏览、编辑、上传、下载并解压 ZIP / RAR / 7z 压缩包——单个自包含 ELF 载荷,无外部服务、无遥测上报。
|
||||
|
||||
**版本:** v1.9.2 · **标题 ID:** `FMGR88888` · **许可证:** GPLv3+ · **目标平台:** `x86_64-sie-ps5`
|
||||
|
||||
---
|
||||
|
||||
## 概述
|
||||
|
||||
一个在已越狱 PS5 上运行的 HTTP 文件管理器载荷。从局域网内任意浏览器(含 PS5 浏览器本身)打开 `http://<PS5_IP>:8888/`,即可管理外接 USB 存储与用户分区的文件。设计初衷是安全地把游戏 dump 文件夹从 USB 拷贝到内置存储,但它同时也支持常规文件管理、原地文本编辑、PKG 预览/安装、图片预览,以及内置防 zip 炸弹保护的解压功能。
|
||||
|
||||
同一套源码树可构建出供开发用的 Linux 二进制,以及供部署的 PS5 载荷 ELF——见下方 `make linux`。
|
||||
|
||||
> **第一次用、不想看技术细节?** 直接看
|
||||
> [《新手使用说明》](docs/USER-GUIDE-zh-CN.md):怎么装、怎么传文件、怎么解压(含带密码与分卷的包)、
|
||||
> 界面上每句话是什么意思,以及与上游原版的差别——全部用大白话写。
|
||||
|
||||
## 未发布内容(下次发版将包含)
|
||||
|
||||
**加密归档现在可以端到端解压——ZIP(两种方案)、RAR,以及带头加密的 7z 都已打通。**
|
||||
|
||||
此前所有加密归档都会被提前拒绝,尽管密码输入框、失败提示与 `extract_password` 文案从 v1.9 起就已就位。真正的缺口在引擎侧,而不在 UI:
|
||||
|
||||
- **ZIP**:vendored 的 minizip-ng 在裁剪时把 crypto 后端一起裁掉了,于是(未被改动的)`mz_zip.c` 里那些 `-DHAVE_WZAES` / `-DHAVE_PKCRYPT` 分支没有实现可调。
|
||||
- **RAR**:rarlab UnRAR 本身能解密,但 `RARSetPassword` 从未被调用。
|
||||
- **7z**:`-mhe=on` 把文件名与 folder 表放进了加密头,归档连列出都做不到。
|
||||
|
||||
三者现在都已接线。密码缺失或错误会统一报为 `extract_password`(引擎层即 `ZIPX_ERR_PASSWORD`),也就是任务浮层已有的密码提示所响应的那个错误码。
|
||||
|
||||
### 变更
|
||||
|
||||
- **版本号加改版标记:`v1.9.3` → `v1.9.3M`。** 上游 owendswang 的发布版是纯 `vX.Y.Z`,因此这个 `M`(Modified,改版)就是「上游原版还是本仓改版」的判据。它是 `VERSION_TAG` 的一部分,所以 `/api/version`、PS5 启动通知、stdout 横幅、网页右下角**与 ELF 文件名**会一次性全部带上;网页右下角另加悬浮提示(`versionTooltip`,中英各一)解释这个字母的含义,免得没读过发行说明的人无从判断。产物名随之改变,也顺带让本仓产物再不可能与上游同版本号的资产同名相撞。
|
||||
|
||||
### 新增
|
||||
|
||||
- **加密 ZIP** —— 传统 PKWARE(「ZipCrypto」,即 `zip -e` 写出的格式)与 WinZip AES-128/192/256(压缩方法 `99` + `0x9901` 扩展字段,即 `7z -mem=AES256` 写出的格式),stored 与 deflated 条目均支持。
|
||||
- **加密 RAR** —— `-p` 内容加密与 `-hp` 头加密。`RARSetPassword` 现在在 `RAROpenArchiveEx` 之后、首次 `RARReadHeaderEx` 之前调用,这正是 unrar 解密 RAR5 头所需的顺序。
|
||||
- `/api/extract` 的 `password=` 现在对两个引擎都能真正走到解密路径。空值或缺失视为「无密码」,因此表单原值可以直接透传。
|
||||
- `third_party/minizip-ng/src/mz_crypt_wfm.c` —— 为裁剪后的 minizip-ng 提供的本地 crypto 后端:SHA-1、HMAC-SHA1、AES-128/192/256;S-box 与 GF(2^8) 表在首次使用时推导,因此二进制不新增任何 `.rodata` 查表。PBKDF2 复用 vendored 的 `mz_crypt.c`;随机数直接读 `/dev/urandom`(不再是 `mz_os_rand()`),从而把 `rand`/`srand` 排除在导入表之外。以下文件按上游 4.2.2 原样恢复:`mz_strm_wzaes.{c,h}`、`mz_strm_pkcrypt.{c,h}`。
|
||||
- **前端在密码失败后可直接重试**:`extract_password` 失败不再只是弹一个错误框,而是弹出密码输入框并按原参数(冲突策略、大文件选配)重新发起同一次解压,最多重试 3 次;取消或留空则回落到原有的失败提示。
|
||||
- **加密 7z 头(`-mhe=on`)现在可以打开。** 这是最后一个已知的格式缺口:`-mhe=on` 时文件名、folder 表**与每个条目的尺寸**全都在加密头里,因此 vendored SDK 在能列出任何条目之前就以 `SZ_ERROR_UNSUPPORTED` 退出。新模块 `src/sevenz_header.c` 读出该头部记录,用它自己的那一个 folder 走项目自研的 7zAES 路径(`src/sevenz_chain.c`)解码,然后交给 SDK 一个虚拟流——把加密头所在区域替换成明文。SDK 于是照常解析它一向解析的那个归档,磁盘上的文件完全不被改动;头部只是被*压缩*(`-mhc=on`,默认)的归档完全不受影响;密码错误则与其它加密归档一样返回 `extract_password`。
|
||||
- `tests/make-zip-enc-fixtures.bat`,以及 `tests/fixtures-real/` 下的三个真实 fixture(`enc-zipcrypto.zip`、`enc-aes256.zip`、`enc-aes256-store.zip`,密码 `secret123`)。
|
||||
|
||||
### 修复
|
||||
|
||||
- **编译选项变化现在会使目标文件失效。** `make` 察觉不到编译选项变化,因此加上 `-DHAVE_WZAES -DHAVE_PKCRYPT` 后旧的 `mz_zip.o` / `mz_crypt.o` 原样保留——又因为此时已没有任何代码引用新流,`--gc-sections` 会在链接「成功」的同时把加密代码再次丢掉(本次改动的第一次构建产物与已发布的 release 逐字节相同)。Makefile 现在把第三方编译选项集记录进 `ps5-obj/.third_party_cflags` / `linux-obj/.third_party_cflags`,只在真正变化时重编——这正是早年 `LzmaDec.o` 规则所规避的同一个陷阱,现已通用化。
|
||||
- `ZIPX_ERR_UNSUPPORTED` 不再涵盖加密,现在仅表示「多卷或不受支持的压缩方法」。
|
||||
- 顺带把 `tests/test_sevenz_extract.c` 里一处会导致截断告警的 `snprintf` 缓冲区调足。
|
||||
|
||||
### 测试
|
||||
|
||||
- `tests/test_zip_extract.c` 对每个加密 fixture 跑四种情况(无密码 → `PASSWORD`、空密码 → `PASSWORD`、错密码 → `PASSWORD`、正确密码 → `ZIPX_OK` 并逐字节校验内容),另加一项证明「提供密码后限额依旧生效」。
|
||||
- `tests/test_rar_extract.c` 对 `enc-v6.rar` 做同样的四种情况验证,包括失败路径绝不发布任何文件。
|
||||
- `tests/test_sevenz_extract.c` 对 `aeshe.7z` 跑三种情况:无密码 → `ZIPX_ERR_PASSWORD`、错密码 → `ZIPX_ERR_PASSWORD`、正确密码 → 成功且逐字节一致,并证明失败后不留下 staging 目录。
|
||||
- 前端重试流程有一份无头检查(`.build/ui_retry_test.mjs`,把 `assets/main.js` 载入桩 DOM):**40 项检查**,覆盖参数记忆、重试上限、取消与空密码的回落,以及「非 ASCII 目录下必须仍然弹出密码框」的回归用例。另有一份 `.build/ui_upload_menu_test.mjs`(**40 项检查**)盯标记侧:i18n 键在两份语言文件里都存在、上传菜单接对了回调、用到的 class 确实有样式、菜单行高亮规则必须带面板作用域(否则会输给通用按钮规则而静默失效),以及**解压按钮不许被隐藏、只许被置灰**(顺带扫 `main.js` 里 117 个 `t("...")` 键是否双语齐全)。
|
||||
- 主机端合计:**140 ZIP + 37 RAR = 177 项检查**,0 失败。
|
||||
- 请求的字典超过本构建支持上限的 RAR 归档不再被误报成「单条目过大」:它有独立的 `extract_dict_too_large` 编码,报错文案同时给出归档需要的字典与构建支持的上限。构建行为**未变** —— 这类归档仍然被拒,因为放行意味着一次性分配整个字典窗口,而这正是 rarlab 自家 CLI 默认拒绝、16 GB 共享内存的主机也承受不了的。
|
||||
- 7z 套件:**27 项用例,0 失败**(`tests/run-sevenz-tests.sh`),且原先登记 `aeshe` 的 `KNOWN_GAPS` 列表现已**清空**——加密头 fixture 同时通过 folder 解码器与解压 façade 两条路径。
|
||||
- 当前源码树构建产物 **903 448 B**,sha256 `8ca47d5aaca75085b32641300cce30fadb7df7749cb6b53d04f129bcecc286b7`,`e_machine` 为 `0x003e`;同一源码树构建两次逐字节一致。产物内已确认包含新的前端代码(前端资源是 gzip 内嵌的,需先解压才能在二进制里检索到)。段数仍为 20,**动态符号零新增**。加密归档那批改动净增 5 712 字节正文(`.text` +4 880、`.rodata` +640、`.eh_frame*` +192);加 `M` 标记再让 `.rodata` 涨 0x100(256 B),修正 `err_extract_unsupported` 文案再涨 0x40(64 B),上传菜单再涨 0x980(2 432 B),第一轮修复的文案与 CSS 再涨 0x100(256 B),菜单行高亮的收敛规则再涨 0x180(384 B),解压按钮常显(去掉 `hidden`、换短标签、三条禁用理由、`.extract-action:disabled`)再涨 0x240(576 B),**其余段尺寸一个都没变**,因此文件总尺寸仍是 903 448 B。**尺寸没变不等于内容没变** —— 判断只看 `readelf -SW` 的段尺寸。
|
||||
|
||||
### 仍未完成
|
||||
|
||||
- 在真机上做端到端验证。
|
||||
|
||||
> 注意:本节描述的是**未发布**状态。已发布版本号仍为 `v1.9.2`,其二进制**不含**上述加密支持;当前未发布的源码树自报版本为 `v1.9.3M`。
|
||||
|
||||
## v1.9.2 与 v1.9.1 新增内容
|
||||
|
||||
> **v1.9.2 与 v1.9.1 的功能完全相同,只换了内嵌版本号。** 原因是原先的 `v1.9.1` tag 指在产出发布二进制的提交**之前 4 个提交**,tag 与产物对不上(clone 该 tag 无法重建出发布的那份 ELF);v1.9.2 重新从产出该二进制的提交上打,使 tag = 源码 = 二进制。
|
||||
|
||||
- **7z 解压引擎**(`src/sevenz_extract.{c,h}`):自研解码子集 + 拉式 codec 链(`src/sevenz_chain.c`,覆盖 LZMA2 / BCJ2 等),由 `src/extract.c` 按扩展名分派,与 ZIP / RAR 共用同一套三阶段模型与限额档位。`.7z` 文件在文件列表中同样带「解压」按钮。
|
||||
- **7zAES 内容解密**(AES-256-CBC):引擎可解密带密码的 7z 内容,密码经 `/api/extract` 的 `password=` 传入;解压 7z 时前端会提前询问密码。ZIP / RAR 的加密当时尚未打通(引擎侧缺口,见顶部「未发布内容」),v1.9.1 时对它们仍会报 `extract_unsupported`。
|
||||
- **7z 分卷**:`.7z.001` / `.z01` 等链式分卷由 `src/sevenz_volstream.c` 按名拼接,打开首个分卷即可。
|
||||
- **性能三项**(纯解码提速,不影响功能面):
|
||||
- SDK 汇编 LZMA 解码器(`Asm/x86/LzmaDecOpt.asm` + jwasm,无 jwasm 自动回退纯 C)≈ 1.26×。
|
||||
- 纯 LZMA2 文件夹多线程解码(`src/sevenz_mt.c` + `Lzma2DecMt`,8 线程)≈ 1.37×。
|
||||
- 移除 ZIP 逐条目 fsync,减少 staging 重命名前的写盘开销。
|
||||
- **当时唯一缺口**:7z `-mhe=on` 加密头(独立单元,读取需自研头解析器),其余 7z 特性均已支持。已在顶部「未发布内容」中补上。
|
||||
|
||||
## v1.9 新增内容
|
||||
|
||||
- **RAR 引擎替换为官方 rarlab UnRAR 7.20.1**(`third_party/unrar7/`,取代 dmc_unrar)。这正是让 RAR 解压在真实文件上可用的一步:dmc_unrar 无法解码 **WinRAR 6.x/7.x** 写出的归档(RAR5「v6」压缩),也不支持多卷;两者现在都能工作。
|
||||
- **RAR5「v6」归档可解压**(v1.8 时代在 WinRAR 6/7 文件上报「归档损坏」的问题已消失)。
|
||||
- **多卷 RAR**(`.part01.rar` 链):当完整卷集与被打开的卷放在同一目录时,unrar 按文件名拼接各部分。
|
||||
- 引擎可解密加密 RAR(`RARSetPassword`)——发 v1.9 时密码 UI / API 接线尚未完成,加密归档会被拒绝;该接线已在顶部「未发布内容」中补齐。
|
||||
- 主机测试现用真实归档(v6 / 加密 / 3 卷 fixture,提交于 `tests/fixtures-real/`):**70 ZIP + 24 RAR = 94 项检查**。
|
||||
|
||||
## v1.8 新增内容
|
||||
|
||||
- **单卷 RAR 解压**,基于内置的 FLOSS 库 [`dmc_unrar`](https://github.com/DrMcCoy/dmc_unrar)(GPL-2.0-or-later)。支持 RAR 1.5、2.x、3.x、4.x、5.x 归档。`.rar` 文件出现在文件列表中且「解压」按钮可用;`.part02+.rar` 子卷上的按钮置灰,提示「请选择主卷」——v1.8 无法拼接多卷 RAR(见下方 [RAR 解压](#rar-解压) 章节)。
|
||||
- 新引擎 `src/rar_extract.c` 与既有 `src/zip_extract.c` 之间**共享解压协议**:相同的 `zipx_status_t` 状态码、相同的 `zipx_limits_t` 档位(默认 / `large=1`)、相同的三阶段模型(`scan → extract → publish → cleanup`)、相同的 staging 目录布局、相同的冲突策略、相同的错误映射到任务 UI。`src/extract.c` 中的分派器只是一个微小的 `ends_with_ci(…)` 判断。
|
||||
- **14 个新增主机端 C 测试**(`tests/test_rar_extract.c`)接入现有 `tests/run-tests.sh`。覆盖:格式分派、每个影响 RAR 用户的 `DMC_UNRAR_*` 错误码翻译、限额档位交接。主机检查总数:**69 ZIP + 14 RAR = 83**。
|
||||
- **文档**:[`CHANGELOG.md`](./CHANGELOG.md)、[`docs/UPGRADE-v1.8-rar-support.md`](./docs/UPGRADE-v1.8-rar-support.md),以及 `third_party/unrar7/VENDORED.md` 中的 vendoring 决策树(v1.8 时为 `third_party/unrar/VENDORED.md`)。
|
||||
|
||||
## v1.8.1 新增内容
|
||||
|
||||
- **放宽默认 ZIP 限额**(配合 v1.7 的大档案档位)。v1.7 默认单条目上限为 64 GiB,对典型 PS5 系统备份 ZIP(200–300 GiB)过于激进。v1.8.1 将默认档位提高到 **总量 1 TiB / 单条目 256 GiB / 500:1 比率**,保留 `large=1` 选项为 2 TiB / 1 TiB / 1000:1。前端阈值从 60 GiB 提升到 240 GiB,使常见系统备份归档不再触发确认提示。
|
||||
- RAR 解压继承这些新默认值(`rar_extract.c` 直接从引擎透传 `c->limits`,无需改引擎)。
|
||||
- 理由:真正的防 zip 炸弹防线是 `check_space()`(staging 前基于 statvfs 的真实磁盘空间检查)+ `max_ratio`(声明的压缩比上限)。尺寸上限只是 UX 护栏,而非安全边界。
|
||||
|
||||
## v1.8.2 新增内容
|
||||
|
||||
- **再次放宽默认 ZIP 限额**,针对 3A 游戏单文件场景。v1.8.1 仍会静默拒绝归档内单个约 300 GiB 的未压缩文件(默认扫描在请求到达前端确认提示之前就返回 `ZIPX_ERR_LIMIT_FILE_SIZE`)。v1.8.2 将默认档位提高到 **总量 2 TiB / 单条目 512 GiB / 500:1 比率**,`large=1` 选件提到 4 TiB / 1 TiB / 1000:1。前端阈值从 240 GiB 提升到 480 GiB。
|
||||
- **两个 PS5 专属构建修复**,在交叉编译 PS5 目标时发现。主机端测试套件(`tests/run-tests.sh`)曾静默接受二者,因为它链接相同源码但使用 gcc 而非 clang 18,且包含路径不同:
|
||||
- `Makefile` CFLAGS:加入 `-Ithird_party/unrar`,使 `src/rar_extract.c` 能找到项目自有的 `dmc_unrar_api.h` 门面头文件。
|
||||
- `src/extract.c`:把 `extract_progress()` 定义移到 `extract_dispatch()` 之前,避免被 `-Werror=implicit-function-declaration` 标记(PS5 SDK 的 clang 18 比测试用的主机 gcc 更严格)。
|
||||
- v1.8.2 发布产物:`web-file-mgr.elf` —— 509 704 字节,sha256 `1b2c3d68b35e32737105f17d14a80a3c159ceca0cabd274ee168cbcd81906f65`,ELF 64 位小端,e_machine `0x003e`(x86_64-sie-ps5)。
|
||||
- 测试:**84 项主机端检查**(70 ZIP + 14 RAR),0 失败。PS5 交叉编译端到端成功。
|
||||
|
||||
## v1.7 新增内容
|
||||
|
||||
- **ZIP 大文件档位**(通过在 `/api/extract` 传入新的 `large=1` 参数选配启用):放宽的限额为 **总量 2 TiB** / **单条目 1 TiB** / **1000:1 压缩比**。当磁盘上归档大于 **60 GiB** 时前端会提示确认;仅当用户明确同意时服务器才启用该档位。
|
||||
- **更严格的默认 ZIP 档位**保持安全:**总量 1 TiB** / **单条目 256 GiB** / **500:1 比率**。一个 4 MiB 压缩包解压到 800 GiB 仍会在打开任何输出文件之前被拒绝。
|
||||
- **69 项主机端 C 测试**(`tests/run-tests.sh`)现已覆盖路径穿越、ZIP64、加密拒绝、压缩比、冲突策略与新增大文件档位(`tests/test_zip_extract.c`)。
|
||||
- 更早的细化——见 v1.6 以来的 `git log`。
|
||||
|
||||
## 截图
|
||||
|
||||
<p>
|
||||
<a href="docs/screenshots/20260617_231827.376.jpg" target="_blank"><img src="docs/screenshots/20260617_231827.376.jpg" width="31%" alt="PS5 网页文件管理器截图 1"></a>
|
||||
<a href="docs/screenshots/20260619_131432.399.jpg" target="_blank"><img src="docs/screenshots/20260619_131432.399.jpg" width="31%" alt="PS5 网页文件管理器截图 2"></a>
|
||||
<a href="docs/screenshots/20260617_232348.855.jpg" target="_blank"><img src="docs/screenshots/20260617_232348.855.jpg" width="31%" alt="PS5 网页文件管理器截图 3"></a>
|
||||
<a href="docs/screenshots/20260619_131811.644.jpg" target="_blank"><img src="docs/screenshots/20260619_131811.644.jpg" width="31%" alt="PS5 网页文件管理器截图 4"></a>
|
||||
<a href="docs/screenshots/20260619_131535.239.jpg" target="_blank"><img src="docs/screenshots/20260619_131535.239.jpg" width="31%" alt="PS5 网页文件管理器截图 5"></a>
|
||||
<a href="docs/screenshots/20260620_232728.533.jpg" target="_blank"><img src="docs/screenshots/20260620_232728.533.jpg" width="31%" alt="PS5 网页文件管理器截图 6"></a>
|
||||
</p>
|
||||
|
||||
## 功能
|
||||
|
||||
- **浏览** —— 列出文件与文件夹;按名称、类型、大小、修改时间或权限排序。上次排序方式持久化在 `localStorage`。
|
||||
- **权限** —— 用复选框切换读/写/执行,或粘贴经过校验的四位八进制模式。
|
||||
- **操作** —— 复制、移动、删除(递归、无回收站)、重命名、创建文件与文件夹。
|
||||
- **编辑器** —— 针对 ≤ 1 MiB 的文件,跨精选扩展名列表的原地 UTF-8 文本编辑器:`.txt .json .xml .ini .cfg .conf .md .log .lua .js .css .html .htm .c .h .cpp .hpp .sh .csv .yaml .yml .shn`。
|
||||
- **多选** —— 一次性复制、移动、删除或打包下载多个项目。
|
||||
- **上传** —— 从局域网内任意设备上传单文件或文件夹树(在 PS5 浏览器中隐藏)。原子化的临时文件 + 重命名。
|
||||
- **下载** —— 单文件以原始字节下载,或文件夹/多选以流式 `.tar` 下载。在 PS5 浏览器中隐藏。
|
||||
- **任务** —— 全屏覆盖层,带延迟显示、实时进度、吞吐率、ETA、取消,以及浏览器中途关闭重开后的恢复能力。
|
||||
- **归档解压** —— ZIP、RAR、7z 三种引擎,均带防 zip 炸弹 / 路径穿越 / 压缩比保护。ZIP 覆盖 stored / deflated / ZIP64 以及**加密**条目(ZipCrypto 与 WinZip AES-128/192/256);RAR 覆盖 RAR4 + RAR5(含 WinRAR 6/7「v6」)、多卷,以及 `-p` / `-hp` 加密;7z 覆盖 LZMA / LZMA2 / PPMd、Delta 与 BCJ2、`.7z.001` 分卷、7zAES 与 `-mhe=on` 加密头。详见下方 [ZIP 解压](#zip-解压)、[RAR 解压](#rar-解压)、[7z 解压](#7z-解压)。
|
||||
- **加密归档重试** —— 解压遇到加密归档时,会弹出密码输入框并按原参数自动重试(最多 3 次);也可以在解压 7z 时提前输入密码以免白跑一次扫描。
|
||||
- **PKG** —— 安装并预览 `.pkg` 文件。
|
||||
- **图片** —— 预览 `.png .jpg .jpeg .gif .bmp .webp`。
|
||||
- **本地化** —— 英文 + 简体中文,根据 `navigator.languages` 自动选择。
|
||||
- **移动端友好** —— 响应式布局,工具栏自动换行,文件列表可横向滚动。
|
||||
|
||||
## 快速上手
|
||||
|
||||
1. **构建** ELF:
|
||||
|
||||
```sh
|
||||
export PS5_PAYLOAD_SDK=/opt/ps5-payload-sdk # 见「构建」章节的 SDK 配置
|
||||
make
|
||||
```
|
||||
2. **发送** 载荷到 PS5(默认 ELF 加载器端口 `9021`):
|
||||
|
||||
```sh
|
||||
nc -q0 "$PS5_HOST" 9021 < web-file-mgr.elf
|
||||
```
|
||||
3. **读取** PS5 屏幕上的通知——它会打印实际监听端口(默认 `8888`)。
|
||||
4. 在**同一局域网**内的任意浏览器中打开 `http://<PS5_IP>:<port>/`——PS5 浏览器也可以。
|
||||
5. 首次运行时,载荷还会写入一个 **Media** 分类的主屏启动器;已有的启动器文件不会被覆盖。
|
||||
|
||||
## 构建
|
||||
|
||||
需要 [ps5-payload-dev/sdk](https://github.com/ps5-payload-dev/sdk#quick-start):
|
||||
|
||||
```sh
|
||||
export PS5_PAYLOAD_SDK=/opt/ps5-payload-sdk
|
||||
```
|
||||
|
||||
本项目链接 `libmicrohttpd`。`make` 在构建前会检查它,缺失时自动运行安装器:
|
||||
|
||||
```sh
|
||||
make
|
||||
```
|
||||
|
||||
若构建主机无网络访问,可提前放入 libmicrohttpd 源码包并手动运行安装器:
|
||||
|
||||
```sh
|
||||
LIBMICROHTTPD_TARBALL=/path/to/libmicrohttpd-1.0.1.tar.gz \
|
||||
./install-libmicrohttpd.sh
|
||||
make
|
||||
```
|
||||
|
||||
输出:
|
||||
|
||||
```text
|
||||
web-file-mgr.elf (约数百 KiB,v1.9.1 含 unrar7 + 7z 后更大;x86_64-sie-ps5)
|
||||
```
|
||||
|
||||
若只想做纯 UI/JS 开发而不需要 PS5 工具链:
|
||||
|
||||
```sh
|
||||
make linux
|
||||
./web-file-mgr-linux
|
||||
```
|
||||
|
||||
Linux 构建**不包含** PS5 主屏启动器安装器。
|
||||
|
||||
## 使用
|
||||
|
||||
在 PS5 上启动一个 ELF 加载器(端口 `9021` 常见)。发送载荷:
|
||||
|
||||
```sh
|
||||
export PS5_HOST=ps5_ip_address
|
||||
nc -q0 "$PS5_HOST" 9021 < web-file-mgr.elf
|
||||
```
|
||||
|
||||
载荷启动后,PS5 通知会显示应用名、版本与实际监听端口。打开它打印的 URL,例如:
|
||||
|
||||
```text
|
||||
http://${PS5_IP_ADDRESS}:8888/
|
||||
```
|
||||
|
||||
若载荷不得不回退到其它端口(如 `8889`),请以通知显示的端口为准——URL 并未硬编码。
|
||||
|
||||
首次启动时,载荷会在需要时于 Media 分类安装一个 `PS5 Web File Manager` 快捷方式。已有的启动器文件会被保留;只补写缺失的文件。
|
||||
|
||||
## ZIP 解压
|
||||
|
||||
支持普通 ZIP 与加密 ZIP——stored / deflated / ZIP64,传统 PKWARE(ZipCrypto)与 WinZip AES-128/192/256 两种加密方案。引擎是一个独立的三阶段模块(`scan → extract → publish → cleanup`),位于 `src/zip_extract.{c,h}`,配有独立的主机端 C 测试套件。每个条目先写入 staging 目录(`*.wfm-part-*`),再原子重命名到目标位置。解压路径里**刻意不做逐条目 `fsync`**——整条流水线是「不 sync、只 rename」,因为 publish 只是 rename、也没有续解功能需要保护(8000 文件档实测 ≥14×,见 `docs/EXTRACTION-PERF.md`)。归档中途任何失败都会回滚部分改动;取消与致命错误总会清理 staging。
|
||||
|
||||
### 限额
|
||||
|
||||
| 限额 | 默认档位 | 大档案档位(`ZIPX_LIMITS_LARGE`) |
|
||||
|---|---|---|
|
||||
| `max_entries` | 200 000 | 500 000 |
|
||||
| `max_total_bytes`(未压缩) | 2 TiB | 4 TiB |
|
||||
| `max_file_bytes`(单条目) | 512 GiB | 1 TiB |
|
||||
| `max_ratio`(未压缩 / 压缩) | 500 : 1 | 1000 : 1 |
|
||||
| `max_depth`(文件夹嵌套) | 32 | 32 |
|
||||
| `max_name_len` / `max_path_len` | 255 / 1024 | 255 / 1024 |
|
||||
|
||||
**默认档位**出厂即安全:一个解压到 800 GiB 的 4 MiB 压缩块会在打开任何输出文件之前被拒绝。**大档案档位**仅在请求携带 `large=1` 时才启用——当磁盘上归档大于 `LARGE_FILE_THRESHOLD_BYTES`(默认 480 GiB;可在 `assets/main.js` 配置)时,解压对话框会自动提示用户。确认提示即为用户的明确选配;服务器自身不会额外记录任何内容。
|
||||
|
||||
### 安全检查
|
||||
|
||||
引擎拒绝解压以下归档:
|
||||
|
||||
- 路径穿越(`..` 段、绝对 POSIX 路径、Windows 盘符)。
|
||||
- 符号链接、设备、FIFO、套接字(`ZIPX_ERR_SPECIAL`)。
|
||||
- 同一归档内的重复条目或目录/文件名冲突。
|
||||
- 解压后尺寸、条目数、嵌套深度、名称长度或压缩比突破当前档位。
|
||||
|
||||
加密条目**不再**属于拒绝项:密码通过 `/api/extract` 的 `password=` 传入,缺失或错误时返回 `zipx` 层的 `ZIPX_ERR_PASSWORD`(前端对应 `err_extract_password`),由界面提示后重试。scan 阶段对加密条目同样生效——限额不会因为提供了密码而被跳过。
|
||||
|
||||
### 冲突策略
|
||||
|
||||
通过 `/api/extract` 上的 `conflict=` 传入:
|
||||
|
||||
- `fail`(默认)—— 拒绝覆盖任何已存在的目标。
|
||||
- `overwrite` —— 替换已存在文件;合并进已存在文件夹。
|
||||
- `merge` —— 保留已存在文件,新增其余文件。
|
||||
|
||||
### 调整阈值
|
||||
|
||||
480 GiB 的前端阈值位于 `assets/main.js`:
|
||||
|
||||
```js
|
||||
const LARGE_FILE_THRESHOLD_BYTES = 480 * 1024 * 1024 * 1024;
|
||||
```
|
||||
|
||||
设为 `Infinity` 可静音提示,调低则更保守,或干脆删掉该调用——无论阈值如何,服务器始终遵循 `large=1`。
|
||||
|
||||
## RAR 解压
|
||||
|
||||
RAR 解压引擎(`src/rar_extract.{c,h}`)由 **官方 rarlab UnRAR 源码** 支撑(`third_party/unrar7/`,版本 7.20.1,编译为静态库并通过其 C 兼容的 DLL API 驱动)。扩展名为 `.rar` 的文件与 `.zip` 文件一样拥有**解压**按钮;引擎由 `src/extract.c` 按扩展名分派。
|
||||
|
||||
> v1.9 替换了 v1.8 的引擎(dmc_unrar 1.7.0)。dmc_unrar 无法解码 WinRAR 6.x/7.x 写出的归档(RAR5「v6」压缩)且不支持多卷;unrar 原生支持两者。
|
||||
|
||||
### 支持范围
|
||||
|
||||
| 格式 | 支持 | 备注 |
|
||||
|---|---|---|
|
||||
| RAR 1.5 → 4.x(含 2.9 / 3.6 / 4.0) | ✅ | |
|
||||
| RAR 5.0 及 **5.0「v6」**(WinRAR 6.x / 7.x) | ✅ | v1.9 的触发点 |
|
||||
| Solid 块、最大 1 GiB 字典 | ✅ | |
|
||||
| PPMd 解压(RAR 3.0+) | ✅ | |
|
||||
| **多卷**(`.part01.rar` + `.part02.rar` + …) | ✅ | 当完整卷集与被打开的卷同处一目录时,unrar 按名拼接。选择首个卷(`name.part1.rar` / `name.part01.rar`);非首卷在 UI 中仍置灰并给出提示。 |
|
||||
| **加密 RAR** | ✅ | `-p` 内容加密与 `-hp` 头加密均可。密码经 `/api/extract` 的 `password=` 传入引擎(`RARSetPassword` 在 `RAROpenArchiveEx` 之后、首次 `RARReadHeaderEx` 之前调用);缺失或错误返回 `ZIPX_ERR_PASSWORD`,界面提示后重试。 |
|
||||
| 符号链接 / FIFO / 套接字 / 设备 | ❌ | 以 `ZIPX_ERR_SPECIAL` 拒绝(与 ZIP 行为一致) |
|
||||
| RAR 1.3(1.4 之前) | ❌ | 被 unrar 上游拒绝 |
|
||||
|
||||
当某归档被拒绝时,用户会收到 `extract_unsupported` 失败,文件名作为详情参数。前端已用典型的双语重试指引显示该错误。
|
||||
|
||||
### 限额
|
||||
|
||||
RAR 引擎原样复用 ZIP 的限额表——其上并无额外的 RAR 档位表。默认值与 `large=1` 选配完全相同:
|
||||
|
||||
| 限额 | 默认档位 | 大档案档位(`large=1`) |
|
||||
|---|---|---|
|
||||
| `max_entries` | 200 000 | 500 000 |
|
||||
| `max_total_bytes`(未压缩) | 2 TiB | 4 TiB |
|
||||
| `max_file_bytes`(单条目) | 512 GiB | 1 TiB |
|
||||
| `max_ratio`(未压缩 / 压缩) | 500 : 1 | 1000 : 1 |
|
||||
| `max_depth`(文件夹嵌套) | 32 | 32 |
|
||||
| `max_name_len` / `max_path_len` | 255 / 1024 | 255 / 1024 |
|
||||
|
||||
大档案档位的 RAR 解压使用与 ZIP 相同的 `LARGE_FILE_THRESHOLD_BYTES`(480 GiB)提示——前端对 `.rar` 与 `.zip` 的提示处理相同,且服务器仅在请求携带 `large=1`(选配)时才启用大限额。
|
||||
|
||||
### 安全检查
|
||||
|
||||
RAR 引擎应用与 ZIP 引擎相同的检查——复用 `zipx_status_t` 状态码,因此任务 UI 的 `err_extract_unsafe_name`、`err_extract_too_deep`、`err_extract_ratio` 等会一致触发:
|
||||
|
||||
- 路径穿越(`..` 段、绝对 POSIX 路径、Windows 盘符、`\` 在 `Rar!\x1a\x07…` 头之后被视为路径分隔符等)。
|
||||
- 符号链接、FIFO、套接字、设备。
|
||||
- 归档内重复条目或目录/文件名冲突。
|
||||
- 归档尺寸、条目数、深度、名称长度或压缩比突破当前档位。
|
||||
|
||||
### Vendoring 与许可
|
||||
|
||||
`third_party/unrar7/` 是官方 **rarlab UnRAR 源码**(7.20.1)的逐字副本,由 [`opello/unrar`](https://github.com/opello/unrar) 在提交 `97e1780` 处镜像。它依 **UnRAR 免费软件许可** 分发(见 `third_party/unrar7/license.txt`):可于任何软件中用于处理 RAR 归档,但不得用于开发 RAR 兼容的*归档器*或重新实现 RAR 压缩算法。项目自有的门面 `third_party/unrar7/unrar_c_api.h` 携带项目自身许可。
|
||||
|
||||
> v1.8 引擎 `third_party/unrar/dmc_unrar.c`(DrMcCoy/dmc_unrar 1.7.0,GPL-2.0-or-later)已在 v1.9 移除;其声明留存于 git 历史。
|
||||
|
||||
### 加密 RAR
|
||||
|
||||
密码通道现已完整接通:`/api/extract` 的 `password=` 会被透传给引擎,并在 `RAROpenArchiveEx` 之后、首次 `RARReadHeaderEx` 之前通过 `RARSetPassword` 交给 unrar(这个顺序是解密 `-hp` 加密头的前提)。密码缺失或错误一律返回 `ZIPX_ERR_PASSWORD`(前端 `err_extract_password`),界面据此弹出密码框并按原参数重试,最多 3 次。
|
||||
|
||||
## 7z 解压
|
||||
|
||||
7z 解压引擎(`src/sevenz_extract.{c,h}`)基于 SDK 解码子集(LZMA2 / LZMA / BCJ2 等)加上项目自研的拉式 codec 链(`src/sevenz_chain.c`,位于 `src/sevenz_chain.h`)。扩展名为 `.7z` 的文件与 ZIP / RAR 一样拥有**解压**按钮;引擎由 `src/extract.c` 按扩展名分派,并复用同一套三阶段模型、限额档位与冲突策略。
|
||||
|
||||
> v1.9.1 新增。SDK 自带的 `SzArEx` 路径仅覆盖 4 个 coder 的文件夹,不足以装下 BCJ2 的 5 coder;本项目改为自研 folder 解析 + 拉式 codec 链,从而原生支持 BCJ2 与多 coder 组合。
|
||||
|
||||
关于头部:7-Zip 把归档头放在文件末尾,并在头部变大时把它压缩(`-mhc=on`,默认行为),所以头部区域通常以一条 `k7zIdEncodedHeader` 记录开头、描述一个 folder。`-mhe=on` 时那个 folder **也被加密**,而它装着文件名、folder 表与每个条目的尺寸,于是 vendored SDK 在能列出任何条目之前就对整个归档放弃。`src/sevenz_header.c` 负责这种情况:读出该记录,用与内容完全相同的那条 7zAES 路径解出它的 folder,再把一个「头部区域是明文」的虚拟流交给 SDK——磁盘上的归档从不被写入,仅仅被*压缩*过头的归档也完全不受影响。
|
||||
|
||||
### 支持范围
|
||||
|
||||
| 格式 | 支持 | 备注 |
|
||||
|---|---|---|
|
||||
| LZMA2 / LZMA(含 ZIP64 式大尺寸) | ✅ | 单 coder 纯 LZMA2 走多线程解码(`src/sevenz_mt.c`,8 线程) |
|
||||
| BCJ2(x86 反汇编后处理) | ✅ | 经自研拉式链;SDK `SzArEx` 装不下 5 coder 时由本项目承载 |
|
||||
| 多 coder 组合文件夹 | ✅ | 自研 `sevenz_chain.c` 解析 |
|
||||
| **分卷**(`.7z.001` / `.z01` 链) | ✅ | `src/sevenz_volstream.c` 按名拼接;打开首个分卷 |
|
||||
| **内容加密**(7zAES,AES-256-CBC) | ✅ | 引擎可解密;解压 7z 时前端会**提前**询问密码(避免为无密码归档白跑一次扫描 + folder 解析),密码经 `password=` 传给引擎,缺失或错误返回 `ZIPX_ERR_PASSWORD` 并可重试 |
|
||||
| **`-mhe=on` 加密头** | ✅ | `src/sevenz_header.c` 自行解码该头部记录(复用同一条 7zAES 路径),再把一个携带明文的虚拟流交给 SDK;密码错误返回 `ZIPX_ERR_PASSWORD`,与其它加密归档一样由提示重试 |
|
||||
| `-mhc=off`(未压缩头) | ✅ | 明文头一向可读;现在按字节探测后完全不做干预 |
|
||||
|
||||
当某归档被拒绝时,用户同样收到 `extract_unsupported` 失败,UI 显示双语重试指引。
|
||||
|
||||
### 限额
|
||||
|
||||
7z 引擎复用与 ZIP / RAR 完全相同的限额表;默认档位与 `large=1` 选配一致(见 [ZIP 解压 → 限额](#限额))。
|
||||
|
||||
### 安全检查
|
||||
|
||||
7z 引擎复用相同的 `zipx_status_t` 错误码与检查集合:路径穿越、特殊文件、重复条目/名冲突、以及突破当前档位的尺寸/条目数/深度/名称长度/压缩比。coder 的 `out_size` 取自 `coder_unpack_sizes[index]`(而非文件夹尺寸),`SzArEx` 失败时重置 `blockIndex` 以避免伪 CRC。
|
||||
|
||||
## 校验
|
||||
|
||||
`make` 之后,对生成的 ELF 做健全性检查:
|
||||
|
||||
```sh
|
||||
ls -la web-file-mgr.elf # v1.9.1 因含 unrar7 + 7z 体积更大;v1.8.3 约 509 KiB
|
||||
sha256sum web-file-mgr.elf # 把摘要记录进你的发布说明
|
||||
file web-file-mgr.elf # 期望 "ELF 64-bit LSB pie executable, x86-64"
|
||||
od -An -tx1 -N20 web-file-mgr.elf | head -2 # 魔数 7f45 4c46 0201 + e_machine 003e
|
||||
```
|
||||
|
||||
`e_machine = 0x003e` 确认了 PS5 目标三元组 `x86_64-sie-ps5`。`e_type = 3`(`ET_DYN`)确认了 ELF 加载器期望的位置无关载荷。
|
||||
|
||||
## 测试
|
||||
|
||||
一套 POSIX / 主机端 C 测试套件覆盖 ZIP、RAR 与 7z 三个引擎,可在任意 Linux / macOS / MSYS shell 下、无需 PS5 SDK 运行:
|
||||
|
||||
```sh
|
||||
cd tests && bash run-tests.sh # ZIP + RAR 套件
|
||||
bash run-sevenz-tests.sh # 7z 套件(需 MinGW gcc 与 7-Zip 二进制)
|
||||
```
|
||||
|
||||
输出为逐用例的 `check` 风格报告。当前 `main` 上为 **177 项检查**(140 ZIP + 37 RAR),0 失败;7z 套件另有 **27 项用例**,同样 0 失败。覆盖:
|
||||
|
||||
- ZIP 条目解析(stored + deflated + ZIP64)
|
||||
- 路径穿越、绝对路径、反斜杠、Windows 盘符
|
||||
- 符号链接、FIFO、坏 CRC、截断归档、非 ZIP 文件
|
||||
- 限额:`entries`、`total_bytes`、`file_bytes`、`ratio`、`depth`、`name_len`
|
||||
- 冲突策略:`fail` / `overwrite` / `merge`
|
||||
- 每个阶段的取消
|
||||
- **加密档案** —— `tests/fixtures-real/` 下的真实归档各跑四种情况(无密码 / 空密码 / 错密码 → `ZIPX_ERR_PASSWORD`;正确密码 → 成功并逐字节校验内容):ZIP 侧覆盖 `enc-zipcrypto.zip`、`enc-aes256.zip`、`enc-aes256-store.zip`,RAR 侧覆盖 `enc-v6.rar`;另验证「提供密码后限额依旧生效」与「失败路径绝不发布文件」
|
||||
- **大文件档位** —— `medium_bomb.zip`(比率 ≈ 238)在默认限额下被拒、在大档位下通过;降低后的大档位仍生效
|
||||
- **RAR 引擎**(`tests/test_rar_extract.c`,37 项检查)—— 格式分派(改名的 ZIP / 垃圾数据均被拒)、每个可达引擎错误码的翻译、限额交接(`large=1` 原样传入 `rar_extract()`)、超过上限的字典路径,以及真实归档覆盖
|
||||
- **7z 引擎**(`tests/test_sevenz_extract.c` + `tests/run-sevenz-tests.sh`)—— 真实 `.7z` fixture 逐字节比对、加密头(三种情况:无密码 / 错密码 / 正确密码)、各策略下的冲突、取消、限额、缺失目标父目录,以及失败时绝不发布且 staging 树被清理的保证
|
||||
|
||||
前端还有一份无头检查 `.build/ui_retry_test.mjs`(`node .build/ui_retry_test.mjs`):把 `assets/main.js` 载入桩 DOM,验证加密失败后的密码重试流程——参数记忆、重试上限、取消与空密码的回落,共 27 项检查。它位于 `.build/`(gitignore 白名单之外),属于开发期验证脚本。
|
||||
|
||||
## 项目结构
|
||||
|
||||
```
|
||||
.
|
||||
├── Makefile # PS5 + Linux 构建(VERSION_TAG v1.9.3M)
|
||||
├── install-libmicrohttpd.sh # 一次性依赖安装器
|
||||
├── gen-asset-module.py # 将 assets/* 内联为 gzip 压缩的 C 数组
|
||||
├── assets/ # HTML / CSS / JS / 图标 / param.json
|
||||
├── src/ # C 载荷源码
|
||||
│ ├── main.c websrv.c filemgr.c # 入口、HTTP 前端、任务模型
|
||||
│ ├── upload.c download.c # 流处理
|
||||
│ ├── extract.c # /api/extract 分派器(ZIP + RAR + 7z)
|
||||
│ ├── zip_extract.{c,h} zipx_common.c # ZIP 引擎
|
||||
│ ├── zipx_volume.c zipx_volstream.c # ZIP 分卷探测 + 拼接流
|
||||
│ ├── rar_extract.{c,h} # RAR 引擎(unrar7 后端)
|
||||
│ ├── sevenz_extract.{c,h} # 7z 引擎
|
||||
│ ├── sevenz_chain.{c,h} # 7z 拉式 codec 链(BCJ2 等)
|
||||
│ ├── sevenz_header.{c,h} # 7z 头部读取 / `-mhe=on` 解密
|
||||
│ ├── sevenz_mt.{c,h} # 7z 多线程 LZMA2 解码
|
||||
│ ├── sevenz_volstream.{c,h} # 7z 分卷流拼接
|
||||
│ └── app_installer.c # PS5 Media 启动器安装器
|
||||
├── third_party/ # vendored:zlib、minizip-ng、unrar7、7z(SDK 子集)
|
||||
│ ├── unrar7/ # rarlab UnRAR 7.20.1,静态库 + C API 门面
|
||||
│ ├── minizip-ng/ # ZIP 读取器
|
||||
│ ├── zlib/ # minizip-ng 的压缩后端
|
||||
│ └── 7z/ # LZMA SDK 解码子集
|
||||
├── tests/ # POSIX / 主机测试套件
|
||||
│ ├── test_zip_extract.c
|
||||
│ ├── test_rar_extract.c
|
||||
│ ├── test_sevenz_extract.c # 7z 用例驱动
|
||||
│ ├── make_fixtures.py # 重新生成测试 fixture
|
||||
│ ├── run-tests.sh # 一次性运行器(ZIP + RAR)
|
||||
│ ├── run-sevenz-tests.sh # 7z 运行器
|
||||
│ ├── compat/ # 小型 Win32 / MSYS 垫片
|
||||
│ └── fixtures/ fixtures-7z/ fixtures-real/ # 生成的测试归档
|
||||
├── docs/
|
||||
│ ├── HANDOVER.md # v1.8 时代的开发手册(历史存档,现行见根目录 HANDOVER.md)
|
||||
│ ├── UPGRADE-v1.7-zip-large-file-profile.md
|
||||
│ ├── UPGRADE-v1.8-rar-support.md
|
||||
│ └── screenshots/ # README 截图
|
||||
├── THIRD_PARTY_NOTICES # 捆绑库署名
|
||||
├── LICENSE # GPLv3+
|
||||
└── README.md
|
||||
```
|
||||
|
||||
## 备注
|
||||
|
||||
- 复制、移动、删除、上传、下载作为单个后台任务运行。一个任务运行时,其它文件操作会被拒绝。
|
||||
- 删除是递归且永久的。没有回收站。
|
||||
- 复制/移动任务可取消。单个文件的部分拷贝会被移除,但部分拷贝的文件夹会保留在原地,以避免在合并进已存在目标文件夹时误删既有文件。
|
||||
- 上传任务可取消。尽可能移除部分上传的临时文件。
|
||||
- 下载文件夹或多个选中项会生成 tar 流。tar 归档由载荷生成,不会先写入 PS5 存储。
|
||||
- 若浏览器在载荷进程仍在运行时被关闭重开,UI 可恢复活动任务显示。
|
||||
- 文本编辑仅限于上述精选扩展名列表。非 UTF-8 与超大文件会被拒绝。
|
||||
- 文件名通过 Web API 以 UTF-8 传输。载荷也会保留挂载文件系统返回的遗留字节序名称,以便混合 USB 文件名编码仍能正确显示与操作。
|
||||
|
||||
## 常见问题
|
||||
|
||||
- **这是自制应用,不应故意修改系统进程或内核内存。** 若遇到内核崩溃(kernel panic),请确保使用较新的越狱方法与 ELF 加载器,或回退到你惯用的稳定方法。
|
||||
- **P2JB 用户** —— 若此载荷触发内核崩溃,请避免在该环境下使用。当每次重试代价高昂时,稳定性比便利更重要。
|
||||
- **准备阶段可能耗时较久** —— 当文件夹含大量文件时,它会累加文件夹大小并检查剩余空间,这有助于避免启动一个无法安全完成的复制 / 移动 / 上传 / 下载。
|
||||
- **`err_extract_entry_too_large`** —— 默认归档上限为单条目 512 GiB / 500:1 比率(覆盖典型 3A 游戏归档中单个约 300 GiB 未压缩文件)。若超过默认,请确认大文件提示(磁盘上 > 480 GiB 的归档会出现),拆分归档,或直接向 API 传入 `large=1`。
|
||||
- **`err_extract_unsupported`** —— 这个包本机读不了:既非 `.zip` / `.rar` / `.7z` 的文件;ZIP 条目用了 stored / deflated 之外的压缩方法;7z 用了不支持的 coder;分卷命名不被识别(RAR 分卷若叫 `x.rar.001`,需改名为 `x.part1.rar`、`x.part2.rar` ……);或早于 RAR 1.4 的归档。**加密归档与多卷归档不属于这一类**——两者都支持。界面会在括号里附上后端原文,指明具体原因。
|
||||
- **`err_extract_dict_too_large`** —— RAR 归档声明的压缩字典超过本构建支持的上限(4096 MiB),且 unrar 请求允许超额。报错文案会同时给出归档需要的尺寸与构建允许的尺寸。这是**刻意拒绝**:另一条路是一次性分配整个字典窗口,rarlab 自家 CLI 默认也会拒绝,16 GB 共享内存的主机更是承受不起。请在 PC 上用不超过 4 GiB 的字典重新压缩(`-md`),或在 PC 上解压。注意 RAR5 格式本身把这个字段卡在 4 GiB,所以这种情况只可能来自更新版 RAR7 头格式写出的归档。
|
||||
- **`err_extract_password`** —— 归档已加密,而本次提交的密码缺失或错误。这也包括带加密头(`-mhe=on`)的 7z:文件名与条目尺寸都在头部里,头解密之前连条目列表都读不出来。ZIP / RAR(以及现在的 7z 加密头)在失败后会弹出密码框(取消或留空即放弃),可用正确密码按原参数重试,最多 3 次;解压 7z 时仍会提前询问一次密码。
|
||||
|
||||
## 署名
|
||||
|
||||
本项目 **fork 自 [owendswang/ps5-web-file-manager](https://github.com/owendswang/ps5-web-file-manager)**(GPL-3.0)。Web UI、任务模型与 PS5 打包方式均源自该项目;上游作者以 GPL-3.0 发布,是本衍生作品得以存在的前提。
|
||||
|
||||
**怎么区分上游原版与本仓改版:** 自 v1.9.3 起版本号带 `M` 后缀(`vX.Y.ZM`),*M* 即 *Modified*(改版);上游 owendswang 的发布版是纯 `vX.Y.Z`。因此 `v1.9.3M` 只可能出自本仓,而这个字母同时出现在 ELF 文件名、PS5 启动通知、`/api/version` 与网页右下角。v1.9.3M 之前的发布早于该约定,保留原本的无后缀编号。
|
||||
|
||||
本项目另参考了以下项目构建:
|
||||
|
||||
- **[ps5-payload-dev/websrv](https://github.com/ps5-payload-dev/websrv):** HTTP 服务器结构、静态资源内联思路、PS5 浏览器/websrv 行为与 PKG 安装函数。许可证:GPLv3+。
|
||||
- **[ps5-payload-dev/ftpsrv](https://github.com/ps5-payload-dev/ftpsrv):** PS5 载荷约定、主屏启动器/安装流程参考、进程处理风格与启动安装参考。许可证:GPLv3+。
|
||||
- **[seregonwar/zftpd](https://github.com/seregonwar/zftpd):** PS5 TCP socket 缓冲调优与高吞吐传输行为参考。许可证:MIT。
|
||||
- **[itsPLK/ps5-payload-manager](https://github.com/itsPLK/ps5-payload-manager):** 载荷构建行为。许可证:GPLv3。
|
||||
- **[libmicrohttpd](https://ftp.gnu.org/gnu/libmicrohttpd/):** 用作内嵌 HTTP 服务器库。由 GNU 以 LGPL 许可;本载荷以 SDK 提供的静态库链接它。
|
||||
- **[ps5-payload-dev/sdk](https://github.com/ps5-payload-dev/sdk):** 载荷构建基础。许可证:GPLv3+。
|
||||
- **[etaHEN](https://github.com/etaHEN/etaHEN):** 退出前用于返回 PS5 主屏的 ShellUI URI 导航。许可证:GPLv3。
|
||||
- **[ezremote](https://github.com/cy33hc/ps5-ezremote-client):** PKG 预览功能的参考出处。许可证:**GPL-2.0-only** —— 其源文件未声明 "or later",因此**无法**与本项目的 GPL-3.0 代码组合。**未取其任何代码**:`src/pkg_info.c` 是独立的 C99 实现(它还负责 `.pkg` 条目表与 `param.json` 字段,而 ezremote 根本没有 `.pkg` 解析器;JSON 走的是 `src/json_util.c` 里自写的分词器,不是 json-c)。详见 `docs/REWRITE-FEASIBILITY.md` §2.2。
|
||||
- **[zlib-ng/minizip-ng](https://github.com/zlib-ng/minizip-ng):** `/api/extract` 端点使用的 ZIP 读取器。vendored 于 `third_party/minizip-ng/`。许可证:zlib。
|
||||
- **[zlib](https://www.zlib.net/):** minizip-ng 的压缩后端。vendored 于 `third_party/zlib/`。许可证:zlib。
|
||||
- **[rarlab UnRAR (opello/unrar)](https://github.com/opello/unrar):** v1.9 起 `/api/extract` 使用的 RAR 读取器(7.20.1)。vendored 于 `third_party/unrar7/`。许可证:UnRAR 免费软件许可。
|
||||
|
||||
## 许可证
|
||||
|
||||
本项目以 **GPLv3 或更高版本** 分发,与作为实现参考的 GPLv3+ 项目保持一致。见 [`LICENSE`](./LICENSE)。
|
||||
|
||||
第三方项目保留各自许可证。请勿在未保留相应许可证声明的情况下,将署名项目的资源或源码复制到其它发行版中。
|
||||
|
||||
若分发二进制,除本项目 GPL 许可外,还需遵守 `libmicrohttpd` 的 LGPL 条款。vendored 的 `zlib` 与 `minizip-ng` 源码以 zlib 许可分发;再分发用此特性构建的二进制时,保留 `third_party/zlib/LICENSE` 与 `third_party/minizip-ng/LICENSE` 中的版权声明。vendored 的 `unrar7`(RAR 引擎)依 UnRAR 免费软件许可分发;再分发用 v1.9 或更高版本构建的二进制时,保留 `third_party/unrar7/license.txt` 中的声明,且不得用其开发 RAR 兼容归档器或重新实现 RAR 压缩算法。
|
||||
|
||||
## 免责声明
|
||||
|
||||
非官方自制软件。仅在已越狱 PS5 主机上运行。使用风险自负——作者不对损坏、数据丢失、账号处罚或保修影响负责。请勿再分发 Sony 专有内容。依 GPLv3+,修改后的再分发必须公开其源码。
|
||||
@@ -0,0 +1,90 @@
|
||||
Third-Party Notices
|
||||
===================
|
||||
|
||||
This project vendors a minimal set of third-party source files under
|
||||
`third_party/` to provide the ZIP, RAR and 7z extraction features (the
|
||||
`/api/extract` endpoint). Their full license texts are included
|
||||
alongside the sources.
|
||||
|
||||
1. minizip-ng
|
||||
-----------
|
||||
Version : 4.2.2
|
||||
Source : https://github.com/zlib-ng/minizip-ng
|
||||
License : zlib (see third_party/minizip-ng/LICENSE)
|
||||
Files : third_party/minizip-ng/** (vendored subset, compiled with
|
||||
relaxed warnings into the final binary)
|
||||
|
||||
2. zlib
|
||||
------
|
||||
Version : 1.3.1
|
||||
Source : https://www.zlib.net/
|
||||
License : zlib (see third_party/zlib/LICENSE)
|
||||
Files : third_party/zlib/** (vendored subset, compiled with relaxed
|
||||
warnings into the final binary)
|
||||
|
||||
3. unrar (rarlab UnRAR source, v7.20.1)
|
||||
-------------------------------------
|
||||
Version : 7.20.1 (RAR 7.23-free source snapshot, 2025-10-28)
|
||||
Source : https://www.rarlab.com/rar_add.htm — mirrored by
|
||||
https://github.com/opello/unrar (commit 97e1780)
|
||||
License : UnRAR freeware license (see third_party/unrar7/license.txt)
|
||||
Files : third_party/unrar7/** (RARDLL source set compiled with
|
||||
relaxed warnings; unrar_c_api.h is project-authored and
|
||||
carries the project's license)
|
||||
|
||||
4. LZMA SDK (7z decoder)
|
||||
----------------------
|
||||
Version : 26.03 (2026-09-03)
|
||||
Source : https://www.7-zip.org/sdk.html — release `lzma2603.7z` from
|
||||
https://github.com/ip7z/7zip/releases
|
||||
License : Public domain ("LZMA SDK is written and placed in the public
|
||||
domain by Igor Pavlov", see third_party/7z/DOC/lzma-sdk.txt)
|
||||
Files : third_party/7z/** (decoder-only subset, compiled with relaxed
|
||||
warnings; see third_party/7z/README.md for the file list)
|
||||
|
||||
The LZMA SDK is the engine behind `src/sevenz_extract.c` (the v1.9.x 7z
|
||||
support: LZMA/LZMA2/PPMd/Copy plus the BCJ, BCJ2 and Delta filters). Only the
|
||||
C implementation is used — it builds with the plain PS5 C toolchain and does
|
||||
not pull in the C++ runtime. The encoder half of the SDK is not vendored.
|
||||
|
||||
unrar is the engine behind `src/rar_extract.c` (the v1.9 RAR support: RAR4,
|
||||
RAR5 including WinRAR 6/7 "v6" compression, and multi-volume archives; the
|
||||
engine can also decrypt via RARSetPassword once a password channel is wired
|
||||
up). The UnRAR source may be used in any software to handle RAR archives,
|
||||
but may not be used to develop a RAR-compatible *archiver* or to re-create
|
||||
the RAR compression algorithm, which is proprietary.
|
||||
|
||||
(The v1.8 engine, DrMcCoy/dmc_unrar 1.7.0 under GPL-2.0-or-later, was
|
||||
replaced by the rarlab UnRAR source in v1.9; see git history under
|
||||
`third_party/unrar/` for its notice.)
|
||||
|
||||
Libraries in sections 1 and 2 are distributed under the zlib license, which
|
||||
permits redistribution in source and binary form provided the copyright
|
||||
notice and this list of conditions are retained. unrar is distributed under
|
||||
its own freeware terms. The LZMA SDK (section 4) is in the public domain and
|
||||
carries no conditions. See the individual LICENSE / license.txt files in
|
||||
each `third_party/` subdirectory for the complete terms.
|
||||
|
||||
|
||||
Reference-only projects (NOT vendored)
|
||||
--------------------------------------
|
||||
|
||||
The README Credits section lists a second class of project: ones this payload
|
||||
was *written with reference to*, whose code is not present in this repository
|
||||
and which therefore impose no obligations here. Keeping the two classes apart
|
||||
matters, because one of them is licence-incompatible with this codebase:
|
||||
|
||||
* websrv, ftpsrv, ps5-payload-manager, ps5-payload-dev/sdk, etaHEN (GPL-3.0 / 3.0+)
|
||||
* zftpd (MIT)
|
||||
* libmicrohttpd (LGPL; linked as the SDK-provided static library)
|
||||
* ezremote (GPL-2.0-only)
|
||||
|
||||
For ezremote specifically: GPL-2.0-only cannot be combined with this project's
|
||||
GPL-3.0, so it matters that no code was taken from it. The PKG-preview code in
|
||||
`src/pkg_info.c` is an independent implementation -- the two share only the
|
||||
on-disk SFO format facts, which no implementation can avoid. Line-by-line
|
||||
comparison and reasoning: `docs/REWRITE-FEASIBILITY.md` section 2.2.
|
||||
|
||||
`owendswang/ps5-web-file-manager` (GPL-3.0) is not a mere reference: it is the
|
||||
fork this project descends from, and the web UI, the task model and the PS5
|
||||
packaging all originate there.
|
||||
|
After Width: | Height: | Size: 268 B |
|
After Width: | Height: | Size: 139 B |
|
After Width: | Height: | Size: 184 B |
|
After Width: | Height: | Size: 1.8 KiB |
@@ -4,13 +4,35 @@
|
||||
<meta charset="utf-8">
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1">
|
||||
<title>PS5 Web File Manager</title>
|
||||
<link rel="icon" type="image/png" href="/icon0.png">
|
||||
<style>
|
||||
html, body { margin: 0; min-height: 100%; background: #111316; color: #edf0f2; }
|
||||
.init-loading {
|
||||
position: fixed;
|
||||
top: 0;
|
||||
right: 0;
|
||||
bottom: 0;
|
||||
left: 0;
|
||||
z-index: 20000;
|
||||
display: flex;
|
||||
align-items: center;
|
||||
justify-content: center;
|
||||
background: #111316;
|
||||
color: #aab4be;
|
||||
font: 18px system-ui, -apple-system, BlinkMacSystemFont, "Segoe UI", sans-serif;
|
||||
}
|
||||
</style>
|
||||
<link rel="stylesheet" href="/main.css">
|
||||
</head>
|
||||
<body>
|
||||
<main class="shell">
|
||||
<header class="topbar">
|
||||
<div id="path" class="path">/</div>
|
||||
<div id="spaceInfo" class="space-info"></div>
|
||||
<div class="top-left">
|
||||
<div id="path" class="path">/</div>
|
||||
</div>
|
||||
<div class="top-right">
|
||||
<div id="spaceInfo" class="space-info"></div>
|
||||
</div>
|
||||
<button id="exitBtn" class="icon-button power-button" type="button" aria-label="Exit"></button>
|
||||
</header>
|
||||
|
||||
@@ -19,56 +41,162 @@
|
||||
<button id="copyBtn" data-i18n="copy"></button>
|
||||
<button id="moveBtn" data-i18n="move"></button>
|
||||
<button id="renameBtn" data-i18n="rename"></button>
|
||||
<button id="downloadBtn" class="remote-only" data-i18n="download"></button>
|
||||
<button id="deleteBtn" class="danger" data-i18n="delete"></button>
|
||||
<button id="installPkgBtn" class="install-action" data-i18n="install" hidden></button>
|
||||
<button id="extractBtn" class="extract-action" data-i18n="extract"></button>
|
||||
<button id="pasteBtn" class="paste-action" hidden>
|
||||
<span id="pasteVerb" data-i18n="paste"></span>
|
||||
<span id="pasteName" class="paste-name"></span>
|
||||
<span id="pasteName" class="paste-name"></span><span id="pasteCount" class="paste-count"></span>
|
||||
<span id="pasteTargetText" data-i18n="pasteToCurrent"></span>
|
||||
</button>
|
||||
<button id="clearClipboardBtn" data-i18n="cancel" hidden></button>
|
||||
</div>
|
||||
<div class="tool-right">
|
||||
<button id="refreshBtn" data-i18n="refresh"></button>
|
||||
<div id="uploadGroup" class="upload-menu remote-only">
|
||||
<button id="uploadBtn" class="upload-main" data-i18n="upload"
|
||||
aria-haspopup="menu" aria-expanded="false"></button>
|
||||
<div id="uploadMenu" class="upload-menu-list" role="menu" hidden>
|
||||
<button id="uploadFilesItem" type="button" role="menuitem" data-i18n="uploadFiles"></button>
|
||||
<button id="uploadFolderItem" type="button" role="menuitem" data-i18n="uploadFolder"></button>
|
||||
</div>
|
||||
</div>
|
||||
<button id="newTextBtn" data-i18n="newText"></button>
|
||||
<button id="mkdirBtn" data-i18n="mkdir"></button>
|
||||
</div>
|
||||
</section>
|
||||
<input id="uploadFiles" type="file" multiple hidden>
|
||||
<input id="uploadFolder" type="file" multiple webkitdirectory hidden>
|
||||
|
||||
<section id="content" class="content">
|
||||
<table>
|
||||
<thead>
|
||||
<tr>
|
||||
<th class="select-col"><label class="select-hit"><input id="selectAll" type="checkbox"></label></th>
|
||||
<th class="name-col" data-i18n="name"></th>
|
||||
<th class="type-col" data-i18n="type"></th>
|
||||
<th class="size-col" data-i18n="size"></th>
|
||||
<th class="time-col" data-i18n="mtime"></th>
|
||||
<th class="mode-col" data-i18n="mode"></th>
|
||||
<th class="name-col sortable" data-sort="name">
|
||||
<div class="name-head">
|
||||
<button id="parentBtn" class="parent-nav-button" type="button" aria-label="Parent directory" disabled>
|
||||
<img src="/icon-back.png" alt="">
|
||||
</button>
|
||||
<span class="name-heading" data-i18n="name"></span>
|
||||
</div>
|
||||
</th>
|
||||
<th class="type-col sortable" data-sort="type" data-i18n="type"></th>
|
||||
<th class="size-col sortable" data-sort="size" data-i18n="size"></th>
|
||||
<th class="time-col sortable" data-sort="mtime" data-i18n="mtime"></th>
|
||||
<th class="mode-col sortable" data-sort="mode" data-i18n="mode"></th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody id="files"></tbody>
|
||||
</table>
|
||||
<div id="empty" class="empty" data-i18n="empty" hidden></div>
|
||||
<div id="contentLoading" class="content-loading" hidden>
|
||||
<div class="content-loading-text" data-i18n="readDir"></div>
|
||||
</div>
|
||||
</section>
|
||||
|
||||
<footer class="status">
|
||||
<div id="statusText" data-i18n="ready"></div>
|
||||
<div id="dropHint" class="status-hint remote-only" data-i18n="dropUploadHint" hidden></div>
|
||||
<div id="versionText" class="version-text"></div>
|
||||
</footer>
|
||||
</main>
|
||||
|
||||
<div id="textEditorOverlay" class="text-editor-overlay" hidden>
|
||||
<section class="text-editor-panel">
|
||||
<div id="textEditorPath" class="text-editor-path"></div>
|
||||
<textarea id="textEditor" class="text-editor" wrap="off" spellcheck="false"
|
||||
autocomplete="off" autocorrect="off" autocapitalize="off"></textarea>
|
||||
<div id="textEditorStatus" class="text-editor-status"></div>
|
||||
<div class="text-editor-actions">
|
||||
<button id="textEditorCloseBtn" class="secondary" data-i18n="close"></button>
|
||||
<button id="textEditorSaveBtn" class="primary" data-i18n="save"></button>
|
||||
</div>
|
||||
</section>
|
||||
</div>
|
||||
|
||||
<div id="imagePreviewOverlay" class="text-editor-overlay" hidden>
|
||||
<section class="text-editor-panel image-preview-panel">
|
||||
<div id="imagePreviewName" class="text-editor-path"></div>
|
||||
<div class="image-preview-stage">
|
||||
<img id="imagePreview" class="image-preview" alt="">
|
||||
</div>
|
||||
<button id="imagePreviewCloseBtn" class="secondary" data-i18n="close"></button>
|
||||
</section>
|
||||
</div>
|
||||
|
||||
<div id="pkgInfoOverlay" class="text-editor-overlay" hidden>
|
||||
<section class="text-editor-panel pkg-info-panel" role="dialog" aria-modal="true" aria-labelledby="pkgInfoTitle">
|
||||
<h2 id="pkgInfoTitle" class="pkg-info-title" data-i18n="pkgInfoTitle"></h2>
|
||||
<div class="pkg-info-content">
|
||||
<div class="pkg-info-image-stage">
|
||||
<img id="pkgInfoImage" class="pkg-info-image" src="/icon-pkg.png" alt="">
|
||||
</div>
|
||||
<div id="pkgInfoFields" class="pkg-info-fields"></div>
|
||||
</div>
|
||||
<div class="text-editor-actions">
|
||||
<button id="pkgInfoCloseBtn" class="secondary" data-i18n="close"></button>
|
||||
<button id="pkgInfoInstallBtn" class="primary" data-i18n="install"></button>
|
||||
</div>
|
||||
</section>
|
||||
</div>
|
||||
|
||||
<div id="permissionOverlay" class="permission-overlay" hidden>
|
||||
<section class="permission-panel" role="dialog" aria-modal="true" aria-labelledby="permissionTitle">
|
||||
<h2 id="permissionTitle" class="permission-title" data-i18n="permissionsTitle"></h2>
|
||||
<div id="permissionPath" class="permission-path"></div>
|
||||
<div class="permission-grid">
|
||||
<div class="permission-row permission-head" aria-hidden="true">
|
||||
<span class="permission-scope"></span>
|
||||
<span>R</span><span>W</span><span>X</span>
|
||||
</div>
|
||||
<div class="permission-row">
|
||||
<span class="permission-scope" data-i18n="permissionOwner"></span>
|
||||
<label><input id="permissionOwnerRead" type="checkbox"><span>R</span></label>
|
||||
<label><input id="permissionOwnerWrite" type="checkbox"><span>W</span></label>
|
||||
<label><input id="permissionOwnerExecute" type="checkbox"><span>X</span></label>
|
||||
</div>
|
||||
<div class="permission-row">
|
||||
<span class="permission-scope" data-i18n="permissionGroup"></span>
|
||||
<label><input id="permissionGroupRead" type="checkbox"><span>R</span></label>
|
||||
<label><input id="permissionGroupWrite" type="checkbox"><span>W</span></label>
|
||||
<label><input id="permissionGroupExecute" type="checkbox"><span>X</span></label>
|
||||
</div>
|
||||
<div class="permission-row">
|
||||
<span class="permission-scope" data-i18n="permissionOther"></span>
|
||||
<label><input id="permissionOtherRead" type="checkbox"><span>R</span></label>
|
||||
<label><input id="permissionOtherWrite" type="checkbox"><span>W</span></label>
|
||||
<label><input id="permissionOtherExecute" type="checkbox"><span>X</span></label>
|
||||
</div>
|
||||
</div>
|
||||
<label class="permission-octal">
|
||||
<span data-i18n="permissionOctal"></span>
|
||||
<input id="permissionMode" type="text" inputmode="numeric" pattern="0[0-7]{3}"
|
||||
minlength="4" maxlength="4" autocomplete="off" spellcheck="false">
|
||||
</label>
|
||||
<label id="permissionRecursiveOption" class="permission-recursive" hidden>
|
||||
<input id="permissionRecursive" type="checkbox" checked>
|
||||
<span data-i18n="permissionRecursive"></span>
|
||||
</label>
|
||||
<div class="permission-actions">
|
||||
<button id="permissionCancelBtn" class="secondary" data-i18n="close"></button>
|
||||
<button id="permissionApplyBtn" class="primary" data-i18n="apply"></button>
|
||||
</div>
|
||||
</section>
|
||||
</div>
|
||||
|
||||
<div id="dropUploadOverlay" class="drop-upload-overlay remote-only" hidden>
|
||||
<div class="drop-upload-message" data-i18n="dropUpload"></div>
|
||||
</div>
|
||||
|
||||
<div id="taskOverlay" class="task-overlay" hidden>
|
||||
<div class="task-panel">
|
||||
<div id="tasks" class="tasks"></div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div id="initLoading" class="init-loading" aria-hidden="true">
|
||||
<div class="loading-dots">
|
||||
<span></span>
|
||||
<span></span>
|
||||
<span></span>
|
||||
</div>
|
||||
</div>
|
||||
<div id="initLoading" class="init-loading" aria-hidden="true">Loading...</div>
|
||||
|
||||
<script src="/main.js"></script>
|
||||
</body>
|
||||
|
||||
@@ -1,16 +1,40 @@
|
||||
window.WFM_LANG = {
|
||||
appTitle: "PS5 Web File Manager",
|
||||
versionTooltip: "Modified build by LisherSong (upstream releases carry no M suffix)",
|
||||
copy: "Copy",
|
||||
move: "Move",
|
||||
delete: "Delete",
|
||||
download: "Download",
|
||||
upload: "Upload",
|
||||
uploadFiles: "Upload Files",
|
||||
uploadFolder: "Upload Folder",
|
||||
dropUploadHint: "Drag files or folders into this window to upload",
|
||||
dropUpload: "Release to upload into this folder",
|
||||
copying: "Copying",
|
||||
moving: "Moving",
|
||||
deleting: "Deleting",
|
||||
changingPermissions: "Changing permissions",
|
||||
rename: "Rename",
|
||||
paste: "Paste",
|
||||
cancel: "Cancel",
|
||||
close: "Close",
|
||||
save: "Save",
|
||||
apply: "Apply",
|
||||
exit: "Exit",
|
||||
exitConfirm: "Exit and stop the file manager process?",
|
||||
exiting: "Exiting...",
|
||||
refresh: "Refresh",
|
||||
mkdir: "New Folder",
|
||||
newText: "New Text",
|
||||
install: "Install",
|
||||
installPackage: "Install package",
|
||||
pkgInfoTitle: "Package Information",
|
||||
pkgInfoLoading: "Reading package information...",
|
||||
pkgInstalling: "Starting package installation: {name}...",
|
||||
pkgInstallStarted: "Package installation started: {name}",
|
||||
file: "File",
|
||||
textFile: "Text",
|
||||
image: "Image",
|
||||
dir: "Folder",
|
||||
parent: "Parent",
|
||||
name: "Name",
|
||||
@@ -18,6 +42,19 @@ window.WFM_LANG = {
|
||||
size: "Size",
|
||||
mtime: "Modified",
|
||||
mode: "Mode",
|
||||
permissionsTitle: "Change permissions",
|
||||
permissionOwner: "Owner",
|
||||
permissionGroup: "Group",
|
||||
permissionOther: "Other",
|
||||
permissionOctal: "Octal mode",
|
||||
permissionRecursive: "Apply to all contents inside this folder",
|
||||
permissionObjectCount: "and {count} objects",
|
||||
permissionChangeTitle: "Change permissions for {name}",
|
||||
permissionInvalid: "Enter exactly four octal digits from 0 to 7, starting with 0. The original value has been restored.",
|
||||
unsavedPermissionConfirm: "The permissions have unsaved changes. Close without applying them?\n\nCancel keeps the permissions dialog open.",
|
||||
permissionChanging: "Changing permissions...",
|
||||
permissionChanged: "Permissions changed: {name} → {mode}",
|
||||
permissionChangeFailed: "Failed to change permissions: {error}",
|
||||
empty: "Folder is empty",
|
||||
ready: "Ready",
|
||||
freeSpace: "Free space",
|
||||
@@ -31,14 +68,21 @@ window.WFM_LANG = {
|
||||
checking: "Checking",
|
||||
pleaseWait: "Please wait",
|
||||
speedLabel: "Speed",
|
||||
itemsPerSecond: "{count} objects/s",
|
||||
progressLabel: "Progress",
|
||||
permissionProgress: "{done} / {total} objects",
|
||||
etaLabel: "ETA",
|
||||
elapsedLabel: "Elapsed",
|
||||
durationHours: "{hours}h {minutes}m",
|
||||
durationMinutes: "{minutes}m {seconds}s",
|
||||
durationSeconds: "{seconds}s",
|
||||
preparingTask: "Preparing task",
|
||||
canceling: "Canceling...",
|
||||
selectedItems: "{name} and {count} items",
|
||||
countItems: "{count} items",
|
||||
totalItems: "{count} items",
|
||||
readDir: "Reading folder...",
|
||||
processing: "Processing...",
|
||||
actionBusy: "{label}...",
|
||||
taskCreated: "{label} task created",
|
||||
actionDone: "{label} done",
|
||||
@@ -47,19 +91,58 @@ window.WFM_LANG = {
|
||||
taskCanceled: "{label} canceled",
|
||||
selectedForPaste: "Selected {name}. Enter the target folder, then choose {verb}",
|
||||
clipboardCleared: "Pending operation canceled",
|
||||
downloadStarted: "Download started: {name}",
|
||||
uploadFolderChoice: "Upload a folder?\n\nOK: choose a folder\nCancel: choose files",
|
||||
uploadOverwriteConfirm: "Items with the same name already exist: {names}\n\nOverwrite matching files and merge matching folders?",
|
||||
uploadingStatus: "Uploading {index}/{count}: {name}",
|
||||
uploadDone: "Upload done, {count} items",
|
||||
uploadFailed: "Upload failed: {error}",
|
||||
downloading: "Downloading",
|
||||
uploading: "Uploading",
|
||||
extract: "Extract",
|
||||
extractToCurrent: "Extract to current folder",
|
||||
extractUpload: "Upload and extract",
|
||||
extracting: "Extracting",
|
||||
extractConfirm: "Extract {name} to {path}?",
|
||||
extractOverwriteAsk: "If a file or folder with the same name already exists in the target:\n\nOK = overwrite same-name files (folders still merge)\nCancel = fail if the target already exists",
|
||||
extractPasswordAsk: "This archive may be encrypted (e.g. 7zAES).\n\nEnter the password to extract, or leave it empty to try without one.",
|
||||
extractPasswordFirstAsk: "This archive is encrypted.\n\nEnter the password to extract, or cancel to stop.",
|
||||
extractPasswordRetryAsk: "This archive is encrypted and the password did not work.\n\nEnter the password to try again, or cancel to stop.",
|
||||
extractUploadConfirm: "Upload and extract {name}?\n\nTarget folder: {path}\nThe uploaded archive will be deleted after success.",
|
||||
extractUploadAsk: "{name} is an archive.\n\nOK: upload and extract here\nCancel: upload only",
|
||||
extractStarted: "Extraction started: {name}",
|
||||
extractDone: "Extraction complete: {name}",
|
||||
extractProgress: "{done} / {total} files",
|
||||
extractLargeAsk: "The archive looks large ({size}). Enable the large-file profile?\n\nOK = yes (single file up to 1 TiB, archive total up to 4 TiB)\nCancel = default limits (single file 512 GiB, archive total 2 TiB); this archive may be rejected",
|
||||
extractLargeActive: "Large-file profile is enabled for this task",
|
||||
extractSelectArchive: "Select one archive to extract (ZIP / RAR / 7z)",
|
||||
extractOneAtATime: "Only one archive can be extracted at a time",
|
||||
extractSelectMainVolume: "Please select the main volume (.rar or .part01.rar)",
|
||||
extractArchivePending: "Preparing to extract {name}",
|
||||
sameSourceTarget: "Source and destination are the same. Cannot {label} {name}",
|
||||
removeConflictFirst: "A {existingType} named {name} already exists. To {label} this {sourceType}, delete that {existingType} first.",
|
||||
overwriteFiles: "Files with the same name will be overwritten: {names}",
|
||||
mergeDirs: "Folders with the same name will be merged. Internal files with the same name will be overwritten: {names}",
|
||||
continueConfirm: "Continue?",
|
||||
preparingPaste: "{label} preparing: calculating size and checking target space...",
|
||||
preparingPaste: "Preparing: calculating size and checking target space...",
|
||||
preparingStatus: "{label} preparing...",
|
||||
cancelPrepare: "Preparation canceled",
|
||||
cancelTaskConfirm: "Cancel current {label} task?",
|
||||
cancelFailed: "Cancel failed: {error}",
|
||||
tasksPollFailed: "Failed to read task status: {error}",
|
||||
transferComplete: "{label} completed: {name}\n\nTotal time: {duration}\nTransferred: {size}",
|
||||
transferCompleteFiles: "{label} completed: {name}\n\nTotal time: {duration}\nTransferred: {size}\nFiles: {count}",
|
||||
renamePrompt: "New name",
|
||||
mkdirPrompt: "Folder name",
|
||||
newTextPrompt: "Text file name",
|
||||
creatingText: "Creating text file...",
|
||||
createTextFailed: "Failed to create text file: {error}",
|
||||
unsavedSaveConfirm: "The text has unsaved changes. Close without saving?\n\nCancel keeps the editor open.",
|
||||
loadingText: "Loading text...",
|
||||
savingText: "Saving...",
|
||||
textSaved: "Text saved",
|
||||
openTextFailed: "Failed to open text: {error}",
|
||||
saveTextFailed: "Failed to save text: {error}",
|
||||
deleteConfirm: "Delete {name}?",
|
||||
deleteConfirmRecursive: "Delete {name}?\n\nFolders will be deleted recursively.",
|
||||
activeTask: "A task is running",
|
||||
@@ -70,11 +153,14 @@ window.WFM_LANG = {
|
||||
storageUsb: "USB",
|
||||
storageM2: "M.2",
|
||||
storageExtended: "Extended",
|
||||
storageRoot: "Root",
|
||||
storageCurrent: "Current",
|
||||
err_target_dir_not_writable: "Target folder is not writable: {path}",
|
||||
err_target_file_not_writable: "Target file is not writable: {path}",
|
||||
err_target_parent_not_writable: "Target parent folder is not writable: {path}",
|
||||
err_target_parent_not_writable: "Current folder is not writable: {path}",
|
||||
err_target_check_failed: "Cannot check target path: {path}",
|
||||
err_target_path_too_long: "Target path is too long",
|
||||
err_path_too_long: "Path is too long",
|
||||
err_space_check_failed: "Cannot read target free space: {path}",
|
||||
err_no_space: "Not enough target space. Required {required}, available {available}",
|
||||
err_active_task: "A task is running",
|
||||
@@ -82,9 +168,42 @@ window.WFM_LANG = {
|
||||
err_source_destination_same: "Source and destination are the same",
|
||||
err_destination_inside_source: "Cannot copy or move a folder inside itself",
|
||||
err_invalid_path: "Invalid path",
|
||||
err_invalid_mode: "Invalid permission mode",
|
||||
err_chmod_failed: "Cannot change permissions: {path}",
|
||||
err_pkg_type_invalid: "Only .pkg files can be installed",
|
||||
err_pkg_info_failed: "Could not read package information: {path}",
|
||||
err_pkg_install_failed: "Package installation failed ({arg})",
|
||||
err_pkg_install_unsupported: "Package installation is only available on PS5",
|
||||
err_file_not_found: "File not found",
|
||||
err_file_already_exists: "A file with the same name already exists",
|
||||
err_invalid_method: "Invalid request method",
|
||||
err_text_type_not_editable: "This file type is not supported by the text editor",
|
||||
err_text_file_too_large: "The text file is too large. The maximum size is 1 MiB",
|
||||
err_text_invalid_utf8: "The file is not valid UTF-8 text",
|
||||
err_text_file_changed: "The file changed after it was opened. Close and reopen it",
|
||||
err_text_file_not_writable: "The text file is not writable",
|
||||
err_request_body_too_large: "Too many selected items. Select fewer items and try again",
|
||||
err_unknown_api: "Unknown API",
|
||||
err_out_of_memory: "Out of memory",
|
||||
err_no_source_paths: "No source paths",
|
||||
err_destination_must_be_directory: "Destination must be a folder for multiple items",
|
||||
err_extract_open_failed: "Cannot open archive: {arg}",
|
||||
err_extract_corrupt: "Archive is corrupt or incomplete: {arg}",
|
||||
err_extract_unsupported: "Unsupported archive (only .zip, .rar and .7z are accepted, including their multi-volume and encrypted forms; this one uses a feature this build cannot handle): {arg}",
|
||||
err_extract_password: "Wrong password or the archive is not encrypted with the one supplied: {arg}",
|
||||
err_extract_unsafe_name: "Archive contains an unsafe path: {arg}",
|
||||
err_extract_special_entry: "Archive contains an unsupported special file: {arg}",
|
||||
err_extract_duplicate: "Archive contains duplicate entries: {arg}",
|
||||
err_extract_too_many_entries: "Archive has too many entries: {arg}",
|
||||
err_extract_entry_too_large: "A file inside the archive is too large: {arg}",
|
||||
err_extract_too_large: "Archive expands to too much data: {arg}",
|
||||
err_extract_ratio: "Suspicious compression ratio (possible zip bomb): {arg}",
|
||||
err_extract_too_deep: "Directory nesting is too deep: {arg}",
|
||||
err_extract_name_too_long: "File name or path is too long: {arg}",
|
||||
err_extract_dict_too_large: "The archive needs a larger dictionary than this device can handle: {arg}",
|
||||
err_extract_conflict: "A file or folder with the same name already exists: {arg}",
|
||||
err_extract_io: "Extraction read/write failed: {arg}",
|
||||
err_extract_crc: "CRC check failed: {arg}",
|
||||
err_extract_failed: "Extraction failed: {arg}",
|
||||
err_system_error: "System error: {arg}"
|
||||
};
|
||||
@@ -1,16 +1,40 @@
|
||||
window.WFM_LANG = {
|
||||
appTitle: "PS5 Web File Manager",
|
||||
versionTooltip: "本版为 LisherSong 改版(上游原版无 M 后缀)",
|
||||
copy: "复制",
|
||||
move: "移动",
|
||||
delete: "删除",
|
||||
download: "下载",
|
||||
upload: "上传",
|
||||
uploadFiles: "上传文件",
|
||||
uploadFolder: "上传文件夹",
|
||||
dropUploadHint: "可直接把文件或文件夹拖进窗口上传",
|
||||
dropUpload: "松开即上传到当前目录",
|
||||
copying: "复制",
|
||||
moving: "移动",
|
||||
deleting: "删除",
|
||||
changingPermissions: "修改权限",
|
||||
rename: "重命名",
|
||||
paste: "粘贴",
|
||||
cancel: "取消",
|
||||
close: "关闭",
|
||||
save: "保存",
|
||||
apply: "应用",
|
||||
exit: "退出",
|
||||
exitConfirm: "是否退出并关闭文件管理器进程?",
|
||||
exiting: "正在退出...",
|
||||
refresh: "刷新",
|
||||
mkdir: "新建目录",
|
||||
newText: "新建文本",
|
||||
install: "安装",
|
||||
installPackage: "安装 PKG",
|
||||
pkgInfoTitle: "PKG 信息",
|
||||
pkgInfoLoading: "正在读取 PKG 信息...",
|
||||
pkgInstalling: "正在提交安装:{name}...",
|
||||
pkgInstallStarted: "已开始安装:{name}",
|
||||
file: "文件",
|
||||
textFile: "文本",
|
||||
image: "图片",
|
||||
dir: "目录",
|
||||
parent: "上级目录",
|
||||
name: "名称",
|
||||
@@ -18,6 +42,19 @@ window.WFM_LANG = {
|
||||
size: "大小",
|
||||
mtime: "修改时间",
|
||||
mode: "权限",
|
||||
permissionsTitle: "修改权限",
|
||||
permissionOwner: "拥有者",
|
||||
permissionGroup: "组",
|
||||
permissionOther: "其他",
|
||||
permissionOctal: "八进制表示",
|
||||
permissionRecursive: "应用于文件夹内的所有内容",
|
||||
permissionObjectCount: "等 {count} 个对象",
|
||||
permissionChangeTitle: "修改 {name} 的权限",
|
||||
permissionInvalid: "请输入以 0 开头、由 0 到 7 组成的四位八进制权限。已恢复为原权限值。",
|
||||
unsavedPermissionConfirm: "权限有未保存的修改,是否不保存并关闭?\n\n取消将留在权限弹窗中。",
|
||||
permissionChanging: "正在修改权限...",
|
||||
permissionChanged: "权限已修改:{name} → {mode}",
|
||||
permissionChangeFailed: "修改权限失败:{error}",
|
||||
empty: "目录为空",
|
||||
ready: "就绪",
|
||||
freeSpace: "可用空间",
|
||||
@@ -31,14 +68,21 @@ window.WFM_LANG = {
|
||||
checking: "检查中",
|
||||
pleaseWait: "请稍候",
|
||||
speedLabel: "速度",
|
||||
itemsPerSecond: "{count} 个对象/秒",
|
||||
progressLabel: "进度",
|
||||
permissionProgress: "{done} / {total} 个对象",
|
||||
etaLabel: "剩余",
|
||||
elapsedLabel: "耗时",
|
||||
durationHours: "{hours}小时 {minutes}分",
|
||||
durationMinutes: "{minutes}分 {seconds}秒",
|
||||
durationSeconds: "{seconds}秒",
|
||||
preparingTask: "正在准备任务",
|
||||
canceling: "正在取消...",
|
||||
selectedItems: "{name} 等 {count} 项",
|
||||
countItems: "{count} 项",
|
||||
totalItems: "共 {count} 项",
|
||||
readDir: "读取目录...",
|
||||
readDir: "读取目录中...",
|
||||
processing: "处理中...",
|
||||
actionBusy: "{label}...",
|
||||
taskCreated: "{label}任务已创建",
|
||||
actionDone: "{label}完成",
|
||||
@@ -47,19 +91,58 @@ window.WFM_LANG = {
|
||||
taskCanceled: "{label}已取消",
|
||||
selectedForPaste: "已选择 {name},进入目标目录后点击{verb}",
|
||||
clipboardCleared: "已清除待操作项目",
|
||||
downloadStarted: "已开始下载 {name}",
|
||||
uploadFolderChoice: "选择要上传的目录?\n\n确定:选择目录\n取消:选择文件",
|
||||
uploadOverwriteConfirm: "目标中已存在同名项目: {names}\n\n是否覆盖同名文件并合并同名目录?",
|
||||
uploadingStatus: "正在上传 {index}/{count}: {name}",
|
||||
uploadDone: "上传完成,共 {count} 项",
|
||||
uploadFailed: "上传失败: {error}",
|
||||
downloading: "下载中",
|
||||
uploading: "上传中",
|
||||
extract: "解压",
|
||||
extractToCurrent: "解压到当前目录",
|
||||
extractUpload: "上传并解压",
|
||||
extracting: "解压",
|
||||
extractConfirm: "解压 {name} 到 {path}?",
|
||||
extractOverwriteAsk: "若目标已存在同名文件或目录:\n\n确定 = 覆盖同名文件(目录仍会合并)\n取消 = 若目标已存在则失败",
|
||||
extractPasswordAsk: "此压缩包可能加密了(如 7zAES)。\n\n输入密码后解压,留空则尝试无密码解压。",
|
||||
extractPasswordFirstAsk: "此压缩包已加密。\n\n输入密码后解压,取消则停止解压。",
|
||||
extractPasswordRetryAsk: "此压缩包已加密,密码不正确。\n\n输入密码重试,取消则停止解压。",
|
||||
extractUploadConfirm: "上传并解压 {name}?\n\n目标目录:{path}\n成功后将删除上传的压缩包。",
|
||||
extractUploadAsk: "这是压缩包 {name}。\n\n确定:上传后自动解压到当前目录\n取消:仅上传,不解压",
|
||||
extractStarted: "已开始解压 {name}",
|
||||
extractDone: "解压完成:{name}",
|
||||
extractProgress: "{done} / {total} 个文件",
|
||||
extractLargeAsk: "压缩包体积较大({size}),是否启用「大文件模式」?\n\n确定 = 启用(单文件最大 1 TiB / 总解压最大 4 TiB)\n取消 = 默认限制(单文件 512 GiB / 总解压 2 TiB),可能拒绝此压缩包",
|
||||
extractLargeActive: "此任务已启用大文件模式",
|
||||
extractSelectArchive: "选中一个压缩包后才能解压(ZIP / RAR / 7z)",
|
||||
extractOneAtATime: "一次只能解压一个压缩包",
|
||||
extractSelectMainVolume: "请改选主卷(如 .rar 或 .part01.rar)",
|
||||
extractArchivePending: "正在准备解压 {name}",
|
||||
sameSourceTarget: "源和目标相同,不能{label} {name}",
|
||||
removeConflictFirst: "目标中已存在同名{existingType} {name}。要{label}{sourceType},请先删除该{existingType}才能继续。",
|
||||
overwriteFiles: "同名文件将被覆盖: {names}",
|
||||
mergeDirs: "同名目录将被合并,内部同名文件会被覆盖: {names}",
|
||||
continueConfirm: "继续?",
|
||||
preparingPaste: "{label}准备中:正在计算大小并检查目标空间...",
|
||||
preparingPaste: "准备中:正在计算大小并检查目标空间...",
|
||||
preparingStatus: "{label}准备中...",
|
||||
cancelPrepare: "已取消准备操作",
|
||||
cancelTaskConfirm: "取消当前{label}任务?",
|
||||
cancelFailed: "取消失败: {error}",
|
||||
tasksPollFailed: "任务状态读取失败: {error}",
|
||||
transferComplete: "{label}完成:{name}\n\n总耗时:{duration}\n传输大小:{size}",
|
||||
transferCompleteFiles: "{label}完成:{name}\n\n总耗时:{duration}\n传输大小:{size}\n文件数量:{count}",
|
||||
renamePrompt: "新名称",
|
||||
mkdirPrompt: "目录名",
|
||||
newTextPrompt: "文本文件名",
|
||||
creatingText: "正在新建文本...",
|
||||
createTextFailed: "新建文本失败: {error}",
|
||||
unsavedSaveConfirm: "文本有未保存的修改,是否不保存并关闭?\n\n取消将留在编辑器中。",
|
||||
loadingText: "正在读取文本...",
|
||||
savingText: "正在保存...",
|
||||
textSaved: "文本已保存",
|
||||
openTextFailed: "打开文本失败: {error}",
|
||||
saveTextFailed: "保存文本失败: {error}",
|
||||
deleteConfirm: "确认删除 {name}?",
|
||||
deleteConfirmRecursive: "确认删除 {name}?\n\n目录会递归删除。",
|
||||
activeTask: "有任务正在执行",
|
||||
@@ -70,11 +153,14 @@ window.WFM_LANG = {
|
||||
storageUsb: "USB存储",
|
||||
storageM2: "M2扩充存储",
|
||||
storageExtended: "扩展存储",
|
||||
storageRoot: "根分区",
|
||||
storageCurrent: "当前目录",
|
||||
err_target_dir_not_writable: "目标目录不可写: {path}",
|
||||
err_target_file_not_writable: "目标文件不可写: {path}",
|
||||
err_target_parent_not_writable: "目标父目录不可写: {path}",
|
||||
err_target_parent_not_writable: "当前目录不可写: {path}",
|
||||
err_target_check_failed: "无法检查目标路径: {path}",
|
||||
err_target_path_too_long: "目标路径过长",
|
||||
err_path_too_long: "路径过长",
|
||||
err_space_check_failed: "无法读取目标剩余空间: {path}",
|
||||
err_no_space: "目标空间不足,需要 {required},可用 {available}",
|
||||
err_active_task: "有任务正在执行",
|
||||
@@ -82,9 +168,42 @@ window.WFM_LANG = {
|
||||
err_source_destination_same: "源和目标相同",
|
||||
err_destination_inside_source: "不能复制或移动目录到它自己的内部",
|
||||
err_invalid_path: "路径无效",
|
||||
err_invalid_mode: "权限格式无效",
|
||||
err_chmod_failed: "无法修改权限:{path}",
|
||||
err_pkg_type_invalid: "只能安装 .pkg 文件",
|
||||
err_pkg_info_failed: "无法读取 PKG 信息:{path}",
|
||||
err_pkg_install_failed: "PKG 安装失败({arg})",
|
||||
err_pkg_install_unsupported: "PKG 安装仅支持 PS5 环境",
|
||||
err_file_not_found: "文件不存在",
|
||||
err_file_already_exists: "同名文件已存在",
|
||||
err_invalid_method: "请求方式无效",
|
||||
err_text_type_not_editable: "该文件类型不支持文本编辑",
|
||||
err_text_file_too_large: "文本文件过大,最大支持 1 MiB",
|
||||
err_text_invalid_utf8: "文件不是有效的 UTF-8 文本",
|
||||
err_text_file_changed: "文件在打开后已发生变化,请关闭后重新打开",
|
||||
err_text_file_not_writable: "文本文件不可写",
|
||||
err_request_body_too_large: "选择的项目过多,请减少选择后重试",
|
||||
err_unknown_api: "未知接口",
|
||||
err_out_of_memory: "内存不足",
|
||||
err_no_source_paths: "没有源路径",
|
||||
err_destination_must_be_directory: "多个项目的目标必须是目录",
|
||||
err_extract_open_failed: "无法打开压缩包: {arg}",
|
||||
err_extract_corrupt: "压缩包损坏或不完整: {arg}",
|
||||
err_extract_unsupported: "不支持的压缩包(只认 .zip / .rar / .7z,含它们的分卷与加密版本;这个包用了本机处理不了的特性):{arg}",
|
||||
err_extract_password: "密码错误,或压缩包未使用所提供的密码加密: {arg}",
|
||||
err_extract_unsafe_name: "压缩包包含不安全的路径: {arg}",
|
||||
err_extract_special_entry: "压缩包包含不支持的特殊文件: {arg}",
|
||||
err_extract_duplicate: "压缩包包含重复条目: {arg}",
|
||||
err_extract_too_many_entries: "压缩包条目过多: {arg}",
|
||||
err_extract_entry_too_large: "压缩包内单个文件过大: {arg}",
|
||||
err_extract_too_large: "压缩包解压后总大小过大: {arg}",
|
||||
err_extract_ratio: "压缩比异常(疑似压缩炸弹): {arg}",
|
||||
err_extract_too_deep: "目录层级过深: {arg}",
|
||||
err_extract_name_too_long: "文件名或路径过长: {arg}",
|
||||
err_extract_dict_too_large: "压缩包需要的字典超出本机可承受范围: {arg}",
|
||||
err_extract_conflict: "目标已存在同名文件或目录: {arg}",
|
||||
err_extract_io: "解压读写失败: {arg}",
|
||||
err_extract_crc: "CRC 校验失败: {arg}",
|
||||
err_extract_failed: "解压失败: {arg}",
|
||||
err_system_error: "系统错误: {arg}"
|
||||
};
|
||||
@@ -0,0 +1,258 @@
|
||||
# v1.9.3M 真机验证清单
|
||||
|
||||
> 目标:在真机上把 **v1.9.3M 相对 v1.9.2 的全部改动**走一遍。
|
||||
> 对应任务 #46(五项规定项)+ 「字典超限报错」修复 + 本次「版本号加 M 改版标记」。
|
||||
> 逐项打勾,失败项记文案原文。
|
||||
|
||||
## 0. 物料
|
||||
|
||||
| 项 | 值 |
|
||||
|---|---|
|
||||
| 待测 ELF | `web-file-mgr-v1.9.3M.elf`(903,448 B,sha256 `8ca47d5a…86b7`) |
|
||||
| **回滚 ELF** | `.build/rel-v1.9.2/web-file-mgr-v1.9.2.elf`(870,488 B,sha256 `177e90fe…8e84`,**从 GitHub Release 下载并已核验**) |
|
||||
| 测试归档 | `.build/device-test/`(22 个文件,2.5 MB,含 `MANIFEST.txt` 指纹) |
|
||||
| 监听端口 | 默认 `8888`,通知栏显示实际端口 |
|
||||
|
||||
> **这一轮(2026-09-24 晚)又重编了三次**,都只动前端资源(上传菜单、拖拽提示、页脚
|
||||
> 状态行钳制、口令提示键、菜单行高亮、解压按钮常显),C 代码一字节没改。六个构建的
|
||||
> **文件尺寸都是 903,448 B**,sha256 各不相同,`.rodata` 逐轮
|
||||
> +0x140 / +0x980 / +0x100 / +0x180 / +0x240:
|
||||
> `7b5ab00c…`(首轮)→ `212107a6…`(+ 上传菜单)→ `da36834d…`(+ 文案与状态行钳制)→
|
||||
> `cf2c0fcf…`(+ 菜单行高亮修复)→ **`8ca47d5a…`(+ 解压按钮常显置灰,本轮待测)**。
|
||||
> **以 sha256 为准,别用文件尺寸判断"包换没换"。**
|
||||
|
||||
⚠️ **回滚只能用 `.build/rel-v1.9.2/` 那个**。项目根目录里曾经并存的、同名的
|
||||
`web-file-mgr-v1.9.2.elf`(903,448 B 的**未发布工作树**)已挪到
|
||||
`.build/elf-v1.9.2-worktree-f3164efa.elf`,根目录现在只剩本次待测的 `v1.9.3M`。
|
||||
|
||||
> **带 `M` = LisherSong 改版,不带 `M` = 上游原版。** 从这一版起版本号统一带 `M`
|
||||
> 后缀(如 `v1.9.3M`),所以「名字里有没有 M」本身就是上游 / 改版的判据。
|
||||
|
||||
### 发送与打开
|
||||
|
||||
```sh
|
||||
nc -q0 <PS5_IP> 9021 < web-file-mgr-v1.9.3M.elf
|
||||
# 看 PS5 左上角通知:应显示 PS5 Web File Manager + v1.9.3M + 监听端口
|
||||
# 浏览器打开 http://<PS5_IP>:8888/
|
||||
```
|
||||
|
||||
### 通用纪律
|
||||
|
||||
1. **每次解压都新建一个空目标目录**。已知遗留问题:含目录条目的包在「覆盖」模式下解到同一
|
||||
目录第二次必失败(三引擎同构,与本次改动无关)—— 别把它记成回归。
|
||||
2. 多卷归档**一次把整个文件夹拖进去**,别只传子卷(UI 对「只选中子卷」会置灰并要求改选首卷)。
|
||||
3. 每项记三样:**通过/失败**、**界面文案原文**、**截图**。
|
||||
4. 任务列表可直接在浏览器看:`http://<PS5_IP>:8888/api/tasks`。
|
||||
|
||||
---
|
||||
|
||||
## 1. 版本号四处一致 + 改版标记(规定项⑤,10 秒)
|
||||
|
||||
| 位置 | 期望 |
|
||||
|---|---|
|
||||
| PS5 启动通知 | `v1.9.3M` |
|
||||
| 页面底部状态栏的版本号(`#versionText`) | `v1.9.3M` |
|
||||
| `http://<PS5_IP>:8888/api/version` | JSON 里 version = `v1.9.3M` |
|
||||
| **鼠标悬停**在版本号上 | 浮出提示「本版为 LisherSong 改版(上游原版无 M 后缀)」/ 英文版同义 |
|
||||
|
||||
三处(+ 悬停)不一致 = 版本宏或前端兜底串没进二进制。ELF 内已核对:`v1.9.3M`
|
||||
出现 1 次、`v1.9.2` 出现 **0** 次;两条 tooltip 文案在解压后的 lang 资源里逐一命中。
|
||||
|
||||
> 若你此前已经刷过不带 M 的 `v1.9.3`:那个包只差字符串,**功能行为与本版完全一致**,
|
||||
> 所以先前测出的结果仍然有效,不必因为加了 M 就重测一遍功能项。
|
||||
> 反过来,只要界面显示的是 `v1.9.3`(无 M)就是旧包,`v1.9.3M` 才是本次待测。
|
||||
|
||||
---
|
||||
|
||||
## 2. 字典超限报错(本次修复的**唯一**新行为,优先做)
|
||||
|
||||
上传 `00-dict-limit.rar`(**7,055 B**,一个 8 GiB 字典的合成归档)→ 解压 → 选冲突策略 → 开始。
|
||||
|
||||
| 检查 | 期望 |
|
||||
|---|---|
|
||||
| 错误码/文案 | `extract_dict_too_large`,文案含 **`8192 MiB (limit 4096 MiB)`** |
|
||||
| **不应**出现 | 「压缩包内单个文件过大: **hello.txt**」← 修复前的错误归因(该文件只有 7 B 级) |
|
||||
| 目标目录 | **不应**出现 `hello.txt`,也不留下垃圾文件 |
|
||||
| 行为 | 干净失败,不是崩溃、不是中途 OOM |
|
||||
|
||||
> 判读:若看到「单个文件过大」⇒ 装的是旧 ELF;若看到字典文案 ⇒ 这一项通过。
|
||||
> 这项改动**只改报错、行为零变化**,所以「拒绝解压」是**正确**结果。
|
||||
|
||||
---
|
||||
|
||||
## 3. 三类分卷解压(规定项①)
|
||||
|
||||
准备:把 `.build/device-test/` 里对应子目录整体上传到 PS5 的某个目录,然后在该目录里解压。
|
||||
|
||||
| 引擎 | 上传哪个目录 | 点哪个文件 | 期望输出 |
|
||||
|---|---|---|---|
|
||||
| **ZIP**(WinRAR 命名,字节切分) | `01-zip-vol/` | `parts.part1.zip` | 4 项:`readme.txt`、`sub/data.bin`、`sub/deep/more.bin`、`tail.bin` |
|
||||
| **ZIP**(Info-ZIP 分盘,偏移按盘算) | `01-zip-vol/` | `disks.zip` | 同上 4 项 |
|
||||
| **RAR**(RAR5,3 卷) | `01-rar-vol/` | `vol.part1.rar`(**必须选首卷**) | `big.bin`,**524,288 B** |
|
||||
| **7z**(7 卷,`.7z.001…`) | `01-7z-vol/` | 任一卷均可触发 | 一个 `_src/` 目录,内含 6 项:`readme.txt`、`binary.bin`、`zeros.bin`、`sub/code.bin`、`sub/nested.txt`、`sub/中文-テスト.txt` |
|
||||
|
||||
要点:
|
||||
- ZIP/RAR/7z 各测一遍**首卷与子卷**的触发行为:ZIP/7z 任一卷都该能触发;RAR 选子卷应**置灰并提示改选首卷**(这是设计行为,不是 bug)。
|
||||
- 7z 那组含**中文 + 日文文件名**,顺手验证 UTF-8 落盘。
|
||||
- 删掉(或改名)某一卷再试一次,应给「找不到分卷」类报错而不是静默成功 —— 顺带查负路径;负向夹具用的是
|
||||
`tests/fixtures/broken.zip.001`(缺后续卷)与 `gap.zip.001 + gap.zip.003`(缺第 2 卷),需要时一并上传。
|
||||
|
||||
---
|
||||
|
||||
## 4. 加密归档(规定项②,改动最集中)
|
||||
|
||||
口令:三个套件用的是**同一个测试口令**,定义在 `tests/run-sevenz-tests.sh`
|
||||
(`FIXTURE_PASSWORD`)与 `tests/make-zip-enc-fixtures.bat` / `tests/make-rar-fixtures.bat` 里,先去看一眼。
|
||||
|
||||
| 文件 | 加密方式 | 期望 |
|
||||
|---|---|---|
|
||||
| `02-encrypted/enc-zipcrypto.zip` | ZipCrypto | 提示输入口令 → 解出 `root.txt`、`dir/nested.txt` |
|
||||
| `02-encrypted/enc-aes256.zip` | WinZip AES-256 | 同上 |
|
||||
| `02-encrypted/enc-aes256-store.zip` | AES-256 + 存储 | 同上 |
|
||||
| `02-encrypted/enc-v6.rar` | RAR5 `-hp`(加密头) | **连列表都要口令** → 提示输入 → 解出同两文件 |
|
||||
| `02-encrypted/aes.7z` | 7zAES(数据加密) | 7z 会在**开始前**先问口令,然后解出 `_src/` 那 6 项 |
|
||||
| `02-encrypted/aeshe.7z` | **7zAES + `-mhe=on`(加密头)** | **本次头号目标**:加密头由新增的 `src/sevenz_header.c` 自解;应能正常列出并解出 `_src/` 那 6 项 |
|
||||
|
||||
必测的负路径:
|
||||
|
||||
| 场景 | 期望 |
|
||||
|---|---|
|
||||
| 口令故意输错(每格式至少一次) | 报 `err_extract_password` 并弹出重试框,**最多 3 次**;取消即结束,不卡死 |
|
||||
| 重试时换正确口令 | 第 2/3 次能成功(证明「记住原请求」的逻辑生效:冲突策略、大文件选配不丢) |
|
||||
| 7z 加密头 + 错口令 | 应是口令错误提示,**不是**「不受支持的归档 / 损坏」 |
|
||||
| **某次提示文案** | 第一次失败说「**此压缩包已加密。输入密码后解压,取消则停止解压。**」;第二次起才是「密码不正确」(第一次没输过密码,不该说你输错了) |
|
||||
| **错误文案里的条目名** | 中文/日文条目名必须**正常显示**,不得出现 `â®…` 或方框乱码 |
|
||||
|
||||
> 判读:`aeshe.7z` 在 v1.9.2 上是**已知缺口**(`tests/run-sevenz-tests.sh` 的 `KNOWN_GAPS`
|
||||
> 里原来就写着 `aeshe`),v1.9.3M 才闭合。这一项通过 = 最后一个 7z 缺口在真机确认关闭。
|
||||
|
||||
---
|
||||
|
||||
## 5. 上传入口 / 口令提示 / 名称编码(本轮修复,优先做,2 分钟)
|
||||
|
||||
这一节全部是**界面与前端 codec** 的改动,测起来最快,也最容易看出装的是不是新包。
|
||||
|
||||
### 5a 上传按钮变成菜单
|
||||
|
||||
| 检查 | 期望 |
|
||||
|---|---|
|
||||
| 按钮外观 | 「上传」右侧带小三角(▾) |
|
||||
| 点一下 | 在按钮正下方弹出列表:**上传文件 / 上传文件夹**(不再是「主按钮 + 一个小箭头」) |
|
||||
| 选「上传文件」 | 弹系统文件选择器,可多选 |
|
||||
| 选「上传文件夹」 | 弹目录选择器(原「小箭头」的功能,没丢) |
|
||||
| 键盘 | 打开后焦点在第一项;`Esc` 关闭并把焦点还给按钮;`↑/↓` 在两项间移动 |
|
||||
| 选中高亮 | 鼠标移到哪一项、或 `↑/↓` 停在哪一项,**只有那一行**亮;颜色一致;高亮完全落在菜单面板内,**不越出边框、不压住相邻行**(修复前是越界蓝框 + 两侧弧线,且 hover 与键盘焦点两行同时亮) |
|
||||
| 点别处 | 菜单关闭 |
|
||||
|
||||
### 5b 拖拽提示文案
|
||||
|
||||
| 检查 | 期望 |
|
||||
|---|---|
|
||||
| 页脚(版本号左边) | 常显灰字:**可直接把文件或文件夹拖进窗口上传** |
|
||||
| 拖文件到窗口 | 仍然出现「松开即上传到当前目录」覆盖层(旧行为不变) |
|
||||
| 在 PS5 自带浏览器里打开 | 上传按钮与这行灰色提示**都不显示**(`remote-only`,控制台浏览器没有拖拽源) |
|
||||
| 上传中/选中文件时 | 页脚状态文案变长时,状态文字**单行截断成 `…`**,提示与版本号都还在同一行(不再撑破页脚) |
|
||||
|
||||
### 5c 加密包「上传即解压」必须问口令(**这就是你报的那个 bug**)
|
||||
|
||||
**复现原步骤**:进一个**中文名字的文件夹**(或任意非 ASCII 目录名),上传一个**有密码的 ZIP**
|
||||
→ 弹「这是压缩包 x.zip,确定:上传后自动解压 / 取消:仅上传」→ 点确定。
|
||||
|
||||
| 检查 | 期望(修复后) |
|
||||
|---|---|
|
||||
| **上传完成后** | 直接弹出「**此压缩包已加密。输入密码后解压,取消则停止解压。**」 |
|
||||
| **不应**出现 | 一个只有「确定」的错误框、且必须自己去按工具栏「解压」才给输密码 ← 修复前的行为 |
|
||||
| 输入正确口令 | 继续解压并成功;结束后源压缩包被删除(上传即解压的既有行为) |
|
||||
| 取消口令框 | 走原来的失败提示,不静默 |
|
||||
| 中文条目名的包解压失败时 | 错误框里的条目名是**中文**,不是 `â®…ç§.psd` 那种乱码 |
|
||||
|
||||
> 根因(供判读):口令重试原来是按**路径**记住原始请求的,而路径是非 ASCII 时
|
||||
> 「页面持有的字符串」与「服务端任务回报的字符串」编码表示不同 ⇒ 查不到 ⇒ 不弹口令框,
|
||||
> 只能手动再解压一次。现在按**任务 id** 记,路径编码再也不会影响它。
|
||||
|
||||
### 5e 解压按钮常显置灰
|
||||
|
||||
工具栏里现在**一直有「解压」按钮**,不再只在选中压缩包时才冒出来。
|
||||
|
||||
| 检查 | 期望 |
|
||||
|---|---|
|
||||
| 什么都不选 | 按钮**在**工具栏里(复制/移动/重命名/下载/删除 之后),**灰色、点不动** |
|
||||
| 停在灰色按钮上 | 出现提示「**选中一个压缩包后才能解压(ZIP / RAR / 7z)**」 |
|
||||
| 选中一个文件夹 | 仍然灰色、点不动 |
|
||||
| 选中**一个**压缩包(`.zip` / `.rar` / `.7z`) | 按钮**变亮可点**,提示变成「解压到当前目录: <包名>」 |
|
||||
| 选中**两个**压缩包 | 又变灰,提示「一次只能解压一个压缩包」 |
|
||||
| 只选中 `.part02.rar` 这类子卷 | 变灰,提示「请改选主卷(如 .rar 或 .part01.rar)」(旧行为) |
|
||||
| 按钮文字 | 是短标签「**解压**」,与相邻按钮等宽(不再是「解压到当前目录」) |
|
||||
| 有任务在跑时 | 变灰(和其他按钮一起被锁) |
|
||||
|
||||
> 为什么改成常显:旧版不选中压缩包就**完全没有**这个按钮,用户不知道有这个功能。
|
||||
> 代价是工具栏常态宽了 96 px ⇒ 窗口窄到 **1190 px** 以下工具栏会换成两行
|
||||
> (英文界面是 1350 px;控制台 1920 px、1280 px 都不受影响)。
|
||||
|
||||
---
|
||||
|
||||
## 6. 分卷 RAR 进度条实时走动(规定项④)
|
||||
|
||||
`01-rar-vol/` 那组只有 512 KB,进度条会一闪而过 ⇒ 用你自己那份大分卷 RAR(之前那个 ≈11.6 GB 的包已经被清理了,
|
||||
可以用 WinRAR 现造一个:`-m1 -md=4g -v1g`,内容选一个几 GB 的可压缩文件)。
|
||||
|
||||
| 检查 | 期望 |
|
||||
|---|---|
|
||||
| 进度条 | 按**字节**持续推进,不是长时间 0% 后直接跳 100% |
|
||||
| 速度读数 | 有合理 MB/s 读数(它是 250 ms 瞬时采样,抖动正常;只信「总字节 ÷ 总耗时」) |
|
||||
| `entries_done` | 跨卷的**单个大条目**场景下可能长时间停在 0 —— 这是设计如此(`assets/main.js:2118`),看字节进度即可 |
|
||||
| 取消 | 中途取消能停下,不残留半成品(取消是条目粒度) |
|
||||
|
||||
---
|
||||
|
||||
## 7. 大 ZIP:160 GB / 9.5 万文件(规定项③,最后做)
|
||||
|
||||
这项只能在真机跑,且最费时。
|
||||
|
||||
| 检查 | 期望 |
|
||||
|---|---|
|
||||
| scan 阶段 | 不 OOM、不长时间无响应;进度条在走 |
|
||||
| 内存 | 峰值平稳(scan 只读中央目录,不解码) |
|
||||
| 空间预检 | 目标分区空间不足时应**提前**报错(`check_space()` 要求双份空间) |
|
||||
| 完成 | 0 报错解完;条目数与源一致 |
|
||||
|
||||
---
|
||||
|
||||
## 8. 取证与回滚
|
||||
|
||||
**失败时提供这三样**(我按这个定位,不用你再复述):
|
||||
1. 界面/通知栏**文案原文**(含错误码,如 `extract_dict_too_large`);
|
||||
2. `http://<PS5_IP>:8888/api/tasks` 的返回;
|
||||
3. 截图(含目标目录文件列表)。
|
||||
|
||||
**回滚**(一条命令,产物已核验):
|
||||
|
||||
```sh
|
||||
nc -q0 <PS5_IP> 9021 < .build/rel-v1.9.2/web-file-mgr-v1.9.2.elf
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 结果记录表
|
||||
|
||||
| # | 项目 | 结果 | 文案/备注 |
|
||||
|---|---|---|---|
|
||||
| 1 | 版本号四处一致 + 悬停改版提示 | ☐ | |
|
||||
| 2 | 字典超限报错(8 GiB 字典) | ☐ | |
|
||||
| 3a | ZIP 分卷(`parts.partN.zip`) | ☐ | |
|
||||
| 3b | ZIP 分盘(`disks.z01+.zip`) | ☐ | |
|
||||
| 3c | RAR 分卷(选首卷 / 选子卷置灰) | ☐ | |
|
||||
| 3d | 7z 分卷(含中日文文件名) | ☐ | |
|
||||
| 4a | ZIP 加密 ×3(ZipCrypto / AES / AES-store) | ☐ | |
|
||||
| 4b | RAR 加密头 `-hp` | ☐ | |
|
||||
| 4c | 7z `aes.7z` | ☐ | |
|
||||
| 4d | **7z `aeshe.7z`(`-mhe=on` 加密头)** | ☐ | |
|
||||
| 4e | 错口令 + 重试(三格式) | ☐ | |
|
||||
| 4f | 首次失败提示文案 / 条目名不乱码 | ☐ | |
|
||||
| 5a | 上传按钮 → 菜单(文件 / 文件夹) | ☐ | |
|
||||
| 5b | 页脚拖拽提示 + PS5 浏览器里隐藏 | ☐ | |
|
||||
| 5c | **加密包上传即解压 → 直接弹口令框**(原 bug) | ☐ | |
|
||||
| 5d | 菜单行高亮(只有一行亮、不越界、hover 与键盘同色) | ☐ | |
|
||||
| 5e | 解压按钮常显置灰(灰 → 选中一个包变亮 → 选两个又变灰) | ☐ | |
|
||||
| 6 | 分卷 RAR 进度条 | ☐ | |
|
||||
| 7 | 大 ZIP 160 GB / 9.5 万文件 | ☐ | |
|
||||
@@ -0,0 +1,618 @@
|
||||
# 解压性能:实测、根因、提速方案
|
||||
|
||||
> 实测 2026-09-15 · 对照物 = 官方 7-Zip 26.03(上游 v1.8 的 helper 就是它)
|
||||
> 复现:`python tests/bench_driver.py --big --runs 3`;WSL 同环境对比见 `.build/bench/wsl-*.sh`
|
||||
|
||||
> **✅ 方案 A 已落地(2026-09-16)**:`Asm/x86/LzmaDecOpt.asm` + `7zAsm.asm` 已 vendor 到
|
||||
> `third_party/7z/`,jwasm `-elf64 -DABI_LINUX` 汇编进 PS5 与 Linux 两条链路,
|
||||
> `LzmaDec.o` 加 `-DZ7_LZMA_DEC_OPT`。实测 **1.39 s → 1.05–1.13 s(1.26×)**,解出字节与
|
||||
> C 版逐字节一致;7z 测试矩阵 28 checks 全过。Makefile 对该优化做了条件化(无 jwasm 自动
|
||||
> 退回纯 C)并依赖 Makefile 本身触发重编(flag 变化不会被 make 察觉)。
|
||||
>
|
||||
> **✅ 方案 B 已落地(2026-09-16)**:`Lzma2DecMt.c` + `MtDec.c` + `Threads.c` 已 vendor,
|
||||
> 单一纯 LZMA2 folder(7-Zip 默认布局)走 SDK 并行解码器(`src/sevenz_mt.c` 适配层),
|
||||
> 8 线程 + 1 MiB inBufSize_MT;`SZ_ERROR_THREAD` 自动降级回单线程 chain(BCJ2/加密/奇异
|
||||
> 布局本来就由 chain 负责)。实测 329 MiB:1.05 s → **0.77 s(1.37×)**,与 7za -mmt=off
|
||||
> 打平(898 ms);7za -mmt=8 = 485 ms。7z/ZIP/RAR 163 checks 全绿。
|
||||
>
|
||||
> **✅ ZIP 引擎逐条目 fsync 移除(2026-09-16)** —— 原计划写的是「批量化(每 64MB/N 条刷一次)」,
|
||||
> **实际落地改为彻底移除**:publish 是纯 rename、又没有续解功能,逐条目 fsync 换不到任何东西
|
||||
> (RAR/7z 引擎本来就没有,三引擎现在统一为「不 sync、只 rename」)。8000 文件 fixture:fsync 版
|
||||
> \>200 s 未完成 → 无 fsync **14.5 s(≥14×)**。同机官方 7-Zip 反而要 >400 s(Defender
|
||||
> 实时扫描逐文件查杀;PS5 无此因素)。
|
||||
>
|
||||
> 与 §二 排除 1 不矛盾:那里测的是**单一大文件**归档,一次 fsync 本来就近乎免费;这里是
|
||||
> **8000 个文件**,成本随条目数线性叠加(且 PS5 无 Defender,比例只会更极端)。
|
||||
> 已知取舍:publish 之后到落盘之间断电,会出现「文件在但内容不完整」;要补只需在 extract
|
||||
> 收尾做**一次**目录/整盘 flush(PS5 是 FreeBSD 系,`syncfs()` 不一定有,`sync()` 是全盘、偏重)。
|
||||
> 代码现状见 `src/zip_extract.c:937-945`。
|
||||
|
||||
## 结论
|
||||
|
||||
**7z 解码我们比 7-Zip 慢 1.57×(单线程),根因已定位到一个具体的编译开关。**
|
||||
|
||||
不是架构问题,不是算法问题,不是编译选项问题 —— 是 **SDK 里有一份汇编版解码器我们没启用**:
|
||||
|
||||
```c
|
||||
/* LzmaDec.c */
|
||||
#ifdef Z7_LZMA_DEC_OPT
|
||||
int Z7_FASTCALL LZMA_DECODE_REAL(CLzmaDec *p, SizeT limit, const Byte *bufLimit); /* asm */
|
||||
#else
|
||||
... 纯 C 宏展开 + LzmaDec_DecodeReal2() /* ← 我们在这里 */
|
||||
#endif
|
||||
```
|
||||
|
||||
`LzmaDecOpt.asm` 是 Igor Pavlov 官方 SDK 的一部分(public domain),**1339 行**,实现同一个函数。开不开这个开关,实测差 1.5 倍。
|
||||
|
||||
---
|
||||
|
||||
## 一、同环境实测(关键:排除跨平台假象)
|
||||
|
||||
第一轮数据是在 Windows 上打的(我们 MinGW 构建 vs `7za.exe`),混了平台因素。重做:**在同一台机器、同一个 WSL Linux 环境、同一份归档、同一类编译器**下对比。
|
||||
|
||||
归档:329 MiB 解压量 / 22 MiB 压缩,LZMA2 solid,单文件
|
||||
|
||||
| 配置 | 单线程 | 8 线程 | 相对我们 |
|
||||
|---|---:|---:|---:|
|
||||
| **ours**(facade,含 staging + publish;当时仍含逐条目 fsync,2026-09-16 已移除) | **1.39 s** | — | 1.00× |
|
||||
| 官方 7-Zip(Linux 构建) | **0.89 s** | **0.51 s** | **1.57× / 2.73×** |
|
||||
|
||||
> 两个数字都是同一台机器上的实测。7-Zip 的 Linux 版和 Windows 版几乎一样快(0.89 vs 0.84 s),说明平台差异不是因素。
|
||||
|
||||
---
|
||||
|
||||
## 二、四个被实测排除的原因
|
||||
|
||||
排查过程里每个假设都先给出过错误结论,所以逐个记录:
|
||||
|
||||
| # | 假设 | 实验 | 结果 |
|
||||
|---|---|---|---|
|
||||
| 1 | **fsync / 写盘开销** | 两边都解到 `/dev/shm`(tmpfs,fsync 近乎免费) | ❌ 我们 1.47 s,磁盘上也是 1.47 s —— **fsync 成本可忽略** |
|
||||
| 2 | **pull 粒度太小**(64 KiB 输出块 → 5000+ 次调用) | 把 `SZ_OUT_CHUNK` 提到 1 MiB | ❌ 1.39 s,与 64 KiB 无差别。profile 显示 `node_pull` **只调用 329 次**,调度开销 ≈ 0 |
|
||||
| 3 | **编译选项保守**(我们用 `-O2 -w`) | `-O3` / `-march=native` / `-march=x86-64-v3` 各跑一遍 | ❌ 全部落在 1.31–1.52 s,无显著差异 |
|
||||
| 4 | **汇编优化只值 6%**(我曾据此推断"不是主因") | 对比 Windows 版(有 asm) 与 Linux 版 | ❌ **这个推断是错的** —— Linux 官方版同样含 asm,所以只看到 6% 的平台差异。见下节 |
|
||||
|
||||
---
|
||||
|
||||
## 三、真正的根因:profile 说话
|
||||
|
||||
`gprof`,同一份归档:
|
||||
|
||||
```
|
||||
% self calls name
|
||||
82.81 1.06 s 81237 LzmaDec_DecodeReal2 ← LZMA 解码核心(C 版)
|
||||
17.19 0.22 s 660 CrcUpdateT12 ← CRC32 校验
|
||||
0.00 0.00 s 329 node_pull ← 我们的链调度,可忽略
|
||||
0.00 0.00 s 329 szx_sink_write ← 写盘,可忽略
|
||||
0.00 0.00 s 1705 LzmaDec_DecodeToDic
|
||||
```
|
||||
|
||||
**82.8% 的时间在一个函数里,而那个函数有一个 asm 版本我们没有使用。**
|
||||
|
||||
这解释了为什么前四个假设全部落空:它们针对的都是那 0% 的部分。
|
||||
|
||||
### 三方交叉验证
|
||||
|
||||
- 我们的构建:未定义 `Z7_LZMA_DEC_OPT` → profile 里是 `LzmaDec_DecodeReal2` ✓
|
||||
- SDK 源码:明确写着 `#ifdef Z7_LZMA_DEC_OPT` 时声明外部 asm 符号 ✓
|
||||
- 官方 GCC 构建规则(`7zip_gcc_c.mak`):`USE_LZMA_DEC_ASM` 开关 + `jwasm` 汇编 `LzmaDecOpt.asm` ✓
|
||||
|
||||
### 附带发现:CRC 占 17%
|
||||
|
||||
`CrcUpdateT12`(slicing-by-12,**纯软件实现**)花掉 0.22 s。
|
||||
|
||||
> ⚠️ **2026-09-23 更正**:本节原先写着「7-Zip 解压时同样校验 CRC,所以这部分**不构成差距**」——**这条是错的**。
|
||||
> 依据是本仓 vendor 的官方构建规则 `third_party/7z/7zip_gcc_c.mak:298-310`:
|
||||
> ```
|
||||
> ifdef USE_X86_ASM
|
||||
> $O/7zCrcOpt.o: ../../../Asm/x86/7zCrcOpt.asm ← 官方走这条:汇编版
|
||||
> else
|
||||
> $O/7zCrcOpt.o: ../../7zCrcOpt.c ← 我们走这条:纯 C
|
||||
> ```
|
||||
> 而 `CpuArch.h:691` 的 `CPU_IsSupported_CRC32()` 说明那份汇编就是 SSE4.2 硬件 `crc32`。
|
||||
> 我们的 `third_party/7z/Asm/x86/` 里**只有** `7zAsm.asm` 与 `LzmaDecOpt.asm`,**没有 `7zCrcOpt.asm`**。
|
||||
> 所以这 17% **是真实差距的一部分**,不只是「可选的净提速」;它同时还是 MT 路径的**串行瓶颈**
|
||||
> (`src/sevenz_mt.c` 的 `mt_seq_write` 在调用线程上算 CRC,Amdahl 意义上压住了多线程上限)。
|
||||
>
|
||||
> **2026-09-23 实测补充(`.build/_crcbench.c`,本机 MinGW x64,329 MiB 同一块数据,跑两次)**:
|
||||
> `CrcUpdateT12` **3.18–3.51 GB/s**(329 MiB → 0.098–0.109 s),zlib `crc32` **2.59–2.80 GB/s**,
|
||||
> SSE4.2 `crc32` 指令(直线写法)**6.03–6.11 GB/s**。
|
||||
> → CRC 实际只占单线程 1.39 s 的 **≈7%**,上面那个「0.22 s / 17%」**大概率是 gprof 插桩放大的**
|
||||
> (`-pg` 对纯循环函数特别吃亏)。**本节往下请按 ≤7% 理解,不要再引用 17%。**
|
||||
> 另外实测挖出一个**会算错的陷阱**,见「方案 C」。
|
||||
|
||||
---
|
||||
|
||||
## 四、提速方案
|
||||
|
||||
### 方案 A(推荐):启用 asm 解码器
|
||||
|
||||
| 步骤 | 内容 |
|
||||
|---|---|
|
||||
| 1 | 取 `Asm/x86/LzmaDecOpt.asm` + `Asm/x86/7zAsm.asm` 入 `third_party/7z/` |
|
||||
| 2 | 用 **jwasm**(MASM 兼容汇编器,支持 ELF64 输出)汇编成 `.o` |
|
||||
| 3 | Makefile 加规则;`LzmaDec.c` 编译时加 `-DZ7_LZMA_DEC_OPT` |
|
||||
| 4 | 完整测试矩阵(163 checks)+ 基准复测 |
|
||||
|
||||
- **预期收益:1.39 s → ~0.95 s(≈1.45×)**,追平 7-Zip 单线程水平
|
||||
- **工作量**:小~中(一个汇编文件 + 一条 Makefile 规则 + 一个宏)
|
||||
- **风险**:中 —— 唯一的不确定点是 **jwasm 能否产出 PS5(prospero-clang / x86-64 ELF)可链接的目标文件**。这一条必须先验证再动手
|
||||
- **为什么"稳"**:asm 是 SDK 官方组成部分(同一位作者维护,与 C 版有链接时版本校验 `_3`,对不上会直接链接失败而不是静默出错);正确性由现有 163 项测试兜底
|
||||
|
||||
### 方案 B:多线程 LZMA2 解码
|
||||
|
||||
SDK 自带 `C/Lzma2DecMt.c`(1095 行,public domain)就是 7-Zip `-mmt` 的并行实现,实测 0.89 → 0.51 s。
|
||||
|
||||
- **预期收益:额外 1.75×**(与 A 叠加后 ≈ 2.6×,基本追平 7-Zip 全核)
|
||||
- **工作量**:大 —— 要重构 chain 的调度(block 级并行 + 字典依赖管理)
|
||||
- **风险**:中高(并发正确性、内存峰值;PS5 只有 8 核且 HTTP/任务系统同进程,建议限制线程数)
|
||||
- **前置**:建议先完成 A,因为 A 不改架构、收益确定、能独立验证
|
||||
|
||||
### 方案 C:CRC 加速(**实测后收益大幅缩水,且有一个会算错的陷阱**)
|
||||
|
||||
`.build/_crcbench.c` 实测(本机 MinGW x64,329 MiB 同一块数据,两次运行):
|
||||
|
||||
| 实现 | 吞吐 | 329 MiB 耗时 | 谁在用 |
|
||||
|---|---:|---:|---|
|
||||
| `CrcUpdateT12`(slicing-by-12) | 3.18–3.51 GB/s | 0.098–0.109 s | 我们:7z chain / MT 输出 / 逐条目 CRC |
|
||||
| zlib `crc32` | 2.59–2.80 GB/s | 0.123–0.133 s | 我们:ZIP 引擎 |
|
||||
| SSE4.2 `crc32` 指令(直线写法) | 6.03–6.11 GB/s | 0.056–0.057 s | 候选 |
|
||||
|
||||
> ⚠️ **陷阱:x86 的 `crc32` 指令算的是 CRC-32C(Castagnoli),不是三个格式要的 IEEE CRC-32。**
|
||||
> 同一份基准里的判定(标准向量 `"123456789"`):
|
||||
> `_mm_crc32_*` 得 **`0xE3069283`(CRC-32C)**,而 `CrcCalc` / zlib 得 **`0xCBF43926`(IEEE)**。
|
||||
> 所以**不能把 `CrcUpdate` 直接换成 `_mm_crc32_u64`** —— 必须补一个多项式转换
|
||||
> (GF(2) 上的 32×32 矩阵),或改用 pshufb / PCLMULQDQ 手写 IEEE 并行 CRC。
|
||||
> **这正是官方 `Asm/x86/7zCrcOpt.asm` 存在的意义**:它不是两行 intrinsic 包装。
|
||||
> UnRAR 那边同理 —— `third_party/unrar7/crc.cpp` 的硬件路径是 `USE_NEON_CRC32`(**ARM 专属**),
|
||||
> x86 上是 slicing-by-16 纯软件。
|
||||
|
||||
**收益重估(按实测)**:CRC 只占单线程 ≈7%(0.098 s / 1.39 s)。硬件指令直线写法 1.8× 于
|
||||
slicing-by-12,但还要扣掉多项式转换的开销 → 乐观估计单线程省 **0.05–0.07 s ≈ 4–5%**;
|
||||
MT 路径里它是串行分量(`sevenz_mt.c:64`),按 0.098 s 算 0.77 s → 约 0.72 s(≈6%),
|
||||
**不足以解释 8 线程下与 `7za -mmt=8`(0.485 s)的 1.6× 差距**。
|
||||
|
||||
**结论**:这仍然是**我们与 7-Zip 差距里确定存在**的一块(官方走 `USE_X86_ASM` 分支编汇编版,
|
||||
我们走 `else` 编纯 C),但**收益是个位数百分比,不是 10–15%,实现也不平凡**。
|
||||
风险倒是低:CRC 算错会**响亮失败**(每个条目报 `ZIPX_ERR_CRC`),现有 177 + 27 项测试会立刻抓住,
|
||||
不会静默写坏数据。**优先级从「最高」降为「可做,但别指望它拉平差距」。**
|
||||
|
||||
### 方案 D(备选,不推荐):上游的 helper 路线
|
||||
|
||||
直接把 7-Zip 做成独立进程,一步到位拿到 1.57×/2.73×。
|
||||
|
||||
不推荐的理由:
|
||||
|
||||
1. 引入外部 ELF 依赖 + IPC + 进程生命周期管理,**故障模式比现在多得多**("最稳"的反面)
|
||||
2. 方案 A 用一个文件 + 一条规则就能拿到 1.45×,D 的增量收益只有多线程那部分
|
||||
3. PS5 上还要处理 elfldr 加载;上游自己都是"单独分发,让用户手动放到 `/data/wfm/`"
|
||||
|
||||
---
|
||||
|
||||
## 五、执行顺序建议
|
||||
|
||||
```
|
||||
第一步 A(asm 解码器) ✅ 已落地 2026-09-16,实测 1.26×
|
||||
└─ 先验证 jwasm → ELF64 → prospero-ld 这条链能否走通 ✅ 走通了
|
||||
第二步 B(多线程) ✅ 已落地 2026-09-16,实测 1.37×
|
||||
└─ 在 A 的基础上做,目标 ~0.55 s ⚠️ 实际 0.77 s:仍慢于 7za -mmt=8 的 0.485 s
|
||||
第三步 C(CRC 加速) ⬜ 未做 —— 但**实测后收益降到 4–5%**,且有 CRC-32C 陷阱
|
||||
└─ 官方走 USE_X86_ASM 编汇编版,我们编纯 C,确是真实差距;但不值得为它冒险
|
||||
```
|
||||
|
||||
> **2026-09-23 补充**:B 之后我们与 `7za` 的对比是 0.77 s vs 单线程 0.898 s / 8 线程 0.485 s。
|
||||
> 也就是说单线程已打平,**8 线程下仍差约 1.6×**。
|
||||
> 曾把这段差距归给「没做的 CRC 硬件化」,但实测否掉了这个假设:CRC 全程只占 ≈7%(**且那是
|
||||
> gprof 放大后的口径,实测 0.098 s 更小**),把它全消掉也只值 4–5%,撑不起 1.6×。
|
||||
> 更可能的来源是 BCJ2 / 7zAES 布局仍走单线程 chain、以及 `Lzma2DecMt` 自身的线程扇出效率
|
||||
> —— 两者都要真机 profile 才能定性。
|
||||
> 其余候选(ZIP inflate 换 libdeflate、条目级并行、AES-NI)同理,均属「收益不可预期」或「工程量大」。
|
||||
|
||||
> **⚠️ 2026-09-23 限定:上面这个 1.59× 是在「最有利的输入形状」上测出来的,不能外推到真实归档。**
|
||||
>
|
||||
> 先看基准归档到底是什么(`tests/bench_driver.py:156-179` + payload 构造 `:144-156`):
|
||||
>
|
||||
> ```
|
||||
> payload = 文本块 ×240 → 一个大文件;再拼 ntoskrnl.exe ×30 → payload_mix.bin
|
||||
> 7za add = -t7z -m0=lzma2 -mx=5 -ms=on
|
||||
> ```
|
||||
>
|
||||
> 也就是 **1 个条目 / 1 个 solid folder / 1 个纯 LZMA2 coder / 未加密**。而 `sz_chain_lzma2_root()`
|
||||
> (`src/sevenz_chain.c:750-762`)要求**恰好** `num_coders == 1 && num_bonds == 0 &&
|
||||
> num_pack_streams == 1 && method == LZMA2` 才走 MT —— 换句话说,这个 fixture 是**唯一能让我们的
|
||||
> MT 生效的形状**,也是 7-Zip 拿不到任何结构优势的形状。它把 per-entry 开销(我们的强项)和
|
||||
> 非 LZMA2 布局(我们的弱项)**同时排除在外**了。
|
||||
>
|
||||
> 真实 PS5 归档(游戏包 / repack)几乎全是相反的形状:
|
||||
>
|
||||
> | 真实特征 | 对我们的后果 | 对 7-Zip 的后果 |
|
||||
> |---|---|---|
|
||||
> | 几千~几万个条目 | per-entry 开销主导(这块我们反而大幅领先,见上方 8000 文件数据) | 同样逐条目,无优势 |
|
||||
> | `-ms=off` / 超大归档切块 → **多 folder** | `num_pack_streams != 1` → **MT 失效**,退回单线程 chain | 跨 folder/块并行,`-mmt` 照常吃满 |
|
||||
> | exe/dll 用 **BCJ2** | `num_coders != 1` → **MT 失效** | 照常多线程 |
|
||||
> | **7zAES 加密** | `sz_chain_needs_password` → **MT 失效** | 照常多线程 |
|
||||
>
|
||||
> 结论:**1.59× 既不是上限也不是下限**,真实方向未知 —— 可能因 I/O 与 per-entry 成本被摊薄到无关,
|
||||
> 也可能因为整条解码退回单线程而比 1.59× 更糟。**在拿到真实归档上的 profile 之前,任何一侧的
|
||||
> 断言都是猜的。**
|
||||
>
|
||||
> 同样,下面这句原话也**未经验证**,暂按假设保留:
|
||||
> 「在真实场景(大游戏包)里受存储 I/O 限制,差距往往比这个倍数更小」——它成立的前提是解码项
|
||||
> 只占 wall-clock 的一小部分;按 0.77 s / 329 MiB ≈ 427 MiB/s 的解码吞吐与 PS5 存储带宽同量级来算,
|
||||
> 这个前提**未必成立**。
|
||||
|
||||
**不做任何优化时的现状也是可接受的**:1.39 s / 329 MiB ≈ 237 MiB/s 单线程吞吐;ZIP/RAR 两个引擎
|
||||
已分别压过/追平各自的官方实现,7z 单线程与 `7za -mmt=off` 打平。
|
||||
|
||||
---
|
||||
|
||||
## 六、真机首次实测(2026-09-23)
|
||||
|
||||
> ### ⚠️⚠️ 第二次更正(2026-09-23 深夜):拿到**真实归档的参数**,并实测了 MT
|
||||
> 用户给了 `D:\PPSA16608-e.part{1,2,3}.rar`,其中 **part3 当时还在盘上**,我用 WinRAR 7.23 的
|
||||
> `UnRAR lt` 直接读了它的头(该文件随后被用户清理掉,故下列数字是**一次性的实测记录**):
|
||||
>
|
||||
> | 项 | 实测值 |
|
||||
> |---|---|
|
||||
> | 格式 | **RAR 5**(不是 RAR4;头 8 字节 `52 61 72 21 1A 07 01 00`) |
|
||||
> | 分卷 | **卷 3 / 锁定(locked)** |
|
||||
> | **固实** | **不是固实** —— 直接解析 part3 主头:archive flags = `0x0013` = `VOLUME|VOLNUMBER|LOCK`,**`MHFL_SOLID=0x0004` 未置位**(`third_party/unrar7/headers5.hpp:25-29`)。旁证两处自洽:`MHFL_VOLNUMBER` ⇒ volnumber=2 ⇒ unrar 显示「卷 3」✓;`MHFL_LOCK` ⇒ unrar 显示「锁定」✓ |
|
||||
> | 关键条目 | `PPSA16608.exfat` **19 493 027 840 B → 打包 3 831 295 727 B**(**ratio 5.09 : 1**,高度可压缩) |
|
||||
> | 压缩参数 | **`RAR 5.0(v50) -m1 -md=4g`** —— **-m1「最快」档 + 4 GiB 字典** |
|
||||
> | 体量 | part3 = **3.568 GiB ≈ 该条目的打包大小** ⇒ 这个 19.5 GB 条目**整段都在 part3 内**(另有两个 KB 级小文件);parts 1+2(各 4 GB)装的是其余 ~1250 个条目 |
|
||||
>
|
||||
> **① UI 速度口径已核实 = 解压后字节。** `src/rar_extract.c:766` 在 `UCM_PROCESSDATA` 回调里
|
||||
> `c->bytes_done += p2`(`p2` 是 unrar 交出的**解压后**长度),`bytes_total` 累加 `hdr.UnpSize`。
|
||||
> ⇒ 用户看到的 **10–40 MB/s 是"吐出数据的速度"**。
|
||||
>
|
||||
> **② 「scan 阶段白解码一遍(2×)」这个最大嫌疑——已排除。**
|
||||
> `third_party/unrar7/dll.cpp:341-342` 只在 **`!Arc.Solid`** 时把 `RAR_SKIP` 走廉价的
|
||||
> `Arc.SeekToNext()` 分支;`extract.cpp:529`(`SkipSolid=Arc.Solid` ⇒ 真解码后丢弃)只在
|
||||
> **固实**时成立。本档**非固实** ⇒ 我们的 scan 只读头 + 跳过字节,**不解码**。
|
||||
> (这条曾是最有希望的一项:真固实的话 `scan + extract` 会把 11.6 GB 解码两遍。)
|
||||
>
|
||||
> **③ MT 收益已实测**(host、WinRAR 自带 **UnRAR 7.23**、`-mt<N>` 开关虽未见于 `-?` 帮助但可解析):
|
||||
>
|
||||
> | 夹具 | 载荷 | `-mt1` | `-mt8` | 加速 |
|
||||
> |---|---|---:|---:|---:|
|
||||
> | **代表性**(`-m1 -md4g`、5.27:1 可压缩、固实、2 GB、16 文件) | 2 GB | **386 MB/s** | **946 MB/s** | **2.45×** |
|
||||
> | 代表性同上但不固实(5.01:1) | 2 GB | 505 MB/s | 1222 MB/s | 2.42× |
|
||||
> | ~~非代表性~~(90% 随机数据 ⇒ 压缩块小) | 240 MB | 129 MB/s | 331 MB/s | ~~2.55×~~ **作废** |
|
||||
>
|
||||
> ⇒ **形状错误会让结论偏乐观**:第一版夹具用 90% 随机数据,压缩块远小于
|
||||
> `unpack50mt.cpp:150-152` 的 `LargeBlockSize=0x20000`(128 KiB)阈值,MT 全程生效;
|
||||
> 真实归档是 5:1 可压缩数据,块更大、会触发 `LargeBlock` 退化路径,实测也确实从 2.55× 降到 2.45×。
|
||||
> 结论:**在真实形状上 MT 值 ~2.4×**,可信。(单位均为**解压后** MB/s。)
|
||||
>
|
||||
> **④ 接线比原计划简单:1 个编译开关 + 1 个链接开关,不用改 vendored 源码、不用增删源文件。**
|
||||
> (下表中间那一行「把 `threadmisc.cpp` 加进源列表」**经实测作废** —— 行内已更正。)
|
||||
> | 改动 | 位置 | 为什么必需 |
|
||||
> |---|---|---|
|
||||
> | `-DRAR_SMP` | `Makefile:UNRAR7_CXX_FLAGS`(PS5 与 host 两处) | `os.hpp:42-45` 的 `#define RAR_SMP` 在 `#ifdef _WIN_ALL` 内;且 `unpack.cpp:7-9` 的 `#include "unpack50mt.cpp"` 就在 `#ifdef RAR_SMP` 里 ⇒ 不定义宏则整个 MT 解码器**根本不参与编译** |
|
||||
> | ~~把 `threadmisc.cpp` 加进 `UNRAR7_SRCS`~~ **← 这一条是错的,不要做** | — | `GetNumberOfThreads()` 确实定义在 `threadmisc.cpp:178`,但 `threadpool.cpp:5` **已经** `#include "threadmisc.cpp"` ⇒ 它**已经**被编进 `threadpool.o`(实测回执:`nm unrar7_threadpool.o` 能查到 `T GetNumberOfThreads` / `T GetNumberOfCPU`)。**再加进源列表 = 重符号链接失败。** 同理 `blake2sp.cpp` 也不必补 —— `blake2s.cpp:27` 已经 `#include "blake2sp.cpp"`。官方 POSIX makefile 只列 `threadpool.o`、不列 `threadmisc.o`/`blake2sp.o`,正是这个原因 |
|
||||
> | `-pthread`(编译+链接) | `Makefile` 的 PS5 link 行 | `threadpool.cpp` 的 `_UNIX` 路径用 pthread cond/mutex(**不用 `sem_t`**)。SDK 侧 `target/lib/libpthread.a` 在位、`target/include/pthread.h:198-237` 声明齐全 |
|
||||
>
|
||||
> **实测回执(宿主 MinGW,2026-09-23)**:用**与 PS5 完全相同的 50 个源文件**、只加 `-DRAR_SMP`,
|
||||
> 链接**一次通过**(`bench_mt.exe` 1,011,448 B):`nm` 里 `Unpack5MT` 出现 1 次、`ThreadPool` 12 次;
|
||||
> 换成 `-DZIPSFX`(关掉 `RAR_SMP`)后 `Unpack5MT` 归 0。
|
||||
> ⇒ **接线 = 1 个编译开关 + 1 个链接开关,零源码改动、零源文件增删。**
|
||||
> **为什么不必改 `dll.cpp`**:`dll.cpp:6-12` 的 `DataSet` 成员顺序是 `CommandData Cmd; Archive Arc; CmdExtract Extract;`,
|
||||
> 构造时 `Cmd` 先完成 → `RAROptions::Init()`(`options.cpp:22-24`)在 `RAR_SMP` 下执行
|
||||
> `Threads=GetNumberOfThreads()` → 随后 `CmdExtract Extract(&Cmd)` 构造函数里
|
||||
> `Unp->SetThreads(Cmd->Threads)`(`extract.cpp:25-27`;上限 `Min(Threads,8)`,`unpack.cpp:65-71`)。
|
||||
> ⇒ **宏一开,RARDLL 路径自动拿到 MT**(这正是 CLI 与 DLL 共用的那条链路)。
|
||||
> 内存代价:MT 下 `UnpackThreadData` × `MaxUserThreads*2`(每个 `Decoded` 预分配 0x4100 项)
|
||||
> + `ReadBufMT` 4 MiB ≈ **15 MB 量级**,PS5 上可忽略。
|
||||
>
|
||||
> **⑤ 但是:新证据把矛头指向「写路径 / 存储」,而不是解码 —— 先别改代码。**
|
||||
> - 同形状**单线程**解码在 PC 上是 **386–505 MB/s(解压后口径)**;PS5 的 Zen 2 单核即使按 1/3 算也有 **~130 MB/s**。
|
||||
> - 真机只算 `.exfat` 一项就是 **19.49 GB ÷ 660 s = ≥29.5 MB/s 解压后**(且与 UI 的 10–40 吻合)。
|
||||
> 若 parts 1+2 的 8 GB 打包数据解开后还有十几 GB,那么全流水线就是 **30–70 MB/s 解压后**,
|
||||
> 比 CPU 能力低 **3–10×**。
|
||||
> - **两次独立操作撞同一个数**:上传(写 11.6 GB)实测 30–40 MB/s;解压(写 ≥19.5 GB)≈30 MB/s。
|
||||
> ⇒ 优先怀疑 **PS5 这条写路径的上限就在 30–40 MB/s**(内置盘 / 外置盘 / 目标目录待确认)。
|
||||
> - ⇒ **决策顺序**:先做 `T_copy`(零改动、纯搬运)。纯搬运也 ~10 分钟 ⇒ 收工,MT 不必做
|
||||
> (做了也会被 I/O 吃掉);纯搬运明显快 ⇒ 再上 MT,那 2.4× 才是真金白银。
|
||||
>
|
||||
> 复现脚本:`.build/rabtest/ab_rar_mt.py`(第一版,形状错误,留作反例)、
|
||||
> `.build/rabtest/ab_rar_mt2.py`(代表性版)。夹具留在 `D:\_wfm_rabtest{,2}\`(≈1 GB)。
|
||||
|
||||
> ### ⚠️ 2026-09-23 晚 更正:**「瓶颈不在解码」这个结论已撤回**
|
||||
>
|
||||
> 本节最初写它时只知道「18 GB / 11 分钟 / 1252 条目」,**不知道归档格式**。随后的补充
|
||||
> (**格式是 RAR**;包在 PC 上、经插件上传进 PS5;上传速度 30–40 MB/s)把两个前提都改了:
|
||||
>
|
||||
> **① 参照物错了 —— 这是方法错误,不是估计偏差。**
|
||||
> 原文拿「PS5 上解 **RAR** 的 28 MiB/s」去比「PC 上解 **7z** 的 427 MiB/s」。**不同格式、
|
||||
> 不同解码器、不同机器**:RAR 我们直接用 rarlab 的 UnRAR 库,7z 走自建 chain,两者毫无
|
||||
> 可比性。⇒ 原文「推论二(量级差 3–10×)」**不成立**,不能作为解码无罪的证据。
|
||||
> 「推论一(摆动)」此前已自行降级为弱证据(250 ms 采样噪声)。**两条都没了。**
|
||||
>
|
||||
> **② 查代码查出一个具体缺口:RAR 解码在我们这里是单线程的。**
|
||||
> `third_party/unrar7/os.hpp:43-45` 的 `#define RAR_SMP` 落在 `#ifdef _WIN_ALL` 分支**内**,
|
||||
> 所以 POSIX(PS5)构建**不定义** `RAR_SMP` —— 我们 Makefile 里 0 次出现;而官方 POSIX
|
||||
> makefile 第 11 行是 `DEFINES=... -DRAR_SMP`,**我们漏了这个开关**。后果:
|
||||
> - `unpack.cpp:185-198` 的 MT 分支整段不参与编译 ⇒ 永远走单线程 `Unpack5()`
|
||||
> - `unpack50mt.cpp`(`Unpack::Unpack5MT`)**不在我们的源列表里**;这是 rarlab 专门调过的
|
||||
> 多线程 RAR5 解压器(文件头注释:「0x400000 和 2 对 i9-12900K 最优」)
|
||||
> - `Unpack::SetThreads()` / `ThreadPool` 随之消失
|
||||
> - ⚠️ **我们在 ELF 上做的符号核查是无效的**:该 ELF 只有 `.dynsym`(513 项)、**没有
|
||||
> `.symtab`**,任何内部符号都查不到(零命中是假象)。上面的结论来自 Makefile 与 `os.hpp`。
|
||||
>
|
||||
> **③ 新的首要假设:28 MiB/s ≈ 单线程 RAR5 解码的典型量级。**
|
||||
> RAR5 `-m5` 单线程在现代桌面 CPU 上约 40–80 MB/s 输出,PS5 的 Zen 2 单核更低。
|
||||
> 另一个角度:`18 GB ÷ 660 s = 27.9 MB/s` 是**解码输入**速率,而上传实测证明**写入端**
|
||||
> 至少能到 30–40 MB/s、**读取通常快于写入** ⇒ 「纯存储上限」解释不了这个数。
|
||||
>
|
||||
> **④ 附带作废一条**:「读粒度只有复制路径 1/32–1/64」是 **7z 的数字**(`SZ_IN_CHUNK`
|
||||
> 256 KiB),对 RAR 不适用 —— RAR 走 `od.ArcName` 按路径打开(`src/rar_extract.c:1059`),
|
||||
> 归档 I/O 由 UnRAR 自己的 `File` 类完成,我们的回调只收到**解压后的数据**,
|
||||
> **插不进 read-ahead**。⇒ 待办 #48 对 RAR 无效。
|
||||
>
|
||||
> **⑤ 归档真实参数(口径已闭合)** —— 两个来源合起来读得通:WinRAR 信息页(待解压版本 5.0、
|
||||
> 加密「缺少」、无恢复记录、压缩文件锁定「存在」、**字典 4 GB**、压缩率 19%、
|
||||
> **总大小 19,493,028,232 B = 18.15 GiB**、打包大小 3,831,296,089 B、**总文件 3**)+ 文件列表
|
||||
> (三卷 `.part1/2/3.rar` = 4 GB + 4 GB + 3.6 GB ≈ 11.6 GB)。
|
||||
> **两者不矛盾**:信息页是把 **part3 当独立归档**打开的 —— part3 = 3.568 GiB,里面就是那 3 个条目
|
||||
> (一个 19.5 GB 的 `PPSA16608.exfat` + 两个 KB 级小文件),上面的「第二次更正」块已用
|
||||
> `UnRAR lt` 直接读头证实;而三卷合计 ≈11.6 GB 才是整个包(另外还有 ~1250 个条目)。
|
||||
> ⚠️ 我先前按"两图互相矛盾、需用户确认"写的那一版**作废**:不是两个归档,是"单卷视图 vs 整包视图"。
|
||||
> 速度口径不受影响:**18.15 GiB ÷ 660 s = 29.5 MB/s(解压后字节)**;就那个大条目而言
|
||||
> 读侧只需 3.83 GB ÷ 660 s ≈ **5.8 MB/s** ⇒ **读侧不是瓶颈,29.5 MB/s 是解码+写盘的真实速率**。
|
||||
>
|
||||
> ## ⚑ 终局:RAR5 多线程**不做**(2026-09-23 18:15,用户决定)
|
||||
>
|
||||
> **#51 结案:保持单线程现状,生产代码不动、不刷机。**
|
||||
>
|
||||
> **为什么不做的依据是"判不了",不是"没收益"** —— 收益本身已实测(块 ⑧:我们的引擎
|
||||
> 1.40–1.74×;rarlab CLI 在用户那种形状上 2.45×)。卡住的是**它能不能兑现**:
|
||||
> - `T_copy` 没做 ⇒ 无法区分「解码慢」与「写路径上限 ≈30 MB/s」。
|
||||
> - 而 MT **只并行解码**:worker 只跑 `UnpackDecodeThread`(`unpack50mt.cpp:190`),
|
||||
> **写盘恒为主线程串行**(`UnpWriteBuf()` 仅由主线程调用 —— `unpack50mt.cpp:283/475/587`)。
|
||||
> ⇒ 若墙在写路径,MT 的收益直接退化成 **1.0×**。
|
||||
> - 成本收益:判定要人上手测一次复制;上线要刷机 + 重跑 11 分钟。而收益可能为 0
|
||||
> ⇒ **不做**。维持 11 分钟,把不确定性留在文档里,比赌一次更划算。
|
||||
>
|
||||
> **下面是已完成的技术取证,全部保留** —— 将来若要重开,它就是现成答案。
|
||||
> ⚠️ **重开的第一个动作是 `T_copy`,不是改构建**(判读见 `docs/REAL-CONSOLE-PROFILE.md` 第 1 步)。
|
||||
|
||||
> **⑥ 开 RAR5 多线程只需一个编译开关**(把 #51 的工作量从"未知"降到"一行"):
|
||||
> - `third_party/unrar7/unpack.cpp:7-9` **已经** `#include "unpack50mt.cpp"`;
|
||||
> `threadpool.cpp:5` **已经** `#include "threadmisc.cpp"`。⇒ 官方 POSIX makefile 不列这两个
|
||||
> `.o` 是**正常的**,**不需要新增源文件**(此前"官方 makefile 自相矛盾"的疑点已消除)。
|
||||
> - `raros.hpp:23-25`:非 Windows 一律 `#define _UNIX` ⇒ PS5 构建自动走 Unix 分支;
|
||||
> `threadpool.cpp` 的 `_UNIX` 路径用 pthread cond/mutex,**不用 `sem_t`**。
|
||||
> - 线程数在 DLL 模式下**自动接上**:`dll.cpp` 每个归档持有一个 `CommandData`,
|
||||
> 构造函数 `cmddata.cpp:6-9 → Init() → RAROptions::Init()` →
|
||||
> `options.cpp:22-24 Threads=GetNumberOfThreads()`;`extract.cpp:25-27` 再
|
||||
> `Unp->SetThreads(Cmd->Threads)` → `unpack.cpp:65-71 MaxUserThreads=Min(Threads,8)`。
|
||||
> ⇒ **不需要 CLI 开关、不需要 vendor 补丁**。
|
||||
> - ⇒ 改动 = 给 unrar 对象加 `-DRAR_SMP` + 链接 pthread。**PS5 侧唯一未验证点是
|
||||
> `sysconf(_SC_NPROCESSORS_ONLN)` 是否返回真核数**(`threadmisc.cpp:111-124`):
|
||||
> SDK 的 `unistd.h:291` 定义了 `_SC_NPROCESSORS_ONLN 58`、`pthread_*` 在
|
||||
> `target/include/pthread.h:198-237` 齐全;但若 `sysconf` 返回 1,**MT 会静默失效**。
|
||||
> ⚠️ 注意 `threadmisc.cpp:116-124` 在 `_UNIX` 且未定义 `_SC_NPROCESSORS_ONLN` 时**没有
|
||||
> return 语句**(UB)—— 真机上要能看到核数才算数。
|
||||
>
|
||||
> **⑦ MT 对固实归档同样生效,唯一例外是分片窗口**:`unpack.cpp:185-198` 在
|
||||
> `MaxUserThreads>1` 时调 `Unpack5MT(Solid)` —— 形参本身就带 `Solid`。会把它挡在外面的只有
|
||||
> `Fragmented`(`unpack.cpp:193`):`unpack.cpp:130-145` 只在 4 GiB 窗口的**连续分配失败**
|
||||
> 且 `WinSize>=16 MiB` 且 64 位时才置位,而且分片窗口路径**本身更慢**。
|
||||
> 64 位下单次 malloc 失败通常是"量不够"而非"地址空间碎",此时 `FragWindow.Init(同一大小)`
|
||||
> 也会失败 ⇒ **分片窗口属于罕见回退,概率低**。但它是可判读的:
|
||||
> **真机 A/B 若"开了 MT 却一点没变",第一嫌疑就是它或 `sysconf`。**
|
||||
>
|
||||
> **⑧ 宿主 A/B 实测:MT 值 1.4–1.75×**(同一台机器、同一份二进制,只差一个 `-DZIPSFX`)。
|
||||
> 方法:MinGW 下 `_WIN32 ⇒ `_WIN_ALL` ⇒ `os.hpp:43` 自动定义 `RAR_SMP`,所以**我们过去所有
|
||||
> 宿主基准跑的都是多线程路径**;反过来单线程基线只能靠 `-DZIPSFX`(该宏在整棵源码树里
|
||||
> **只出现一次**,就是 `os.hpp:43`,干净可用)。回执:`nm` 查 `unpack.o`,base 的
|
||||
> `Unpack5MT` 符号数 = 0、mt = 1。样本:341 MiB 现实混合数据(127 MiB 真实二进制
|
||||
> + 158 MiB 短匹配文本 + 48 MiB 随机),RAR5 固实 `-m3`,压缩后 125–153 MB(比率 37–45%):
|
||||
>
|
||||
> | 样本 | base(单线程) | mt | 加速 |
|
||||
> |---|---:|---:|---:|
|
||||
> | 固实 `-md1m` | 69.7 MiB/s | 97.3 MiB/s | **1.40×** |
|
||||
> | 固实 `-md256m` | 63.4 MiB/s | 110.6 MiB/s | **1.74×** |
|
||||
> | 分卷 `-md256m`(96+23 MiB) | 64.2 MiB/s | 100.8 MiB/s | **1.57×** |
|
||||
>
|
||||
> 两点附带信息:①**字典越大 MT 越划算** —— 单线程随字典从 1 MiB 涨到 256 MiB 掉到
|
||||
> 63–70 MiB/s(内存局部性),而 MT 稳定在 97–110;用户的包字典 4 GB,比这里最大的样本
|
||||
> 还大 16×,**方向上有理由期望 MT 收益不小于 1.6×**。②**跨机器外推不算结论**:
|
||||
> 宿主单线程 63–70 MiB/s vs PS5 的 28.1 MiB/s 是不同 CPU,只作量级参考。
|
||||
> ⇒ 若 PS5 上同样拿到 1.5–1.75×,11 分钟 → **约 6.3–7.3 分钟**。
|
||||
>
|
||||
> **⑧b 与上面「第二次更正」块的 2.45× 不矛盾 —— 差在样本形状,不在实现。**
|
||||
> 那块用 rarlab 自带 **UnRAR 7.23 CLI** 的 `-mt1` vs `-mt8`,夹具是 `-m1 -md4g`、**5.27:1**
|
||||
> 可压缩的 2 GB /16 文件;我这边是 `-m3`、只有 **2.4:1** 的短匹配数据。方向一致:
|
||||
> **数据越可压缩(匹配越长)MT 越划算** —— 长匹配让一条符号吐出更多字节,串行 apply 被摊薄、
|
||||
> 并行解码占比上升,同时更容易越过 `unpack50mt.cpp:150-152` 的 `LargeBlockSize=0x20000` 退化阈值。
|
||||
> ⇒ **对用户这个包应以 ~2.4× 为预期**(它是 `-m1 -md4g`、5.09:1,正落在那个夹具的形状上),
|
||||
> 我测到的 1.4–1.75× 作为**更难数据下的下界**。绝对吞吐那 10× 的差距(386 MB/s vs 63–70 MiB/s)
|
||||
> **纯粹是夹具可压缩性差异,不能拿来比较两个实现**(这正是"基准代表性"那类错误)。
|
||||
> 综合估计:11 分钟 → **约 4.6 分钟**;悲观情形(1.5×)→ 6.3–7.3 分钟。
|
||||
>
|
||||
> **⑨ 样本代表性教训(第一版 A/B 是废的)**:初版样本是「同一个 60 KiB 区块重复 2400 次」,
|
||||
> 压缩到 0.2%、解码 **602 MiB/s** —— 长匹配让范围解码器的每条符号吐出上千字节,
|
||||
> 这个形状**真实归档里不存在**,而且它恰恰是 MT 最不擅长的形状(串行 apply 占主导)。
|
||||
> 后改用"真实二进制 + 短匹配文本 + 随机"混合,比率 37–45%,吞吐落到 63–110 MiB/s 的正常带。
|
||||
> 另:初版把 `-v96m` 的目标名写成 `vol.part1.rar`,rar 会翻倍成 `vol.part1.part1.rar`
|
||||
> (`tests/make-rar-fixtures.bat` 里已记过这个坑,我复现了一遍)。
|
||||
>
|
||||
> **⑩ 顺带查出两个真实缺陷/隐患**:
|
||||
> 1. **字典 > 4 GiB 的归档会直接失败** —— **已修,但修的是"说清楚",不是"放行"**。
|
||||
> 机制:`extract.cpp:1748-1767 CheckWinLimit()` 在 `WinSize > Cmd->WinSizeLimit` 时调
|
||||
> `uiDictLimit()`;DLL/silent 构建里 `uisilent.cpp:65-73` 只在
|
||||
> `Cmd->Callback(UCM_LARGEDICT, …) == 1` 时才放行,否则 `DllError=ERAR_LARGE_DICT` 并跳过
|
||||
> 该文件。我们的回调原先**只处理 `UCM_PROCESSDATA`**,其余一律返回 0。默认
|
||||
> `Cmd->WinSize`/`WinSizeLimit` = `0x2000000` / `0x100000000`(`options.cpp:12-13`)
|
||||
> ⇒ 分界线正好是 4 GiB(比较是 `<=`)。
|
||||
>
|
||||
> **⚠️ 三条订正(2026-09-23 18:40,把先前"1 行可修"的判断推翻):**
|
||||
> - **RAR5 根本到不了这里。** `arcread.cpp:871` 把 RAR5 字典读成
|
||||
> `0x20000 << ((CompInfo>>10) & 0x0f)` —— **只有 4 bit** ⇒ 格式自身上限 = `0x20000<<15`
|
||||
> = **正好 4 GiB**,与我们的 limit 相等。⇒ 任何 `-ma5`(含用户这个包)**永远不触发**。
|
||||
> 只有 **RAR7 头**(`UnpVer==1`,5 bit,上限 `UNPACK_MAX_DICT` = 64 GiB)才可能超。
|
||||
> - **而 RAR7 造不出来。** 实测 `Rar.exe 7.23`:`-ma4` / `-ma6` / `-ma7` **全部 exit 7**
|
||||
> (命令行错误),只有 `-ma5` 可用。⇒ 本机连验证样本都得手工合成。
|
||||
> - **"放行"是陷阱不是修复。** 放行后 unrar 会去 `new` 一个**完整的字典窗口**;rarlab 自己的
|
||||
> CLI 对这个样本的答复是:「8 GB 字典超过 4 GB 限制,而且需要大于 8 GB 内存来解压缩。
|
||||
> 使用 `-md8g` 或 `-mdx8g` 参数来解压缩。」PS5 只有 16 GB **共享**内存 ⇒ >4 GiB 字典在
|
||||
> 该设备上本就解不动;**中途被 OOM 杀掉(整个 payload/UI 一起没)比干净失败更糟**。
|
||||
> ⇒ 决定:**保持拒绝**,与上游 CLI 默认一致。
|
||||
>
|
||||
> **实际改动(在生产代码里,2026-09-23):** 拒绝时不再把锅甩给条目。原先
|
||||
> `ERAR_LARGE_DICT → ZIPX_ERR_LIMIT_FILE` → i18n `extract_entry_too_large`
|
||||
> ⇒ 用户看到的是「**压缩包内单个文件过大: hello.txt**」,而那个条目只有 7 KB —— 完全错。
|
||||
> 现在新增 `ZIPX_ERR_LIMIT_DICT`(`src/zip_extract.h`)+ `extract_dict_too_large`
|
||||
> (中/英),并在 `rar_data_cb` 里从 `UCM_LARGEDICT` 的 `p1/p2` 取回真实数字,
|
||||
> 报成「需要 8192 MiB(上限 4096 MiB)」。回归用例 `tests/test_rar_extract.c:test_dict_limit`
|
||||
> + 固定样本 `tests/fixtures/dict-8g.rar`(合成器 **`tests/make_fixtures.py:bigdict()`**,
|
||||
> 纯 Python 手写最小合法 RAR5 归档、不依赖任何压缩器;说明见该函数 docstring)。
|
||||
> ⚠️ **样本必须由 `make_fixtures.py` 生成** —— `fresh()` 会 `shutil.rmtree()` 整个
|
||||
> `tests/fixtures/`,提交进去的二进制会被抹掉,所以不能单独放一个生成脚本。
|
||||
> 2. **归档里的目录条目 +「覆盖」策略 = 第二次解压到同一目录必失败**。
|
||||
> `src/rar_extract.c:884-892`:目标已存在且是目录、而归档条目也是目录时,
|
||||
> 只有 `ZIPX_CONFLICT_MERGE` 能过;`OVERWRITE` 报 `ZIPX_ERR_CONFLICT`
|
||||
> (状态串是 "target already exists",`detail` 才是 "directory already exists")。
|
||||
> `zip_extract.c:1123` / `sevenz_extract.c:1526` 同构。⇒ 用户对含目录的包用「覆盖」
|
||||
> 解两次,第二次会失败。**这条是我在搭 A/B 时被挡了才知道的**,需要确认是否设计意图。
|
||||
>
|
||||
> 下文数据与推论**保留原文**(它们是当时判断的依据),但**结论以本块为准**。
|
||||
|
||||
### 数据
|
||||
用户在真机上解一个 **18 GB 的包**,UI 上报的解压速度在 **10–40 MB/s** 之间摆动;
|
||||
随后补上两个关键数字:**总耗时 11 分钟(660 s)**、**条目数 1252**(平均 14.7 MB/条目);
|
||||
再确认 **18 GB 是压缩包自身的大小**(解压后多大未知)。
|
||||
|
||||
⇒ **平均吞吐 ≥ 28 MiB/s**(18 GB = 18 432 MiB ÷ 660 s;解压后更大则更高,故为下限)。
|
||||
**这个数字才是基线**,UI 上那个 10–40 的区间只是瞬时值。
|
||||
|
||||
「18 GB = 压缩包大小」顺带给出一个**与解码无关的硬上界**:源盘必须在 660 s 内交出
|
||||
18 GB 归档数据 ⇒ 整条流水线的平均吞吐上界就 ≈ 28 MiB/s。**无论解码多快,源盘只有这个交付速度。**
|
||||
这也是为什么第 1 步的 `T_copy`(同一个 18 GB 文件的纯搬运)能与 660 s 直接比大小。
|
||||
|
||||
顺带排除一项:**per-entry(小文件)开销不是主因** —— 1252 个文件、平均 14.7 MB,不是
|
||||
"几万个小文件"那种形态,建文件 + rename 分摊到 0.53 s/文件里微乎其微。
|
||||
|
||||
口径先确认(不是猜):进度条的"字节"是**解压后的字节**——
|
||||
`src/zip_extract.c:651-678` 把 `bytes_total` 累加自 `info->uncompressed_size`;
|
||||
`:875` 的 `mz_zip_entry_read()` 返回解压字节,`:900/910` 用它累加 `bytes_done`。
|
||||
所以 10–40 MB/s 是**吐出数据的速度**,正是用户关心的那个口径。
|
||||
|
||||
### 推论一:摆动说明"负载不恒定",但它是弱证据
|
||||
solid 块(同一字典、同一条码流)的解码速率**几乎是恒定的**。要出现 4× 的摆动,
|
||||
更像是 I/O 侧在变:源盘读取、目标盘写入、逐条目同步开销、或存储设备自身在忙。
|
||||
|
||||
但这条**不能单独定案** —— 那个 MB/s 是 250 ms 窗口 + 1 MiB 上报阈值的**瞬时值**
|
||||
(`src/zip_extract.c:123`、`src/task.c:202-206`),写缓冲突发本身就能在窗口里造成大幅跳动。
|
||||
它能说的只有"负载不恒定",不构成"解码无罪"的证明。有解释力的是推论二(量级)和推论三(粒度)。
|
||||
|
||||
### 推论二:量级上差 3–10×,解码没有解释力
|
||||
| | 吞吐 |
|
||||
|---|---:|
|
||||
| PC / WSL,329 MiB 混合数据,8 线程(本仓实测) | **427 MiB/s** |
|
||||
| PS5 单线程解码的乐观上界(按核数×频率外推,**未实测**) | ~100 MB/s |
|
||||
| **真机实测(18 GB 包,端到端)** | **10–40 MB/s** |
|
||||
|
||||
18 GB @ 10–40 MB/s = **7.7 ~ 31 分钟**;同样数据在 PC 上纯解码约 43 s。
|
||||
⇒ 解码最多占 10–25%,**很可能远低于此**。
|
||||
|
||||
### 推论三:三个引擎的读请求粒度都只有复制路径的 1/32–1/64
|
||||
粒度审计(已核对源码):
|
||||
|
||||
| 路径 | 源侧读粒度 | 目标侧写粒度 |
|
||||
|---|---|---|
|
||||
| 复制(`copy_file_pipeline`,≥256 MiB) | **8 MiB** × 3 slot,4096 对齐,独立读线程 | 8 MiB |
|
||||
| 7z | chain `SZ_IN_CHUNK = 256 KiB`(`src/sevenz_chain.c:87`);MT 路径 `inBufSize_MT = 1 MiB` | `SZ_OUT_CHUNK = 64 KiB`(`:94`) |
|
||||
| ZIP | `ZIPX_IO_BUFFER = 128 KiB`(`src/zip_extract.c:32`) | 128 KiB |
|
||||
| RAR | UnRAR `File::CopyBufferSize() = 4 MiB`(`third_party/unrar7/file.hpp:148-153`) | 4 MiB |
|
||||
|
||||
都不是 4 KB 那种「小读」灾难,但**都比复制小 32–64 倍**。在延迟主导的设备上,
|
||||
吞吐 ≈ 单次请求大小 ÷ 每次请求的等效延迟:
|
||||
|
||||
```
|
||||
256 KiB / 10 ms = 25 MB/s ← 正好落在实测 10–40 MB/s 的中间
|
||||
8 MiB / 10 ms = 800 MB/s ← 复制路径不会撞这个上限
|
||||
```
|
||||
|
||||
若源与目标在**同一块盘**(例如外置 USB HDD 上解压到同一块盘),读流与写流并存,
|
||||
磁头来回跑、read-ahead 被写回刷打断 → 每次请求退化成一次寻道,上面这个算术即成立。
|
||||
**这是目前唯一可疑的代码级病因**,而它的改动面极小:所有 7z 的读都只经过
|
||||
`src/sevenz_volstream.c` 的 `vol_read()` 一个函数(ZIP/RAR 同理在 `src/zipx_volstream.c`)。
|
||||
|
||||
### ~~因此:剩余解码优化项全部搁置~~(**已撤回,见文首更正块**)
|
||||
原文在这一段把 CRC 硬件化、MT 扩到 BCJ2 / 多 folder、ZIP inflate 换 libdeflate、AES-NI 全部判为
|
||||
「在 10–40 MB/s 的现实面前没有意义」。**建立在「解码不是瓶颈」之上,而那个前提已撤回。**
|
||||
|
||||
更正后的分层判断(按真实工作负载 **RAR** 重排):
|
||||
|
||||
| 项 | 更正后的判断 |
|
||||
|---|---|
|
||||
| **UnRAR RAR5 多线程解压(`RAR_SMP`)** | **❌ 不做(2026-09-23 用户决定,见文首「终局」块)**。收益已实测(rarlab CLI `-mt1` vs `-mt8` **2.45×**,见「第二次更正」块 ③;我们引擎在更难数据上 1.40–1.74×,块 ⑧b),接线也只是 **1 个编译开关 + 1 个链接开关、零源码改动**(块 ④)—— **但 `T_copy` 没做,无法排除「写路径上限 30–40 MB/s」**;MT 只并行解码、写盘恒为主线程串行(`unpack50mt.cpp:190` / `:283,475,587`)⇒ 收益可能是 1.0×。刷机 + 重跑 11 分钟的成本压在不确定收益上 ⇒ 搁置 |
|
||||
| CRC 硬件化(4–5%) | 仅对 7z / ZIP 有意义;**对 RAR 完全无关**(CRC 由 UnRAR 自己算) |
|
||||
| MT 扩到 BCJ2 / 多 folder | 仅 7z;RAR 工作负载下无意义 |
|
||||
| ZIP inflate 换 libdeflate / AES-NI | 同上,与 RAR 无关 |
|
||||
| 读粒度 / read-ahead(#48) | **对 RAR 无效**(UnRAR 按路径自读,回调只收解压后数据);只对 7z/ZIP 有意义 |
|
||||
|
||||
「先 profile 再排序」这个结论**仍然成立**,但目标从「查清为什么只有 10–40 MB/s」变成
|
||||
**「先把 RAR 单线程这条确认掉 / 排除掉」** —— 见 `docs/REAL-CONSOLE-PROFILE.md`。
|
||||
|
||||
### 尚未定案的部分
|
||||
- ~~**归档格式未知**~~ → **已确认(2026-09-23):格式是 RAR**,包原本在 PC 上、经插件上传进
|
||||
PS5(上传速度 30–40 MB/s)。⇒ 解码用的是 rarlab 的 UnRAR 库本身,**我们自己的代码在 RAR
|
||||
解码路径上只剩一层很薄的 facade**;「我们的解码器慢」这个方向基本不成立,
|
||||
但「**我们的构建没打开 UnRAR 的多线程**」成立(见文首更正块 ②)。
|
||||
仍未确认:**RAR4 还是 RAR5**(`Unpack5MT` 只对 RAR5/7.0 生效)、是否固实、是否加密、
|
||||
**解压后多大**(决定输出吞吐)。
|
||||
- ~~**18 GB 是压缩后还是解压后**~~ → **已确认(2026-09-23):18 GB 是压缩包自身的大小**。
|
||||
于是有了一个**不需要知道解压后大小的硬上界**:源盘要在 660 s 内交出 18 GB 归档数据
|
||||
⇒ 整条流水线平均吞吐上界 ≈ **28 MiB/s**;`T_copy` 与 660 s 可直接比大小。
|
||||
- **存储未知**:包在哪个设备上(内置 SSD / 外置 USB / 同盘还是异盘)、输出写到哪个设备。
|
||||
- ~~**下一步(决定性、零改动)**~~ → **未执行(2026-09-23 用户决定搁置,见文首「终局」块)**:
|
||||
用插件自己的**复制**功能把那个包搬到解压输出所在的那块盘上,记耗时。复制的
|
||||
`copy_file_pipeline` 是 8 MiB × 3 slot 的双缓冲搬运,⇒ 它就是"同一台机器、同一对设备、
|
||||
只把解压换掉"的对照。**将来重开时这是第一个动作**,判读见 `docs/REAL-CONSOLE-PROFILE.md` 第 1 步。
|
||||
|
||||
> 一句话:**"我们比 7-Zip 慢 1.59×" 这个议题在真机上不成立** —— 两边都被同一个 I/O
|
||||
> 上限压着,谁先撞墙取决于存储,而不是解码器。真机目标从"提速解码"改成"查清 28 MiB/s"。
|
||||
> ~~目前指向一个具体、可改的位置:**读请求粒度只有复制路径的 1/32–1/64**(推论三),
|
||||
> 而全部 7z 读都经过 `vol_read()` 一个函数。**先用复制做基线验证它,再决定改不改。**~~
|
||||
> → **推论三对 RAR 不适用**(那是 7z 的 256 KiB 数字;UnRAR 按路径自读,见文首更正块 ④)。
|
||||
> 更正后:真机工作负载是 RAR,而**我们的 RAR 解码是单线程的**(`RAR_SMP` 未定义 ⇒
|
||||
> `unpack50mt.cpp` 不参与编译)—— 这是一项有据可查、且直接针对真实负载的改进点,
|
||||
> 在真实形状上**已实测 2.4×**。
|
||||
> ~~**但优先级已被文首二次更正块 ⑤ 改写**:先做零改动的 `T_copy`,确认墙在解码还是在写路径。~~
|
||||
> → **闭环(2026-09-23 18:15):`T_copy` 没做,用户决定不做了** ⇒ 见文首「终局」块。
|
||||
> 单线程现状保留,此项转入"已评估、搁置(重开先测 `T_copy`)"。
|
||||
|
||||
---
|
||||
|
||||
## 七、复现
|
||||
|
||||
```bash
|
||||
export PATH="/c/mingw64/bin:/c/Users/songl/.workbuddy/binaries/PortableGit/versions/1.2.0/mingw64/bin:/c/Users/songl/.workbuddy/binaries/python/versions/3.13.12:/usr/bin:/bin:/c/Windows/System32:/c/Windows"
|
||||
cd "/c/Users/songl/Desktop/Web File Manager/ps5-web-file-manager"
|
||||
|
||||
# 先各跑一次,生成引擎对象
|
||||
/usr/bin/bash tests/run-sevenz-tests.sh
|
||||
/usr/bin/bash tests/run-tests.sh --rebuild
|
||||
|
||||
# Windows 基准(我们的引擎 vs 7za.exe)
|
||||
python tests/bench_driver.py --big --runs 3
|
||||
|
||||
# 同环境对比(WSL):官方 Linux 7-Zip / 我们的引擎 / 编译选项 / profile
|
||||
wsl.exe -d Ubuntu-22.04 -- bash < .build/bench/wsl-perf.sh
|
||||
wsl.exe -d Ubuntu-22.04 -- bash < .build/bench/wsl-ours.sh
|
||||
wsl.exe -d Ubuntu-22.04 -- bash < .build/bench/wsl-flags.sh
|
||||
wsl.exe -d Ubuntu-22.04 -- bash < .build/bench/wsl-prof.sh
|
||||
```
|
||||
|
||||
## 附:测量方法备忘
|
||||
|
||||
这个沙箱里 bash 计时不可用,三个坑都先给出过错误数字:
|
||||
|
||||
- 每次 `date` 要 ~350 ms → `t0/t1` 对给 ~600 ms 的测量注入 700 ms 误差
|
||||
- `time` 的 user/sys 看不到 native 子进程(解 82 MiB 报 `user 0.031s`)
|
||||
- 反复跑堆积 2.7 GB 输出目录 → 测出过"内部时间 > 外部时间"
|
||||
|
||||
正确做法:Python 驱动(单次 spawn + 单调时钟)+ 候选程序自带内部计时 + 显式扣除 spawn tax(~190–240 ms)+ best of N + 每次跑前清输出目录。
|
||||
@@ -0,0 +1,143 @@
|
||||
# PS5 网页文件管理器 v1.9.2(自编译版)发布 + 与上游原版对比 + 使用简介
|
||||
|
||||
> 这是我自己维护的一个分支版本,基于 owendswang 的 `ps5-web-file-manager` 二次开发。本文讲三件事:**它和上游原版差在哪、我这个版本强在哪、怎么用**。
|
||||
|
||||
---
|
||||
|
||||
## 一、它是什么
|
||||
|
||||
一个在已越狱 PS5 上跑的 HTTP 文件管理器(单文件 ELF 载荷)。局域网内任意浏览器(含 PS5 自带浏览器)打开 `http://<PS5的IP>:8888/` 就能管理外接 USB 和内置存储:浏览、改权限、上传、下载、原地编辑文本、装/预览 PKG、看图,以及**解压 ZIP / RAR / 7z**。
|
||||
|
||||
- 版本:**v1.9.2**
|
||||
- 标题 ID:`FMGR88888`
|
||||
- 许可证:GPLv3+
|
||||
- 目标平台:`x86_64-sie-ps5`(Zen 2,不是 ARM)
|
||||
|
||||
> **关于版本号**:v1.9.2 与 v1.9.1 功能完全相同,只更新了内嵌版本号。原先的 `v1.9.1` tag 打在了产出发布二进制的提交**之前 4 个提交**,导致"tag 对应的源码"重建不出发布的那份 ELF;v1.9.2 重新从正确提交上打,使 **tag = 源码 = 二进制**。
|
||||
|
||||
---
|
||||
|
||||
## 二、和上游原版(owendswang v1.8)对比
|
||||
|
||||
先说结论:**不是同一条路线,各有胜负手。**
|
||||
|
||||
| 维度 | 上游 owendswang v1.8 | 我的 v1.9.2 |
|
||||
|---|---|---|
|
||||
| 一句话定位 | 把 7-Zip 本体做成**外部 helper 进程**,靠 IPC 调用 | **自研 / vendor 解码库,全部内嵌同一进程** |
|
||||
| 支持的压缩格式 | **30 种**(zip/7z/rar/tar/gz/xz/zst/bz2/cab/arj…) | **3 种**:ZIP / RAR / 7z |
|
||||
| 部署方式 | **两个文件**,helper 必须放 `/data/wfm/` 指定路径 | **单个 ELF,零外部依赖** |
|
||||
| 载荷体积 | 主程序 + 7-Zip 本体(两份) | **850 KiB 单文件**(含 ZIP/RAR/7z 三个解码引擎) |
|
||||
| 防压缩炸弹(zip bomb) | ❌ 无 | ✅ 压缩比上限 + 1 GiB 下限豁免 |
|
||||
| 磁盘写满保护 | ❌ 无 | ✅ 解压前按实际剩余空间 `statvfs` 预检 |
|
||||
| 路径穿越防护 | ❌ 无(交给 7-Zip) | ✅ 有专项测试 |
|
||||
| 解压中断留残留 | ⚠️ 可能留半成品 | ✅ staging 隔离,失败即清 |
|
||||
| 解压任务跨重启恢复 | ✅ helper 独立进程,重启不丢 | ❌ 暂无 |
|
||||
| 内存隔离 | ✅ 独立进程 | ❌ 与主程序共享地址空间 |
|
||||
| 错误信息详细度 | 中等 | ✅ 含条目名 / errno / 字节数 |
|
||||
| 解压核心正确性 | 7-Zip 本体(20 年验证) | 自研 7z 链 + 成熟 vendor 库 |
|
||||
|
||||
### 上游原版强在哪
|
||||
- **格式多**:30 种,常见游戏包/备份/Mod 里 `.tar.gz`、`.xz`、`.zst`、`.bz2` 都能直接解。
|
||||
- **任务恢复**:helper 是独立进程,主程序被系统杀掉或浏览器重开,大包解压不丢。
|
||||
- **内存隔离**:解压峰值不影响主文件服务。
|
||||
|
||||
### 我的版本强在哪
|
||||
- **安全护栏齐全**:压缩炸弹、写满磁盘、路径穿越、中断残留——这四项上游一个都没有,而我这边都有实现和测试(测试矩阵 163 项检查,0 失败)。说白了,**一个恶意压缩包不会把你的内置存储搞崩**。
|
||||
- **单文件部署**:丢一个 ELF 就行,不用记第二个文件该放哪;helper 丢了功能全废的事在我这不存在。
|
||||
- **7z 引擎是硬啃出来的**:上游靠 7-Zip 本体"白嫖",我这边是自己解析 7z folder + 拉式 codec 链,原生支持 BCJ2 反汇编后处理和多 coder 组合,还顺手做了多线程 LZMA2 解码(约 1.37×)和汇编 LZMA 解码器(约 1.26×)。
|
||||
- **RAR 用官方 UnRAR 7.20.1**:能解 WinRAR 6.x/7.x 写的 RAR5「v6」归档和多卷 RAR——上游原版的旧引擎在这类文件上会报"归档损坏"。
|
||||
|
||||
### 我不回避的短板
|
||||
- **格式覆盖只有 3 种**,这是最明显的弱项。`.tar.gz`、`.xz`、`.zst`、`.bz2` 你现在还得在 PC 上先解开。这是我接下来增量要补的方向(tar+zlib 已有、xz/lzma 复用 LZMA SDK、bz2/zst 可 vendor 单文件解码器),但**不打算照抄上游的 helper 路线**——那样会丢掉上面那些安全护栏和单文件部署优势。
|
||||
- **任务跨重启恢复**和**内存隔离**暂时没有(和"单进程内嵌"的架构取舍有关)。
|
||||
|
||||
> 一句话:**上游赢在"格式广度 + 进程架构",我赢在"安全 + 部署 + 错误质量"。**
|
||||
|
||||
---
|
||||
|
||||
## 三、v1.9.2 / v1.9.1 我做了哪些具体改进
|
||||
|
||||
- **7z 解压引擎**:自研解码子集 + 拉式 codec 链(`sevenz_chain.c`),支持 LZMA2 / LZMA / BCJ2,覆盖 SDK `SzArEx` 装不下的 5-coder 文件夹。
|
||||
- **分卷(多卷)支持**:三种格式都支持分卷,但命名约定不同(引擎按文件名自动识别,打开首个分卷即可):
|
||||
- **ZIP 分卷**:① 经典多磁盘 `name.z01 … name.zNN … name.zip`(索引目录固定在最后一个 `.zip`);② 7-Zip 字节分割 `name.zip.001 / .002 / …`;③ WinRAR 卷 `name.part1.zip … name.partN.zip`。最多 512 卷。
|
||||
- **7z 分卷**:7-Zip 字节分割 `name.7z.001 / .002 / …`,各分卷需等大,打开首个即可。最多 512 卷。
|
||||
- **RAR 分卷**:标准多卷 `name.part1.rar / .part2.rar / …`(unrar 自动按名合并)。**注意**:用 7-Zip 切出来的 `name.rar.001` 这种命名**暂不支持**,引擎会提示你先把分卷改名成 `.partN.rar` 再解。
|
||||
- **性能三项**(纯解码提速,不影响功能):
|
||||
- 汇编 LZMA 解码器 ≈ **1.26×**
|
||||
- 纯 LZMA2 多线程解码(8 线程)≈ **1.37×**
|
||||
- 移除 ZIP 逐条目 fsync,减少写盘开销
|
||||
- **RAR 升级到官方 UnRAR 7.20.1**:RAR5「v6」+ 多卷可用(v1.8 时代在 WinRAR 6/7 文件上报"归档损坏"的问题已消失)。
|
||||
- **载荷瘦了 15.8%**(1,034,328 → 870,488 字节):链接期去掉了 libc++abi 里一段**永远不会执行**的 C++ 名字还原器(只服务于"未捕获异常打印类型名"这条路径),再加上相同机器码折叠。**不损失任何功能,也不影响解压速度**。
|
||||
- **上传更顺手**:点上传按钮直接选文件或文件夹(不再弹二级菜单);支持**把文件/文件夹直接拖到页面上**上传。
|
||||
- **中英双语界面** + 项目主页右上角可切换语言说明。
|
||||
|
||||
### 已知缺口(诚实列出)
|
||||
|
||||
> 本节描述的是 **v1.9.2 发布时**的状态。此后有两条已在工作树中补齐、尚未发版:
|
||||
> 加密 ZIP/RAR 与 7z `-mhe=on` 加密头。详见 README「未发布内容」与
|
||||
> `HANDOVER.md` §十一 / §十二。
|
||||
|
||||
- 带密码的 ZIP / RAR / 7z:**拒绝解压**(引擎有解密能力,但密码输入 UI/API 还没接,临时先挡掉)。
|
||||
- 7z `-mhe=on` **加密头**:暂不支持(需要自研头解析器)。这是 7z 侧唯一已知缺口。
|
||||
|
||||
---
|
||||
|
||||
## 四、使用简介(三步上手)
|
||||
|
||||
### 1. 把 ELF 发到 PS5
|
||||
PS5 上先运行一个 ELF 加载器(端口 `9021` 常见),然后在 PC 上:
|
||||
|
||||
```sh
|
||||
# 把 PS5_IP 换成你主机实际 IP
|
||||
nc -q0 <PS5的IP> 9021 < web-file-mgr-v1.9.2.elf
|
||||
```
|
||||
|
||||
PS5 屏幕会弹通知,显示实际监听端口(默认 `8888`)。首次运行还会在主屏 Media 分类装一个「PS5 Web File Manager」快捷方式。
|
||||
|
||||
### 2. 浏览器打开
|
||||
同一局域网下,任意设备浏览器打开:
|
||||
|
||||
```
|
||||
http://<PS5的IP>:8888/
|
||||
```
|
||||
|
||||
PS5 自带浏览器也能开。
|
||||
|
||||
### 3. 日常操作
|
||||
- **浏览/管理**:列文件、排序、改权限、复制/移动/重命名/删除、原地编辑文本(≤1 MiB 的 txt/json/js/c/h…)。
|
||||
- **上传下载**:单文件或文件夹树上传(也可直接拖拽到页面);文件夹/多选以 `.tar` 流式下载。
|
||||
- **解压**:在文件列表里点 ZIP / RAR / 7z 的「解压」按钮即可。大档案(>480 GiB)会弹确认提示,确认后用更宽松的限额解压。
|
||||
- **冲突策略**:同名文件默认拒绝覆盖;需要时可选覆盖或合并。
|
||||
|
||||
---
|
||||
|
||||
## 五、下载与校验
|
||||
|
||||
> 请在 GitHub Release 页面下载:`https://github.com/lishersong/ps5-web-file-manager/releases/tag/v1.9.2`
|
||||
|
||||
发布文件:`web-file-mgr-v1.9.2.elf`
|
||||
|
||||
| 项目 | 值 |
|
||||
|---|---|
|
||||
| 大小 | 870,488 字节(约 850 KiB) |
|
||||
| sha256 | `177e90fecf93a0251e83f67884fba4551051be248330b0d70fda8ea732f88e84` |
|
||||
| 文件类型 | ELF 64-bit LSB,x86-64(e_machine `0x003e`,即 PS5 目标三元组) |
|
||||
|
||||
下载后建议先核对 sha256 再发到主机:
|
||||
|
||||
```sh
|
||||
sha256sum web-file-mgr-v1.9.2.elf
|
||||
# 应等于 177e90fecf93a0251e83f67884fba4551051be248330b0d70fda8ea732f88e84
|
||||
```
|
||||
|
||||
> 本次构建是**可复现**的:同一份源码树连编两次 sha256 完全相同。所以这个哈希可以放心当交付指纹用。
|
||||
|
||||
---
|
||||
|
||||
## 六、免责声明
|
||||
|
||||
非官方自制软件,仅在已越狱 PS5 上运行。使用风险自负——作者不对损坏、数据丢失、账号处罚或保修影响负责。请勿再分发 Sony 专有内容。依 GPLv3+,本分支修改后的源码已公开在上面的仓库。
|
||||
|
||||
---
|
||||
|
||||
*想继续补 `.tar.gz` / `.xz` 等格式、或接上密码解压 UI 的,欢迎在仓库提 issue。*
|
||||
@@ -0,0 +1,40 @@
|
||||
# PS5 Web File Manager — v1.8 开发方案(已废弃)
|
||||
|
||||
> **这份文件已不再维护,不要拿它当参考。**
|
||||
>
|
||||
> 原文写于 **2026-09-04**,是一份 **v1.8(RAR 支持)开发计划**,对应代码版本 `aef4a44`
|
||||
> (v1.7 时代)。其中大量结论已被后续版本推翻,例如它至今仍写着:
|
||||
>
|
||||
> - 「❌ RAR 分卷」「❌ 加密 RAR」
|
||||
> - 「dmc_unrar 不支持 / 已移除」
|
||||
> - 「7z 尚不支持」「加密解压尚未接线」
|
||||
>
|
||||
> 而 v1.9 已用 rarlab UnRAR 7.20.1 解决 RAR 分卷与加密,v1.9.2/v1.9.3 补齐了 ZIP/RAR/7z
|
||||
> 三格式加密内容与 7z `-mhe=on` 加密头,dmc_unrar 也已整体移除。
|
||||
|
||||
## 当前权威文档(按优先级)
|
||||
|
||||
| 文档 | 用途 |
|
||||
|---|---|
|
||||
| 仓库根 [`../HANDOVER.md`](../HANDOVER.md) | **状态速览、真机待验证清单、坑清单 —— 以此为唯一准绳** |
|
||||
| [`../README.md`](../README.md) / [`../README.zh-CN.md`](../README.zh-CN.md) | 功能范围与使用方式 |
|
||||
| [`../CHANGELOG.md`](../CHANGELOG.md) | 逐版本变更 |
|
||||
| [`EXTRACTION-PERF.md`](./EXTRACTION-PERF.md) | 解压性能实测与优化账目 |
|
||||
| [`REAL-CONSOLE-PROFILE.md`](./REAL-CONSOLE-PROFILE.md) | 真机性能验证清单 |
|
||||
| [`REWRITE-FEASIBILITY.md`](./REWRITE-FEASIBILITY.md) | 许可与拆库可行性分析 |
|
||||
| [`SIZE-OPTIMIZATION.md`](./SIZE-OPTIMIZATION.md) | 二进制体积优化记录 |
|
||||
| [`UPSTREAM-V1.8-COMPARISON.md`](./UPSTREAM-V1.8-COMPARISON.md) | 与上游 v1.8 helper 路线的对比 |
|
||||
|
||||
**任何关于「现在支持什么」的陈述,一律以根 `HANDOVER.md` 与源码为准。**
|
||||
|
||||
## 为什么归档
|
||||
|
||||
这份 1713 行(65 KB)的文件停留在 v1.7/v1.8 时代,却与根 `HANDOVER.md` 并行存在。
|
||||
它的过时结论在多轮开发中**反复误导整仓 grep**(本项目自己就踩过数次:搜 "RAR 分卷"/"加密"
|
||||
会命中这里,得到与现状完全相反的答案),因此 2026-09-23 把它移出主文档树。
|
||||
|
||||
保留它的唯一理由是**历史价值**:§9「坑清单」以及 v1.7/v1.8 时代的设计推导,
|
||||
对理解「当时为什么这么取舍、踩过什么坑」仍有考古意义。
|
||||
|
||||
原文完整副本(内容一字未改,含归档前的 md5):
|
||||
**[`archive/HANDOVER-v1.8-planning.md`](./archive/HANDOVER-v1.8-planning.md)**
|
||||
@@ -0,0 +1,285 @@
|
||||
# 真机验证清单:解压速度(10–40 MB/s 到底卡在哪)
|
||||
|
||||
> ### ⚑ 本清单的终局:停在第 1 步之前(2026-09-23 18:15,用户决定)
|
||||
>
|
||||
> **`T_copy` 没做;RAR 多线程(#51)不做;生产代码一行不动、不刷机 —— 11 分钟维持现状。**
|
||||
>
|
||||
> 理由:MT 的收益在 PC 上**已实测**(下面「第三件」2.45×/我们引擎 1.40–1.74×),
|
||||
> 但**能不能兑现,刚好吊在本文档第 1 步那个实验上** —— 而 MT 只并行解码、写盘恒为主线程
|
||||
> 串行(`unpack50mt.cpp` 的 `UnpWriteBuf()` 只在主线程调,见 §六 终局块),
|
||||
> 所以那个实验的结果很可能是「收工」。赌一次的成本(改构建 + 刷机 + 重跑 11 分钟 + 真机验证)
|
||||
> 压在不确定收益上 ⇒ **不做**。
|
||||
>
|
||||
> **下面全部内容保留**,作为「将来重开时的现成答案」。
|
||||
> ⚠️ 重开的**第一个动作是第 1 步的 `T_copy`,不是改构建**。
|
||||
|
||||
> ### ⚠️⚠️ 2026-09-23 深夜(第二次更正):**归档参数已拿到、MT 已实测、最大嫌疑已排除**
|
||||
>
|
||||
> **口径修正**:包不是 18 GB,是 **3 个分卷 ≈ 11.6 GB**(4 GB + 4 GB + 3.6 GB)。
|
||||
> **下文凡写「18 GB」处,一律按「≈11.6 GB」读**;核心算术已在本块重算。
|
||||
>
|
||||
> **① 真实归档参数(从 part3 的头实测,该文件随后被清理)**
|
||||
> RAR **5**、**卷 3/3**、**锁定**、**不是固实**(直解主头 `archive flags=0x0013`,`MHFL_SOLID=0x0004` 未置位)、
|
||||
> 压缩参数 **`-m1 -md=4g`**(**最快档 + 4 GiB 字典**)。关键条目 `PPSA16608.exfat`
|
||||
> **19.49 GB → 打包 3.83 GB(ratio 5.09)**,并**整段都在 part3 内**。
|
||||
> 详见 `docs/EXTRACTION-PERF.md` §六 第二次更正块。
|
||||
>
|
||||
> **② 最大嫌疑(scan 白解码一遍 = 2×)已排除。** 那建立在「固实 skip 必须解码」
|
||||
> (`dll.cpp:341` 只在 `!Arc.Solid` 走廉价 `SeekToNext()`)之上;**本档非固实** ⇒ 我们的 scan 不解码。
|
||||
>
|
||||
> **③ MT 收益已实测 = ~2.4×**(真实形状夹具:`-m1 -md4g`、5:1 可压缩、2 GB、`-mt1` 386→`-mt8` 946 MB/s 解压后)。
|
||||
> 接线只需 **1 个编译开关 + 1 个链接开关**(`-DRAR_SMP` + `-lpthread`),**不必改 vendored 源码、
|
||||
> 不必增删源文件** —— 早期版本写的「还要加 `threadmisc.cpp`」**是错的**:`threadpool.cpp:5` 已经
|
||||
> `#include "threadmisc.cpp"`(实测 `nm unrar7_threadpool.o` 里有 `T GetNumberOfThreads`),再加会重符号。
|
||||
>
|
||||
> **④ 但矛头现在指向「PS5 写路径」——所以本清单第 1 件事(`T_copy`)不变,而且更该先做。**
|
||||
> - 单线程解码在 PC 上同形状就是 **386–505 MB/s**;PS5 单核按 1/3 算也 ~130 MB/s。
|
||||
> - 真机:**只算 `.exfat` 一项**就是 19.49 GB ÷ 660 s = **≥29.5 MB/s 解压后**。
|
||||
> - **两次独立操作撞同一个数**:上传(写 11.6 GB)**30–40 MB/s** = 解压(写 ≥19.5 GB)≈30 MB/s。
|
||||
> ⇒ 优先怀疑这条写路径的上限就是 **30–40 MB/s**。**若 `T_copy` 也 ~10 分钟 ⇒ 收工,MT 不做**
|
||||
> (做了会被 I/O 吃掉);**若 `T_copy` 明显快 ⇒ 立刻上 MT,那 2.4× 是真的**。
|
||||
>
|
||||
> 下面是原始判断(保留),**优先级以本块为准**。
|
||||
|
||||
> ### ⚠️ 2026-09-23 晚 更正 —— 本文档的起点判断已反转。
|
||||
> 本文最初的前提是「已排除解码器」,那建立在两条现已作废的论证上:
|
||||
> ① 拿 **PS5 上解 RAR** 的 28 MiB/s 去比 **PC 上解 7z** 的 427 MiB/s(不同格式/解码器/机器);
|
||||
> ② UI 速度的 4× 摆动(已自行降级为 250 ms 采样噪声的弱证据)。
|
||||
>
|
||||
> **新增的三条事实把方向翻了过来**:格式是 **RAR**(解码就是 rarlab 的 UnRAR 库本身);
|
||||
> 包原本在 PC 上、经插件上传进 PS5;上传实测 30–40 MB/s。随后查代码发现:
|
||||
> **我们的 POSIX 构建没有定义 `RAR_SMP`**(`os.hpp:43-45` 的 `#define RAR_SMP` 在
|
||||
> `#ifdef _WIN_ALL` 内;官方 POSIX makefile 第 11 行是 `DEFINES=... -DRAR_SMP`),
|
||||
> 后果是 `unpack50mt.cpp`(`Unpack::Unpack5MT`,多线程 RAR5 解压器)**根本没编进来**,
|
||||
> `unpack.cpp` 的 MT 分支整段不参与编译 ⇒ **RAR 解码在 PS5 上是单线程的**。
|
||||
>
|
||||
> ⇒ **新的首要假设:28 MiB/s ≈ 单线程 RAR5 解码的正常量级。**
|
||||
> 详细更正与代码行号见 `docs/EXTRACTION-PERF.md` §六 文首更正块。
|
||||
> **下文按「实测 → 判读」仍有效;但第 2 步(读粒度)对 RAR 不适用**,已加注。
|
||||
|
||||
## 已到手的数据(2026-09-23 真机)
|
||||
|
||||
| 项 | 值 |
|
||||
|---|---|
|
||||
| 包大小 | **≈11.6 GB 压缩包**(4 GB + 4 GB + 3.6 GB 三个分卷;✅ 2026-09-23 晚更正,原写 18 GB 有误) |
|
||||
| 条目数 | 1252 |
|
||||
| 总耗时 | **11 分钟 = 660 s**(UI 显示) |
|
||||
| **UI 速度口径** | **解压后字节**(`src/rar_extract.c:766` 在 `UCM_PROCESSDATA` 里 `bytes_done += p2`)⇒ 10–40 MB/s 是「吐出数据的速度」 |
|
||||
| 平均吞吐(输入侧) | **≥ 17.6 MB/s 打包字节**(11.6 GB ÷ 660 s) |
|
||||
| 平均吞吐(输出侧) | **≥ 29.5 MB/s 解压后**——只算 `PPSA16608.exfat` 一项就是 19.49 GB ÷ 660 s |
|
||||
| 格式 | **RAR 5** ✅(头 8 字节 `52 61 72 21 1A 07 01 00`) |
|
||||
| 固实 | **不是固实** ✅(主头 `MHFL_SOLID=0x0004` 未置位) |
|
||||
| 压缩参数 | **`-m1 -md=4g`**(最快档 + 4 GiB 字典)✅ |
|
||||
| 关键条目 | `PPSA16608.exfat` 19 493 027 840 B → 打包 3 831 295 727 B(ratio **5.09**),整段在 part3 内 |
|
||||
| 来源 | 包原本在 **PC 上**,**经插件本身上传**进 PS5,上传速度 **30–40 MB/s**(写 11.6 GB) |
|
||||
| 源 / 目标设备 | **待确认** ⬅ **现在最关键的一条**(内置 SSD / 外置 USB / 是否同盘) |
|
||||
| 解压后总大小 | **待确认** ⬅ 决定输出侧吞吐的确切值(≥19.49 GB 已知) |
|
||||
|
||||
「18 GB 是压缩包大小」这条把口径钉住了,而且给出一个**不需要知道解压后大小的硬上界**:
|
||||
|
||||
```
|
||||
源盘必须在 660 s 内交出 18 GB 的归档数据
|
||||
⇒ 整条流水线的平均吞吐上界 = 18 GB ÷ 660 s ≈ 28 MiB/s
|
||||
```
|
||||
|
||||
解压后有多大都不影响这个上界 —— **无论解码多快,源盘就只有这个交付速度**。
|
||||
所以第 1 步的 `T_copy` 和它**可以直接比大小**(同一个 18 GB 文件、同一对设备)。
|
||||
|
||||
两个立刻可用的推论:
|
||||
|
||||
1. **per-entry(小文件)开销解释不了它。** 1252 个文件、平均 14.7 MB,不是「几万个小文件」那种
|
||||
形态;建文件 + rename 这两种 per-entry 成本加起来分摊到 0.53 s/文件 里微乎其微。
|
||||
2. **UI 上那个 10–40 MB/s 是瞬时值,不是平均值。** 进度回调只在累计 ≥ 1 MiB 时上报
|
||||
(`src/zip_extract.c:123`),`src/task.c:202-206` 每 **250 ms** 采一次样。250 ms 窗口 +
|
||||
写缓冲突发会让速度在「忽停忽走」之间跳,**波动里含相当比例的采样噪声**。
|
||||
⇒ 只有「解压后字节 ÷ 总耗时」可信;那 4× 的摆动只能说明「不是恒定负载」,不能单独定案。
|
||||
|
||||
---
|
||||
|
||||
## ★ 你要在真机上做的事(就两件)
|
||||
|
||||
### 第一件:补一个数字 `T_copy`(3 分钟操作 + 一次等待)
|
||||
|
||||
1. 把那个 **18 GB 的包**,用插件自己的**复制**功能,复制到**解压时输出所在的那块盘**
|
||||
2. **记下耗时**(UI 上有)
|
||||
3. 顺便记下:这个包**原本在哪**(内置 SSD / 外置 USB),**解压输出到哪**(同盘还是另一块)
|
||||
|
||||
判读(就在 `T_copy` 与 **660 s** 之间比):
|
||||
|
||||
| `T_copy` | 结论 | 我接下来做什么 |
|
||||
|---|---|---|
|
||||
| **≥ 10 分钟** | 存储物理极限,与代码无关 | **收工**,解码优化全部关闭(#48 取消) |
|
||||
| **1–2 分钟** | 解压比同一对设备上的纯搬运慢 5–10× ⇒ **代码里有真问题** | 做 #48:`vol_read()` 加 read-ahead |
|
||||
| 介于中间 | 混合 | 按字节数扣掉 I/O 分量,差值才是能改的部分 |
|
||||
|
||||
> 为什么这个数字这么关键:它是**同一台机器、同一对设备、同一份代码**,只把「解压」换成
|
||||
> 「纯搬运」。不需要造新包、不需要第二块盘、不需要任何假设。上面那 4 个对照实验(第 3 步)
|
||||
> 只在它的结论模糊时才需要。
|
||||
|
||||
### 第二件:两个小事实 —— **已答一半,剩两个待确认**
|
||||
|
||||
1. ~~RAR4 还是 RAR5?~~ **RAR5 ✅**;~~是否固实?~~ **非固实 ✅**;压缩参数 **`-m1 -md=4g` ✅**
|
||||
(都从 part3 的头上直接读到,见文首第二次更正块 ①)。
|
||||
顺带确认:**UI 的口径是解压后字节**(`src/rar_extract.c:766`)。
|
||||
2. **还缺两个数(都很便宜)**:
|
||||
- **解压后总大小** —— 决定输出侧吞吐。已知 ≥19.49 GB(仅 `.exfat` 一项)。
|
||||
插件跑解压时 UI 上的 `total` 就是它(`task->total = p->bytes_total`,`src/extract.c:64`)。
|
||||
- **源 / 目标设备** —— 包在哪块盘、解压输出写到哪块盘、是否同一块盘。
|
||||
这一条现在比什么都重要:写路径被怀疑是墙(文首块 ④)。
|
||||
|
||||
### 第三件(我已做完,不用上真机):多线程 RAR5 的收益
|
||||
|
||||
用 WinRAR 自带的 **UnRAR 7.23**(`-mt<N>` 开关可解析,虽然不出现在 `-?` 帮助里)
|
||||
在一份**按真实形状**造的夹具上做 A/B(`-m1 -md4g`、5.27:1 可压缩、2 GB、分卷):
|
||||
|
||||
| 线程 | 耗时 | 解压后吞吐 | 加速 |
|
||||
|---:|---:|---:|---:|
|
||||
| 1 | 5.18 s | 386 MB/s | 1.00× |
|
||||
| 4 | 2.39 s | 835 MB/s | 2.16× |
|
||||
| **8** | **2.11 s** | **946 MB/s** | **2.45×** |
|
||||
|
||||
⇒ **收益 2.4×,且已在真实形状上验证**(非固实版本 2.42×,量级一致)。
|
||||
接线只需 **1 个编译开关 + 1 个链接开关**(`-DRAR_SMP` / `-lpthread`),**不用改 vendored 源码、
|
||||
不用增删源文件**(早期写的 `threadmisc.cpp` 那一条经实测作废,见 §六 块 ④ 行内更正)——
|
||||
原因与行号见 `docs/EXTRACTION-PERF.md` §六 第二次更正块 ④。
|
||||
|
||||
> ⚠️ 但结论顺序仍以第一件为准:**先 `T_copy`**。如果墙是 30–40 MB/s 的写路径,
|
||||
> 这 2.45× 会被 I/O 完全吃掉,做了等于白做。
|
||||
> ⇒ **闭环(2026-09-23 18:15):`T_copy` 没做,用户决定不做这项了**(见文首终局块)。
|
||||
> 本节的实测数据**保留** —— 将来重开时它就是收益依据;但**第一步仍是 `T_copy`**。
|
||||
|
||||
---
|
||||
|
||||
## 第 1 步(决定性实验,= 上面「第一件」):用插件自己的「复制」功能做 I/O 基线
|
||||
|
||||
**这是目前唯一能把「存储」和「我们的代码」一刀切开的实验,而且零改动。**
|
||||
|
||||
插件里 **`TASK_COPY` 是已实现功能**,并且对 ≥ 256 MiB 的文件自动走
|
||||
`copy_file_pipeline()`(`src/filemgr.c:836`,阈值见 `:37`):**3 个 slot × 8 MiB、4096 字节对齐、
|
||||
独立读线程 + 独立写线程**。也就是说,它就是「同一台 PS5、同一对设备、同一份代码,
|
||||
只把解压那一步换成纯搬运」。
|
||||
|
||||
**操作**:把那个 18 GB 的包,用插件自己的复制功能,复制到**解压时输出所在的那块盘**上。
|
||||
记下耗时 `T_copy`。
|
||||
|
||||
**判读:**
|
||||
|
||||
| 观察到 | 结论 | 下一步 |
|
||||
|---|---|---|
|
||||
| `T_copy` ≈ 10 分钟或更多 | **存储物理极限**,与我们的代码无关 | 收工。解码优化全部搁置 |
|
||||
| `T_copy` ≈ 1–2 分钟(远小于 660 s) | 我们的解压比同一对设备上的纯 I/O **慢 5–10×** ⇒ **代码里有真问题** | 进第 2、第 4 步 |
|
||||
| 介于两者之间 | 混合 | 按字节数把 I/O 分量扣掉,差值才是我们能动的部分 |
|
||||
|
||||
严格比较要看**总移动字节**:复制读 18 GB + 写 18 GB = 36 GB;解压读 ≤ 18 GB、写 = 解压后大小。
|
||||
所以「复制 36 GB 用了多久」和「解压搬了 (18 GB + 解压后大小) 用了 660 s」才是同一口径。
|
||||
|
||||
> 为什么这个实验优于第 3 步那一堆:它不需要造新包、不需要两块盘、不需要任何假设,
|
||||
> 而且**测的就是出事的那对设备**。第 3 步只在第 1 步结论模糊时才需要。
|
||||
|
||||
---
|
||||
|
||||
## 第 2 步:读写粒度审计 —— ~~目前唯一可疑的代码级病因~~(⚠️ **对 RAR 不适用**)
|
||||
|
||||
> **2026-09-23 晚加注:本节整段只对 7z / ZIP 有效。** 它比较的是「引擎内部缓冲大小」,
|
||||
> 而 RAR 的归档 I/O 在 UnRAR 自己的 `File` 类里(我们按路径打开,
|
||||
> `src/rar_extract.c:1059`),我们的回调只收到解压后的数据 —— **读粒度我们改不了**。
|
||||
> 唯一还能对上 RAR 的那半句是「**写入**粒度」,而 RAR 的写出也在 UnRAR 内。
|
||||
> ⇒ **真机工作负载是 RAR 时,本节没有可操作性**;留着是为了 7z/ZIP 场景。
|
||||
> 顺带:本表里 RAR 那行写的 4 MiB 是 `File::CopyBufferSize()`
|
||||
> (`third_party/unrar7/file.hpp:148-153`),那是**文件复制**的缓冲,**不是解压读取缓冲** ——
|
||||
> 原文引用错了行,这条勘误一并记在这里。
|
||||
|
||||
把四条路径的**单次请求大小**摊开看(已核对源码):
|
||||
|
||||
| 路径 | 源侧读粒度 | 目标侧写粒度 |
|
||||
|---|---|---|
|
||||
| **复制**(pipeline,≥ 256 MiB) | **8 MiB** × 3 slot,4096 对齐,独立读线程 | 8 MiB |
|
||||
| **7z** | chain `SZ_IN_CHUNK = 256 KiB`(`src/sevenz_chain.c:87`);MT 路径 `inBufSize_MT = 1 MiB` | `SZ_OUT_CHUNK = 64 KiB`(`src/sevenz_chain.c:94`) |
|
||||
| **ZIP** | `ZIPX_IO_BUFFER = 128 KiB`(`src/zip_extract.c:32`) | 128 KiB(`src/zip_extract.c:900`) |
|
||||
| **RAR** | UnRAR 内部 `File::CopyBufferSize() = 4 MiB`(`third_party/unrar7/file.hpp:148-153`) | 4 MiB |
|
||||
|
||||
⇒ 结论一:**三个引擎都不是「小读」病理**,最小也有 128 KiB,不是 4 KB 那种灾难。
|
||||
|
||||
⇒ 结论二:**但都比复制小 32–64 倍**。
|
||||
|
||||
为什么这可能正是病根 —— 在**延迟主导**的设备上,吞吐 ≈ 单次请求大小 ÷ 每次请求的等效延迟:
|
||||
|
||||
```
|
||||
256 KiB / 10 ms = 25 MB/s ← 正好落在实测 10–40 MB/s 的中间
|
||||
8 MiB / 10 ms = 800 MB/s ← 复制路径不会撞这个上限
|
||||
```
|
||||
|
||||
这个算术顺带解释两件事:
|
||||
|
||||
- **为什么吞吐会 4× 摆动**:等效延迟随设备状态(HDD 寻道、USB 桥接、写缓存回刷)变化,
|
||||
线性映射到吞吐上就是大幅波动。
|
||||
- **为什么「条目平均 14.7 MB」没能救我们**:per-entry 成本的确不是主因,但**读请求粒度**是另一回事 ——
|
||||
它由引擎内部缓冲决定,与条目大小无关。
|
||||
|
||||
最可能的场景是**源和目标在同一块盘**(外置 USB HDD 上解压到同一块盘):读流和写流同时存在,
|
||||
磁头来回跑,OS read-ahead 被写回刷反复打断,于是每次请求退化成一次寻道 —— 这正好让上面那个
|
||||
算术成立。第 1 步如果出现「复制明显快于解压」,就是在支持它(复制的寻道次数只有解压的 1/32)。
|
||||
|
||||
**如果第 1 步指向这里,改动面其实很小**:所有 7z 的读都只经过 `src/sevenz_volstream.c` 的
|
||||
`vol_read()` 这一个函数(ZIP/RAR 同理在 `src/zipx_volstream.c`)。在那里加一层 read-ahead
|
||||
(向后 seek 时丢弃缓存)即可,**不需要动解码器**。但这要等第 1 步的结论,现在不写。
|
||||
|
||||
---
|
||||
|
||||
## 第 3 步:四个对照实验(零改动)—— 第 1 步结论模糊时才需要
|
||||
|
||||
| 实验 | 做法 | 若结果为 A | 若结果为 B |
|
||||
|---|---|---|---|
|
||||
| **A 换简单包** | 造一个「**单一大文件**(如 10 GB 伪随机数据)」的 zip,放内置 SSD,解到内置 SSD | 吞吐跳到 **150+ MiB/s** ⇒ 慢在**条目数 / per-entry 开销** | 仍是 10–40 ⇒ 慢在**存储或 I/O 模式**,与条目数无关 |
|
||||
| **B 换源设备** | 同一个包,分别放**内置 SSD** 与**外置 USB** | 两者差别巨大 ⇒ 就是源盘带宽 | 两者一样慢 ⇒ 不是源盘 |
|
||||
| **C 换目标设备** | 同一个包,分别解到**内置**与**外置** | 差别巨大 ⇒ 写入端是瓶颈 | 一样慢 ⇒ 不是目标盘 |
|
||||
| **D 同盘 vs 异盘** | 包与输出在同一块盘 / 在两块盘 | 同盘明显更慢 ⇒ 读写争用(这也是第 2 步最看好的假设) | — |
|
||||
|
||||
> 参考量级:机械/低成本 USB HDD 顺序读写约 30–80 MB/s,小文件随机访问降到 1–10 MB/s。
|
||||
|
||||
---
|
||||
|
||||
## 第 4 步:只有在第 1/3 步指向「我们的代码」时才做(要改代码)
|
||||
|
||||
1. **阶段计时**:在四个阶段边界累加时间戳 —— `scan` / `read+decode` / `write staging` /
|
||||
`publish`。用 `clock_gettime(CLOCK_MONOTONIC, ...)`(该原语已在三个引擎里用于进度上报,
|
||||
真机可用)。收尾用 `printf` 打进日志。
|
||||
2. **拆开 `read+decode`**:这一段目前分不开。可用「同一归档跑两遍(第二遍吃页缓存)」或
|
||||
「解到 /dev/null 类目标」来逼近 I/O 与解码的分界。
|
||||
3. **`SZX_MT_THREADS` 运行时可配**:现在是 `src/sevenz_mt.h:21` 的编译期 `#define 8`,
|
||||
做线程数 A/B 必须重编四次。改成运行时读(默认仍 8)会让实验便宜很多。
|
||||
4. **读粒度 A/B**:`SZ_IN_CHUNK` / `ZIPX_IO_BUFFER` 加大到 1–4 MiB 各构建一版,
|
||||
看第 1 步指出的那条曲线是否真的跟着动。
|
||||
5. 埋点要用**编译开关**控制,关闭时零开销 —— 避免影响已发布的产物指纹。
|
||||
|
||||
---
|
||||
|
||||
## 已排除的项(不要再花时间)
|
||||
|
||||
> ⚠️ 本表 2026-09-23 晚已按「格式 = RAR」重排。原表里「解码整体就不是瓶颈」这条前提已撤回,
|
||||
> 所以**同一批项现在被分成两类**:与 RAR 无关(不用看)、以及真已排除。**两类都不要花时间。**
|
||||
|
||||
**A. 与 RAR 工作负载无关**(它们是 7z / ZIP 的项;RAR 解码在 rarlab 库里,我们碰不到)
|
||||
|
||||
| 项 | 为什么无关 |
|
||||
|---|---|
|
||||
| CRC 硬件化 | 对 7z/ZIP 才值 4–5%;**RAR 的 CRC 由 UnRAR 自己算** |
|
||||
| MT 扩到 BCJ2 / 多 folder | 纯 7z 概念 |
|
||||
| ZIP inflate 换 libdeflate | ZIP 专属 |
|
||||
| AES-NI | 只影响 ZIP 加密流与 7zAES;RAR 加密走 UnRAR 自己的 rijndael |
|
||||
| 读粒度 / read-ahead(#48) | RAR 按路径自读(`src/rar_extract.c:1059` 的 `od.ArcName`),回调只收解压后数据,**插不进去** |
|
||||
|
||||
**B. 真已排除**(与格式无关,或已结论)
|
||||
|
||||
| 项 | 为什么排除 |
|
||||
|---|---|
|
||||
| 与 7-Zip 比倍数 | 真机上没有可比对象(上游那个 `wfm-7zip-helper.elf` 单独分发,本仓刻意不走 helper 路线);而且**那是 7z 的对比,与 RAR 负载无关** |
|
||||
| 逐条目 fsync | 已经全部移除(2026-09-16) |
|
||||
| 「条目太小」 | 1252 条 / 18 GB,平均 14.7 MB,不是小文件场景 |
|
||||
| 「我们的 RAR 解码器写得慢」 | 解码就是 vendored 的 UnRAR 库本体,我们只剩一层 facade |
|
||||
|
||||
**C. 唯一还站着的、且直接针对真实负载的一项** ⬅ **优先看这个**
|
||||
|
||||
| 项 | 状态 |
|
||||
|---|---|
|
||||
| **UnRAR RAR5 多线程解压(`RAR_SMP` + `-lpthread`)** | **❌ 不做(2026-09-23 用户决定,见文首终局块)**。我们没打开它;收益已实测(PC 上 2.45× / 我们引擎 1.40–1.74×)但被「写路径是否为墙」卡住,而验证它的 `T_copy` 未执行 ⇒ 搁置 |
|
||||
@@ -0,0 +1,297 @@
|
||||
# 重写可行性评估报告
|
||||
|
||||
> 评估日期:2026-09-15 · 评估对象:`LisherSong/ps5-web-file-manager`(v1.9.1)
|
||||
> 目标问题:项目源自他人代码、怀疑授权不清;若从零重写,改动量多大?能否做得更好?
|
||||
>
|
||||
> ⚠️ 本文是**工程视角**的合规盘点,不是法律意见。若涉及商用/闭源决策,请咨询律师。
|
||||
|
||||
---
|
||||
|
||||
## 0. 结论先行
|
||||
|
||||
| 问题 | 答案 |
|
||||
|---|---|
|
||||
| 上游真的没有许可吗? | **否。上游是 GPL-3.0**,我们自己也是 GPL-3.0,两者一致 |
|
||||
| 现在能合法发布吗? | **能**。GPL-3.0 允许修改和再分发,只需满足归因 + 源码可得 |
|
||||
| 有没有真实风险? | 原本 5 个,**4 个已于 2026-09-23 关闭、第 5 个经确认接受现状**(见 §2);**从头到尾没有一个需要重写** |
|
||||
| 全量重写要多少人日? | **35–50 人日**(约 1.4–2 万行需重写,第三方 7.1 万行可直接复用) |
|
||||
| 值得重写吗? | **取决于目标**:想闭源/商用 → 必须重写;想开源分享 → **完全不必** |
|
||||
|
||||
**最反直觉的一点**:我们自建的 7z 引擎核心(3,661 行)**只依赖公共领域和 zlib 许可的第三方库,与上游 GPL 代码零耦合** —— 它是完全干净的资产,今天就能单独抽成 MIT 授权的独立库。
|
||||
|
||||
---
|
||||
|
||||
## 1. 许可现状核查(事实,非猜测)
|
||||
|
||||
### 1.1 上游授权 —— 你的前提是错的
|
||||
|
||||
通过 GitHub API 查询(2026-09-15):
|
||||
|
||||
```
|
||||
owendswang/ps5-web-file-manager
|
||||
license: GPL-3.0 stars: 78 forks: 6
|
||||
created: 2026-06-16 last push: 2026-09-08 archived: false
|
||||
```
|
||||
|
||||
上游**有明确许可**,且是 GPL-3.0。我们的 `LICENSE` 同样是 GPL-3.0,在 root commit `5cb0b76`(Initial import)时加入 —— **两边一致,不存在"无授权"的灰色地带**。
|
||||
|
||||
### 1.2 授权链条
|
||||
|
||||
```
|
||||
ps5-payload-dev/websrv (John Törnblom, GPLv3+)
|
||||
│ 源码里仍保留 "Copyright (C) 2024/2025 John Törnblom"
|
||||
│ —— asset.c / asset.h / mime.h / websrv.h 四个文件
|
||||
▼
|
||||
owendswang/ps5-web-file-manager (GPL-3.0)
|
||||
▼
|
||||
LisherSong/ps5-web-file-manager (GPL-3.0) ← 本项目
|
||||
```
|
||||
|
||||
`src/asset.c:1` 等文件里的 Törnblom 版权声明**至今完整保留** ✅ —— 说明 GPL §5(a)「保留版权声明」这一条在最上游那一环是满足的。
|
||||
|
||||
---
|
||||
|
||||
## 2. 真实风险清单
|
||||
|
||||
按严重度排序。**注意:没有一条需要重写代码来解决。**
|
||||
|
||||
| # | 风险 | 严重度 | 具体位置 | 修法 | 成本 |
|
||||
|---|---|---|---|---|---|
|
||||
| 1 | ~~**归因缺失**~~ → **已修正(2026-09-23)**:README Credits 已补上 `owendswang` 与 rarlab UnRAR / opello 镜像、并把已删除的 `third_party/unrar/` 从 Credits 里清理掉;`THIRD_PARTY_NOTICES` 本就正确。**残留**:git 历史里的 `5cb0b76 Initial import` 无法追溯上游提交 | 🟡 中 → 🟢 低 | `README.md` Credits | 历史归属只能在 Release 说明与 Credits 里声明;如需彻底重建历史得重写仓库 | 已完成 |
|
||||
| 2 | **unRAR 与 GPL-3.0 的附加限制冲突**:UnRAR 许可禁止"用于开发 RAR 兼容压缩器",GPL-3.0 §7 禁止附加限制,严格讲不兼容 | 🟡 中 | `third_party/unrar7/` | 见 §2.1 | 0(接受) |
|
||||
| 3 | ~~**ezremote 是 GPLv2**:README 只写 "GPLv2",未标 "or later"。GPLv2-only 与 GPL-3.0 **不兼容**~~ → **已定性并关闭(2026-09-23)**:确为 **GPL-2.0-only**,但逐行比对确认**我们未取其代码** | ✅ 已关闭 | `src/pkg_info.c`(PKG 预览) | 无需重写;README 中英措辞已改准,代码加了来源注记 —— 见 §2.2 | 0 |
|
||||
| 4 | ~~**二进制分发需提供源码**(GPL §6)~~ → **已补(2026-09-23)**:v1.9.2 Release 说明的 License 段原先只链了**上游**仓库,未链本仓库 | ✅ 已关闭 | Release 里的 ELF | 已用 `gh release edit` 加入 "Corresponding source for this binary" 段(资产未动、仍非 draft/pre、仍是 latest);今后发版沿用 `.build/release-notes-*.md` 模板 | 0 |
|
||||
| 5 | Title ID `FMGR88888` 与上游相同,可能与他人 payload 冲突 | 🟢 低 | `Makefile:21` | ~~换一个自定义 ID~~ → **2026-09-23 决定维持现状(用户确认)**。核对结论:`TITLE_ID` 只出现在两处 —— `src/app_installer.c:86`(PKG 安装目标目录 `/user/app/<TITLE_ID>`)和 `src/version.c:31`(`/api/version` 上报),**与解压 / 上传 / 浏览功能无关**,且本项目是上游 fork 的直接替代者(同 ID 便于覆盖安装)。代价:与上游 payload **不能共存**,同时装会撞目录 | 0(接受) |
|
||||
|
||||
### 2.1 关于 unRAR(风险 2 详解)
|
||||
|
||||
UnRAR 许可原文允许"在任何软件中处理 RAR 归档",但**禁止用它开发 RAR 兼容的压缩器**。这是一个"附加限制",与 GPL-3.0 §7 冲突。
|
||||
|
||||
实务上的三种处理:
|
||||
|
||||
| 方案 | 做法 | 代价 |
|
||||
|---|---|---|
|
||||
| **A. 接受现状(推荐)** | 明确声明 unrar 部分适用其自有条款,其余部分 GPL-3.0 | 0 —— rarlab 官方自己就这么分发,社区普遍接受 |
|
||||
| B. 移除 RAR 支持 | 删掉 `rar_extract.c` + `third_party/unrar7` | 失去 RAR(含分卷/加密)—— 不划算 |
|
||||
| C. 换实现 | 找自由许可的 RAR 解码器 | **市面上不存在可用的**,死路 |
|
||||
|
||||
**建议 A**。风险等级实际很低:你只做解压不做压缩,本来就不触碰被禁止的那一条。
|
||||
|
||||
### 2.2 关于 ezremote(风险 3 详解,2026-09-23 结案)
|
||||
|
||||
**结论:风险不成立 —— 确为 `GPL-2.0-only`,但我们一行代码都没取。**
|
||||
|
||||
#### 第一步:许可措辞(用 `gh api` 查上游,不是猜)
|
||||
|
||||
```
|
||||
gh api repos/cy33hc/ps5-ezremote-client → license.spdx_id = "GPL-2.0"
|
||||
gh api .../contents/LICENSE → GNU GPL v2 全文(June 1991)
|
||||
gh api .../contents/source/actions.cpp → 无任何版权 / GPL 声明头
|
||||
gh api .../contents/source/clients/*.h → 同上,一个声明头都没有
|
||||
```
|
||||
|
||||
上游**全部源文件都不带许可声明**,唯一的许可陈述就是那份 GPLv2 全文。
|
||||
GPLv2 的 "or later" 只能由版权人**明示**授予(LICENSE 全文本身不含该授予),
|
||||
所以按其自身现状应认定为 **`GPL-2.0-only`**。
|
||||
|
||||
这一点很关键:**`GPL-2.0-only` 与 GPL-3.0 不兼容**(这正是 FSF 发明
|
||||
"GPLv2 or later" 惯例的原因)。所以「到底抄没抄」不是学术问题 ——
|
||||
抄了就必须重写这个模块。
|
||||
|
||||
#### 第二步:逐行比对(决定性的一步)
|
||||
|
||||
上游与 PKG 相关的**只有一个文件**:`source/sfo.cpp`(4,209 B / 141 行 / C++)。
|
||||
上游**没有 `.pkg` 容器解析器** —— tree 里的 `Ps5_ezRemote_Client_2.00.pkg`
|
||||
是一个已编译的 payload(10 MB),不是源码。
|
||||
|
||||
| 维度 | 上游 `sfo.cpp` | 我们的 `src/pkg_info.c` |
|
||||
|---|---|---|
|
||||
| 语言 | C++(`reinterpret_cast` / `std::map` / `namespace SFO`) | **C99** |
|
||||
| 函数分解 | 三个独立函数 `GetString` / `GetParams` / `GetParamsFromParamJson` | 单个 `append_sfo_fields(strbuf_t *, ...)` 直接流式产出 JSON 片段 |
|
||||
| 返回值 | `std::map<std::string,std::string>` | 写进 `strbuf_t`,无中间容器 |
|
||||
| JSON | 依赖 **json-c**(`json_tokener_parse` / `json_object_object_get`) | **自写分词器**(`parse_json_tokens()` / `json_object_value()`,在 `json_util.c`) |
|
||||
| 越界防护 | 仅两处 `size <` 检查,其余裸指针 + `reinterpret_cast` | 逐项校验(`count > SFO_ENTRY_MAX`、`index_end > size`、`key_offset < index_end`、`memchr` 找 NUL、`read_le32` 定长读) |
|
||||
| 覆盖范围 | 只有 SFO + param.json | 另有 **`.pkg` 条目表**(`PKG_CNT_MAGIC` / FIH / LIH / 条目类型 `0x1000` / `0x1200` / `0x121f` / `0x2000`)、本地化图标选择、两个 HTTP 端点 |
|
||||
|
||||
**唯一重合的是格式事实**:SFO magic `0x46535000`、20 字节头 / 16 字节条目、
|
||||
`keyofs`+`nameofs` 与 `valofs`+`dataofs` 的间接寻址。这些是 PS5 文件格式的客观规定,
|
||||
也是解析它的**唯一办法**(等同合并原则),不构成可保护的表达。
|
||||
|
||||
⇒ **不存在代码衍生关系**,`src/pkg_info.c` 无需重写。
|
||||
|
||||
#### 第三步:已落地的处置
|
||||
|
||||
1. `README.md` / `README.zh-CN.md` 的 Credits 条目:由 "Preview PKG info. License: GPLv2"
|
||||
改为明确写出 **GPL-2.0-only、与本项目不兼容、未取其代码**,并指向本节 ——
|
||||
后来人不会再把它当成"我们的依赖"去理解授权链。
|
||||
2. `src/pkg_info.c` 文件头补了来源注记(含比对理由与本节指引)。
|
||||
**纯注释,不改变编译产物** —— 已确认 `src/` 内没有 `__LINE__` / `__FILE__` 依赖,
|
||||
注释被预处理器丢弃后目标文件逐字节相同。
|
||||
3. 本节即为 provenance 留档。
|
||||
|
||||
> 适用范围:这套「先查许可措辞 → 再做逐行比对 → 最后把结论写进代码注记」的流程,
|
||||
> 对任何「README 里 credits 了某个项目」的情形都适用。因为 README 的 Credits 段落里
|
||||
> 混着两类东西:**真正 vendored 的代码**(minizip-ng / zlib / UnRAR)和
|
||||
> **只是参考了思路的项目**(websrv / ftpsrv / zftpd / etaHEN / ezremote)。
|
||||
> 两者在授权义务上完全不同,但排版把它们放在同一张列表里 —— 这就是这个疑问的由来。
|
||||
|
||||
---
|
||||
|
||||
## 3. 代码归属盘点(决定重写成本的关键)
|
||||
|
||||
### 3.1 总量分布
|
||||
|
||||
| 类别 | 行数 | 重写时怎么办 |
|
||||
|---|---:|---|
|
||||
| **第三方 vendored** | **71,528** | ♻️ **直接重新 vendor,一行都不用写** |
|
||||
| ├ zlib 1.3.1 | 20,106 | zlib 许可 |
|
||||
| ├ unrar7 7.20.1 | 27,710 | UnRAR 许可 |
|
||||
| ├ LZMA SDK 26.03 | 17,248 | **公共领域** |
|
||||
| └ minizip-ng 4.2.2 | 6,464 | zlib 许可 |
|
||||
| **第一方代码** | **20,844** | 这是唯一需要写的部分 |
|
||||
| ├ v1.7 上游遗产 | 13,860 | 🔴 受 GPL 约束 |
|
||||
| └ 我们新增 | 6,984 | 🟢 版权归我们 |
|
||||
|
||||
**这是整份报告最重要的数字**:项目里 **77% 的代码是第三方库**,重写时原样搬走即可。真正需要动手的只有 2 万行第一方代码,其中又只有不到 1.4 万行受 GPL 约束。
|
||||
|
||||
### 3.2 我们新增代码的独立性分析
|
||||
|
||||
v1.7 之后**新建**的文件(6,745 行),逐个检查其依赖:
|
||||
|
||||
| 文件 | 行数 | 依赖 | 能否独立授权 |
|
||||
|---|---:|---|:---:|
|
||||
| `sevenz_chain.c/.h` | 2,094 | 仅 LZMA SDK(**公共领域**) | ✅ **完全干净** |
|
||||
| `zipx_volume.c/.h` | 658 | 仅自身 | ✅ **完全干净** |
|
||||
| `zipx_volstream.c/.h` | 494 | minizip-ng(zlib)+ zipx_volume.h | ✅ **完全干净** |
|
||||
| `sevenz_volstream.c/.h` | 415 | LZMA SDK + zipx_volume.h | ✅ **完全干净** |
|
||||
| `rar_extract.c/.h` | 1,169 | `zip_extract.h`(共享协议) | ⚠️ 弱耦合,抽协议即可解绑 |
|
||||
| `sevenz_extract.c/.h` | 1,790 | `zip_extract.h`(共享协议) | ⚠️ 弱耦合,抽协议即可解绑 |
|
||||
| `zipx_common.c` | 99 | `zip_extract.h` | ⚠️ 内容仅限额/状态串,30 分钟可重写 |
|
||||
| `cpu_support_stub.c` | 26 | 无 | ✅ 干净 |
|
||||
|
||||
**核心结论**:
|
||||
- **3,661 行(7z 解码链 + 分卷流抽象)与 GPL 代码零耦合** —— 这是项目最有价值的部分(自解析 folder + pull 式 codec 链 + BCJ2 + 7zAES + 三种分卷语义),也是投入最多的部分。它们**今天就可以抽成独立的 MIT/Apache 库**,不受上游任何影响。
|
||||
- 另有 2,959 行只通过 `zip_extract.h` 的**共享协议**(状态枚举、进度回调、限额结构)与上游耦合。把那 100 行协议定义抽成独立的 `archive_api.h` 就能解绑。
|
||||
|
||||
---
|
||||
|
||||
## 4. 三条路径对比
|
||||
|
||||
### 路径 A:维持 GPL-3.0 + 补齐合规(推荐)
|
||||
|
||||
| 项 | 内容 |
|
||||
|---|---|
|
||||
| 做什么 | README 补 owendswang 署名、加 NOTICE、确认 ezremote 许可、Release 附源码链接、换 Title ID(→ 末项已于 2026-09-23 决定维持现状,见风险 5) |
|
||||
| 成本 | **0.5–1 人日** |
|
||||
| 收益 | 合规闭环,零功能损失,保留全部现有能力 |
|
||||
| 风险 | 无 |
|
||||
| 适合 | **想开源分享、想让成果被保护** |
|
||||
|
||||
### 路径 B:架构重构(保留 GPL-3.0)
|
||||
|
||||
| 项 | 内容 |
|
||||
|---|---|
|
||||
| 做什么 | 分层重写:platform 层 / archive 引擎层 / HTTP 层 / 前端模块化;抽 `archive_api.h` 解耦;统一进度模型 |
|
||||
| 成本 | **12–18 人日** |
|
||||
| 收益 | 代码可维护性大幅提升,引擎可独立成库,修掉 §6 的已知缺陷 |
|
||||
| 风险 | 中 —— 需真机回归,163 个 host check 要全绿 |
|
||||
| 适合 | **觉得现在代码烂、想长期维护** |
|
||||
|
||||
### 路径 C:Clean-room 全量重写(换许可)
|
||||
|
||||
| 项 | 内容 |
|
||||
|---|---|
|
||||
| 做什么 | 不参考上游代码,按功能规格从零写 ~13,860 行受 GPL 约束的部分 |
|
||||
| 成本 | **35–50 人日**(含真机调试) |
|
||||
| 收益 | 完全自有版权,可任选许可(含闭源商用) |
|
||||
| 风险 | **高** —— clean-room 必须严格隔离:写代码的人不能看过上游源码;否则法律上无效 |
|
||||
| 适合 | **确定要闭源/商用** |
|
||||
|
||||
### 4.1 路径 C 的工作量拆解
|
||||
|
||||
| 模块 | 行数 | 难度 | 人日 |
|
||||
|---|---:|---:|---:|
|
||||
| HTTP 服务 + 路由(websrv/main/asset/mime/file_response) | ~700 | 低 | 3 |
|
||||
| 文件管理核心(filemgr.c) | 2,458 | **高** | 8 |
|
||||
| 上传 / 下载 / 断点续传 | 1,343 | 中 | 5 |
|
||||
| 文件系统工具(fs_util/path_util/list/space/text) | 1,229 | 中 | 4 |
|
||||
| PKG 安装 / 信息解析 / app_installer | 847 | 中 | 4 |
|
||||
| 任务调度 + 通知(task/notify) | 253 | 低 | 1.5 |
|
||||
| ZIP 引擎 + 分卷(zip_extract/zipx_*) | 3,091 | **高** | 8 |
|
||||
| 前端(main.js / main.css / index.html / i18n) | 4,779 | 中 | 6 |
|
||||
| 构建系统 + 资产内嵌 | ~200 | 低 | 1 |
|
||||
| 测试矩阵(对齐现有 163 checks) | — | 中 | 5 |
|
||||
| 真机调试 + PS5 平台适配 | — | **高** | 6 |
|
||||
| **合计** | **~14,900** | | **≈ 51 人日** |
|
||||
|
||||
可复用的:7z 全套(4,299 行)+ zipx 分卷(1,193 行)+ 第三方(71,528 行)—— 这三项占了重头戏,所以才是 50 人日而不是 150 人日。
|
||||
|
||||
---
|
||||
|
||||
## 5. 重写能做到"比现在更好"的地方
|
||||
|
||||
如果真要走 B 或 C,这些是现在已知的技术债,**顺手一起解决才值得动**:
|
||||
|
||||
| # | 现有问题 | 位置 | 改进方案 |
|
||||
|---|---|---|---|
|
||||
| 1 | **进度口径三处不一致**:进度条按字节、文字按条目数、ETA 按字节速度,混合大包上体验割裂 | `main.js:1985` / `main.js:2014` / `task.c:129-177` | 统一为字节口径,ETA 用滑动窗口 |
|
||||
| 2 | **PS5 `*at()` 族运行时损坏**(返回 -1 且 errno=0),现有绕行逻辑散落在 `zip_extract.c` | `zip_extract.c` | 抽 platform 层集中处理,写新代码不再踩 |
|
||||
| 3 | 引擎与 HTTP 层耦合:解压协议定义在 `zip_extract.h` 里 | `zip_extract.h` | 抽 `archive_api.h`,引擎可独立成库 |
|
||||
| 4 | `main.js` 2,652 行单文件,无模块拆分 | `assets/main.js` | 按 view / api / task 拆模块 |
|
||||
| 5 | `filemgr.c` 2,458 行,路由 + 业务逻辑 + 平台调用混在一起 | `src/filemgr.c` | 分 handler / service / platform 三层 |
|
||||
| 6 | 测试靠手工脚本,未接入 `make test` | `tests/` | 接 CI,覆盖率可量化 |
|
||||
| 7 | ~~唯一功能缺口:7z `-mhe=on` 加密头~~ **已闭合**(2026-09-23,`src/sevenz_header.c`) | `sevenz_extract.c` | 无剩余格式缺口;重写时该项可删 |
|
||||
|
||||
---
|
||||
|
||||
## 6. 建议
|
||||
|
||||
### 6.1 我的推荐:路径 A,外加一条"资产剥离"
|
||||
|
||||
**不要全量重写。** 三个理由:
|
||||
|
||||
1. **GPL-3.0 对你有利,不是负担**。它保证别人拿走你的 7z 引擎成果后**必须同样开源**。换成 MIT,别人可以直接闭源拿去卖 —— 你花了大量精力做的 BCJ2 链、7zAES、分卷流抽象会被白嫖。
|
||||
2. **重写的法律风险比不重写更高**。你已经看过上游源码了,clean-room 的前提已被破坏。真重写必须找没看过上游的人来做,还得隔离沟通 —— 成本远超 50 人日。
|
||||
3. **你的核心资产本来就是干净的**。3,661 行的 7z 解码链 + 分卷流只依赖公共领域和 zlib 许可,**现在就能单独抽出来做 MIT 授权的独立库**,不需要动主项目一根指头。
|
||||
|
||||
### 6.2 立刻可做的三件事(共 1 天)
|
||||
|
||||
```
|
||||
1. README.md Credits 补一行:
|
||||
- [owendswang/ps5-web-file-manager](...): base implementation. License: GPL-3.0.
|
||||
|
||||
2. 把 sevenz_chain.{c,h} + sevenz_volstream.{c,h} + zipx_volume.{c,h}
|
||||
+ zipx_volstream.{c,h} 抽成独立仓库,MIT 授权,主项目作为 submodule 引用。
|
||||
→ 3,661 行成果立刻获得独立身份,且证明这部分是你的原创。
|
||||
|
||||
3. ~~确认 ezremote 是 "GPLv2" 还是 "GPLv2 or later";若是 v2-only,重写 pkg_info.c~~
|
||||
→ **已结案(2026-09-23)**:是 `GPL-2.0-only`,但逐行比对确认**我们未取其代码**,
|
||||
**无需重写**。比对记录与处置见 §2.2。
|
||||
4. ~~v1.9.2 Release 说明补本仓库链接(GPL §6 源码提供义务)~~
|
||||
→ **已补(2026-09-23)**,见风险 4 行。
|
||||
```
|
||||
|
||||
### 6.3 需要你回答的问题
|
||||
|
||||
**你重写的动机到底是什么?** 不同答案对应完全不同的方案:
|
||||
|
||||
| 如果你的目标是… | 应该走 | 成本 |
|
||||
|---|---|---|
|
||||
| 想闭源 / 商业化 | C(且必须找没看过上游的人写) | 50+ 人日 |
|
||||
| 只是担心"没许可"不合法 | **A** —— 你的担心不成立 | 0.5 天 |
|
||||
| 想让别人知道这是你写的 | **A** —— GPL 允许你在修改部分署名 | 0.5 天 |
|
||||
| 觉得代码质量差、想重构 | B | 12–18 人日 |
|
||||
| 想保护成果不被闭源 | **A** —— GPL-3.0 已经是最佳选择 | 0 天 |
|
||||
|
||||
---
|
||||
|
||||
## 附录:核查方法与数据来源
|
||||
|
||||
| 项 | 来源 |
|
||||
|---|---|
|
||||
| 上游许可 | GitHub API `repos/owendswang/ps5-web-file-manager`,2026-09-15 查询 |
|
||||
| 本项目许可 | `LICENSE`(35,149 B,GPL-3.0 全文),root commit `5cb0b76` 引入 |
|
||||
| 代码行数 | `wc -l` 于 v1.9.1 工作树;上游基线取 `git show 5cb0b76:<file>` |
|
||||
| 文件归属 | `git ls-tree -r 5cb0b76 -- src assets` 与当前工作树的差集 |
|
||||
| 依赖分析 | 逐个 grep `#include "` 于自有文件 |
|
||||
| 已知缺陷 | 项目 `HANDOVER.md` 第四节 + `.workbuddy/memory/MEMORY.md` |
|
||||
@@ -0,0 +1,354 @@
|
||||
# ELF 瘦身可行性分析
|
||||
|
||||
> 2026-09-20 · 环境:WSL Ubuntu-22.04 + `/opt/ps5-payload-sdk` + LLD 18.1.8
|
||||
> 基线产物 `web-file-mgr-v1.9.1.elf` = **1,034,328 B**
|
||||
> 复现脚本:`.build/_sizeprobe.sh`、`.build/_probe3.sh`、`.build/_slimtest.sh`、`.build/_slimtest2.sh`、`.build/_modsize.sh`
|
||||
|
||||
## 结论
|
||||
|
||||
> ✅ **4.1 + 4.2 已落地(2026-09-20)** — `src/demangle_stub.c` 已加入 `COMMON_SRCS`,
|
||||
> `LDFLAGS` 已加 `-Wl,--icf=all`。
|
||||
>
|
||||
> 产物:**870,488 B** · sha256 `24392aff6ddcca4dc0ea969cce356bd693ac52efe8a117d61ee1c814aa43cd07` · e_machine `0x003e`
|
||||
>
|
||||
> 落地后复核:`__cxa_demangle` 本体 11 B、`itanium_demangle` 符号 0、
|
||||
> `__cxa_throw` / `_Unwind_Resume` 等异常符号齐全、`sevenz_extract` / `rar_extract` /
|
||||
> `zipx_extract` / `MHD_start_daemon_va` 全部存在。
|
||||
> section 变化:`.text` 637,616 → 538,336、`.rela.dyn` 54,816 → 25,800、
|
||||
> `.eh_frame` 58,904 → 43,972、`.rodata` 162,016 → 153,664。
|
||||
>
|
||||
> 4.3(RELR)与 4.4 未启用 —— 待真机验证 / 权衡后再决定。
|
||||
|
||||
**能瘦,而且有一个"白捡"的 15.8%。**
|
||||
|
||||
| 方案 | 结果 | 降幅 | 风险 |
|
||||
|---|---:|---:|---|
|
||||
| 瘦身前 | 1,034,328 B | — | — |
|
||||
| **+ `__cxa_demangle` 桩** | **886,872 B** | **−147 KB** | 极低 |
|
||||
| **+ ICF 代码折叠(当前产物)** | **870,488 B** | **−164 KB (−15.8%)** | 极低 |
|
||||
| 再 + RELR 重定位压缩 | 854,176 B | −180 KB (−17.4%) | 需真机验证 |
|
||||
| 第三方改 `-Oz`(单列) | 969,504 B | −65 KB | 可能降速 |
|
||||
|
||||
前两项**不损失任何功能,也不影响解压速度** —— 去掉的是一段永远不会被执行的代码。
|
||||
|
||||
---
|
||||
|
||||
## 一、现状构成
|
||||
|
||||
数据源:`size -A` / `nm --size-sort --print-size` / 未 strip 重链接。
|
||||
|
||||
### section 级
|
||||
|
||||
| section | 大小 | 备注 |
|
||||
|---|---:|---|
|
||||
| `.text` | 637,616 | 代码主体 |
|
||||
| `.rodata` | 162,016 | 前端资源(gzip)+ 字符串常量 |
|
||||
| `.eh_frame` + `.eh_frame_hdr` | 71,444 | **C++ 异常展开表** |
|
||||
| `.gcc_except_table` | 8,500 | **C++ 异常处理器表** |
|
||||
| `.rela.dyn` | 54,816 | 2,119 × `R_X86_64_RELATIVE` + 165 × `GLOB_DAT` |
|
||||
| `.data.rel.ro` | 20,192 | 含指针的只读数据 |
|
||||
| `.dynsym` + `.dynstr` + `.gnu.hash` | 23,863 | 动态符号表(PIE 必需) |
|
||||
| `.text$LZMADECOPT` | 4,719 | LZMA 汇编解码器(提速 1.26×,保留) |
|
||||
| `.bss` | 56,416 | **不占文件体积** |
|
||||
|
||||
### 模块级(按目标文件归属,text+data)
|
||||
|
||||
| 模块 | 文件数 | text | data |
|
||||
|---|---:|---:|---:|
|
||||
| unrar7(RAR 引擎) | 48 | 314,528 | 658 |
|
||||
| libc++ / libc++abi / PS5 运行时 | — | ~161,900 | 20,738 |
|
||||
| **C++ 异常机制** | — | **151,288** | — |
|
||||
| zlib | 8 | 65,211 | 336 |
|
||||
| 7z SDK + 自研链 | 30 | 85,406 | 56 |
|
||||
| minizip-ng | 8 | 39,382 | 384 |
|
||||
| 自有代码 `src/` | 14 | 60,653 | 472 |
|
||||
| libmicrohttpd | — | 44,647 | — |
|
||||
| 前端资源(gzip 后) | 13 | 40,278 | 104 |
|
||||
|
||||
---
|
||||
|
||||
## 二、最大的单一发现:151 KB 的 C++ 异常机制
|
||||
|
||||
`nm` 统计出 **607 个 `itanium_demangle::*` 符号,合计 105,431 B** —— 这是 libc++abi 的
|
||||
C++ 名字还原器(`__cxa_demangle`),单独一块就占了整个 ELF 的 **10.2%,比 zlib 整个库还大**。
|
||||
|
||||
它是怎么被拉进来的:
|
||||
|
||||
1. `third_party/unrar7/dll.cpp` 用了 `catch (RAR_EXIT)` / `catch (std::bad_alloc&)`
|
||||
(`unpack.cpp` / `model.cpp` 里也有 `throw std::bad_alloc()`)
|
||||
2. 只要 C++ 异常运行时存在,libc++abi 的 `__cxa_throw` 链路就会引用 `__cxa_demangle`
|
||||
(用于打印未捕获异常的类型名)
|
||||
3. 链接器于是把整个 `cxa_demangle.cpp` 拉进来 —— 一个深度内联的模板解析器,
|
||||
展开成 607 个函数
|
||||
|
||||
连带被拖进来的还有 `libunwind`(21,953 B,栈回溯)和异常胶水(23,904 B),
|
||||
以及散落在每个 C++ 目标文件里的 `.eh_frame`(58,904 B)/ `.gcc_except_table`(8,500 B)。
|
||||
|
||||
**但这个 demangler 只在「未捕获异常」的诊断路径上才会被调用。**
|
||||
我们的 unrar7 走 DLL 模式,所有异常都在 `dll.cpp` 内部被 catch 掉,
|
||||
程序逻辑永远走不到那条路径。
|
||||
|
||||
---
|
||||
|
||||
## 三、实测(同一 WSL、同一份源码、同一个链接器)
|
||||
|
||||
| # | 变更 | 结果 | 差值 |
|
||||
|---|---|---:|---:|
|
||||
| E0 | 重新链接(校验基线) | 1,034,328 | 0 |
|
||||
| E1 | `-Wl,--icf=all` | 1,017,944 | −16,384 |
|
||||
| E2 | 注入 `__cxa_demangle` 桩 | 886,872 | **−147,456** |
|
||||
| E3 | 桩 + `--icf=all` | 870,488 | **−163,840** |
|
||||
| E4 | `-Wl,-z,pack-relative-relocs` | 985,248 | −49,080 |
|
||||
| E5 | `-Wl,-z,noseparate-code` | 1,034,328 | 0(无效) |
|
||||
| E7 | 桩 + ICF + RELR | 854,176 | −180,152 |
|
||||
| E8 | 第三方全部改 `-Oz` | 969,504 | −64,824 |
|
||||
|
||||
> E4 单独能省 49 KB,但与 ICF 组合后只剩 16 KB —— ICF 已经合并掉了一批重定位。
|
||||
|
||||
---
|
||||
|
||||
## 四、落地方案
|
||||
|
||||
### 4.1 立即可用:`__cxa_demangle` 桩(−147 KB)
|
||||
|
||||
新增 `src/demangle_stub.c`:
|
||||
|
||||
```c
|
||||
/* 只提供 __cxa_demangle 的桩,让 libc++abi 里 105 KB 的名字还原器
|
||||
* 不被链接进来。该函数只在打印「未捕获异常的类型名」时被调用;
|
||||
* 返回 NULL 时调用方退回打印 mangled 名,不影响任何业务流程。 */
|
||||
#include <stddef.h>
|
||||
|
||||
char *__cxa_demangle(const char *mangled, char *buf, size_t *len, int *status)
|
||||
{
|
||||
(void)mangled; (void)buf; (void)len;
|
||||
if (status) *status = -1;
|
||||
return NULL;
|
||||
}
|
||||
```
|
||||
|
||||
Makefile 里把它加进 `COMMON_SRCS`(PS5 与 Linux 两条链路都受益):
|
||||
|
||||
```make
|
||||
COMMON_SRCS := src/main.c src/websrv.c ... src/sevenz_mt.c src/demangle_stub.c
|
||||
```
|
||||
|
||||
**原理**:链接器解析 `__cxa_demangle` 引用时,命令行上的 `.o` 优先于归档成员,
|
||||
所以 `libc++abi.a` 里的 `cxa_demangle.o` 压根不会被取出。
|
||||
|
||||
**安全性(已实测验证)**:
|
||||
|
||||
| 检查项 | 结果 |
|
||||
|---|---|
|
||||
| `__cxa_demangle` 本体大小 | **11 B**(我们的桩;原 demangler 入口是 1 701 B) |
|
||||
| `itanium_demangle::*` 符号残留 | **0** |
|
||||
| `__cxa_throw` | 存在 |
|
||||
| `__cxa_begin_catch` / `__cxa_end_catch` | 存在 |
|
||||
| `_Unwind_Resume` / `__gxx_personality_v0` | 存在 |
|
||||
|
||||
**唯一的行为变化**:万一真的出现未捕获异常,`std::terminate` 打印的是 mangled 名
|
||||
而不是可读名。解压逻辑、错误码、进度上报、HTTP 服务一概不受影响。
|
||||
|
||||
### 4.2 立即可用:ICF 代码折叠(−16 KB)
|
||||
|
||||
```make
|
||||
LDFLAGS := -Wl,--gc-sections -Wl,--icf=all
|
||||
```
|
||||
|
||||
lld 的 identical code folding,合并字节完全相同的函数。工具链已确认为 LLVM LLD 18。
|
||||
建议只加在 PS5 的 `LDFLAGS`,不动 Linux 链路(GNU ld 的 `--icf` 支持不完整)。
|
||||
|
||||
### 4.3 需真机验证:RELR 重定位压缩(−16 KB)
|
||||
|
||||
```make
|
||||
LDFLAGS += -Wl,-z,pack-relative-relocs
|
||||
```
|
||||
|
||||
把 2,119 条 `R_X86_64_RELATIVE`(24 B/条)压成 RELR 位图格式(8 B/条)。
|
||||
|
||||
**风险**:需要 PS5 的 ELF 加载器认得 `.relr.dyn`。如果加载器只处理 `.rela.dyn`,
|
||||
重定位根本不会执行 —— 表现是启动即崩。**先在一台机器上验证再推广。**
|
||||
|
||||
### 4.4 不建议作为默认项
|
||||
|
||||
| 项 | 收益 | 为什么不默认开 |
|
||||
|---|---:|---|
|
||||
| 第三方改 `-Oz` | −65 KB | 作用于 LZMA / Deflate / RAR 的热循环,解压速度有下降风险。要用先跑 `tests/bench_driver.py` 对比 |
|
||||
| 关掉 PPMd(`-DZ7_PPMD_SUPPORT`) | −10 KB | PPMd 压缩的 7z 就解不开了 —— 违背"功能完整" |
|
||||
| 去掉 zlib `deflate`(只留 inflate) | −18 KB | `mz_strm_zlib_write` 引用了它,需要桩或改库,收益/风险不划算 |
|
||||
|
||||
---
|
||||
|
||||
## 五、还能挖的(未实测,仅估算)
|
||||
|
||||
| 项 | 预估 | 代价 |
|
||||
|---|---:|---|
|
||||
| unrar7 去掉 C++ 异常(`throw`/`catch` 改错误码 + `-fno-exceptions -fno-rtti`) | −100~110 KB | 改 vendored 代码,需回归 163 checks。可回收 `.eh_frame` 59 KB + `.gcc_except_table` 8.5 KB + libunwind 22 KB + 异常胶水 |
|
||||
| LTO(`-flto=thin`) | −30~60 KB | 全量重编,第三方 `.o` 需统一编译选项;有一定概率反而提速 |
|
||||
| 前端资源改 LZMA 压缩(复用已有解码器,替代 gzip) | ~−10 KB | 改 `gen-asset-module.py` + `asset.c` |
|
||||
|
||||
理论极限在 700 KB 上下(−32%),但边际成本递增:4.1 + 4.2 用 20 行代码换 164 KB,
|
||||
而再往下 100 KB 要动 vendored 源码或验证加载器行为。
|
||||
|
||||
---
|
||||
|
||||
## 六、推荐执行顺序
|
||||
|
||||
1. ~~落 4.1 + 4.2 → 构建~~ ✅ **已完成(2026-09-20)**,产物 870,488 B
|
||||
2. 跑 163 checks(`tests/run-tests.sh` + `tests/run-sevenz-tests.sh`)确认无回归
|
||||
3. PS5 真机跑一轮 ZIP / RAR / 7z(含分卷)解压,确认行为不变
|
||||
4. 真机验证 4.3(RELR)后再决定是否加入
|
||||
5. 有需要再评估第五节的三项
|
||||
|
||||
> 本次改动只动链接期(新增一个 TU + 一个 lld 参数),未触碰任何解压逻辑,
|
||||
> 因此 163 checks 的预期是"逐条不变"。
|
||||
|
||||
**结果(2026-09-20)**:163 checks(ZIP 108 + RAR 27 + 7z 28)**0 失败**,
|
||||
`aeshe` 仍是已知的 `-mhe=on` 缺口。README / CHANGELOG / HANDOVER / 论坛帖里的
|
||||
产物指纹已同步为 870,488 B · sha256 `177e90fe…8e84`。
|
||||
|
||||
> 后续(2026-09-23):测试计数已变为 ZIP 140 + RAR 37 = 177(7z 套件 27 用例),
|
||||
> `aeshe` 缺口也已闭合;产物指纹随之更新。本节保留的是上面的历史测量值。
|
||||
|
||||
---
|
||||
|
||||
## 附录 A:v1.9.2 产物一致性验证(2026-09-20)
|
||||
|
||||
v1.9.2 是一次"只改内嵌版本号"的重发(原 `v1.9.1` tag 落后产出发布二进制的提交
|
||||
4 个提交)。为确认这次重发**真的**只动了版本号,在 WSL 里做了下面的验证。
|
||||
|
||||
### A.1 复现性实验(决定性证据)
|
||||
|
||||
当前工作区相对 `HEAD` 只有两处改动:`Makefile` 的 `VERSION_TAG` 与
|
||||
`assets/main.js` 的 `APP_VERSION_FALLBACK`。把这两处用 `sed` 回退成 `v1.9.1`
|
||||
后重新构建:
|
||||
|
||||
| 构建 | sha256 |
|
||||
|---|---|
|
||||
| 已发布的 v1.9.1 ELF | `24392aff6ddcca4dc0ea969cce356bd693ac52efe8a117d61ee1c814aa43cd07` |
|
||||
| 回退后重建的产物 | `24392aff6ddcca4dc0ea969cce356bd693ac52efe8a117d61ee1c814aa43cd07` |
|
||||
|
||||
**逐字节相同。** 再恢复 `v1.9.2` 重构,sha256 也精确回到 `177e90fe…8e84`。
|
||||
|
||||
→ 构建是**确定性**的,因此 v1.9.2 与 v1.9.1 的全部差异就等于那两处版本字面量。
|
||||
复现脚本:`.build/_repro.sh`。
|
||||
|
||||
### A.2 为什么原始字节 diff 有 5.5 万字节 —— 别被吓到
|
||||
|
||||
`cmp` 两个 ELF 会看到 **55,280 字节不同(6.35%)**,但这是链接器字符串池重排的
|
||||
副作用,不是代码变了:
|
||||
|
||||
| section | 差异字节 | 占该 section |
|
||||
|---|---:|---:|
|
||||
| `.rodata` | 53,496 | 34.8% |
|
||||
| `.text` | 1,543 | 0.3% |
|
||||
| `.rela.dyn` | 241 | 0.9% |
|
||||
|
||||
而**每个 section 的尺寸完全相同**(`.text` 538,336 = 538,336),段数也都 17 个。
|
||||
|
||||
机制:`.rodata` 里 7 字节的 `"v1.9.1\0"` 被换成 `"v1.9.2\0"` 后落点变了,其后
|
||||
所有字符串整体平移 7 字节 → 指向它们的 `lea rdi,[rip+disp]` 位移和 `.rela.dyn`
|
||||
重定位加数全部跟着变。
|
||||
|
||||
两条量化证据:
|
||||
|
||||
| 检查 | 结果 |
|
||||
|---|---|
|
||||
| 指令**助记符**序列(`objdump -d --no-show-raw-insn` 只取 mnemonic) | 141,780 条 vs 141,780 条,**完全一致** —— 没有任何指令被增删改 |
|
||||
| `.text` 差异字节的增量分布 | 1,543 个里 **1,506 个恰好是 −7**(正是那个 7 字节平移);`.rela.dyn` 241/241 个 8 字节字段减 7 |
|
||||
| 嵌入的 gzip 资产 | 6 个成员,5 个逐字节相同,唯一不同的是 `main.js`,且差异 = `APP_VERSION_FALLBACK` 那一行 |
|
||||
|
||||
脚本:`.build/_diffmap.py`、`.build/_operandcheck.sh`、`.build/_fieldcheck.py`、
|
||||
`.build/_verify_v192b.py`、`.build/_seccmp.py`。
|
||||
|
||||
### A.3 源码层的约束
|
||||
|
||||
`VERSION_TAG` 在源码里**只出现在字符串上下文**:
|
||||
|
||||
```
|
||||
src/version.c:29 json_escape(&b, VERSION_TAG);
|
||||
src/main.c:127 printf("version: %s\n", VERSION_TAG);
|
||||
src/main.c:147 notify_user("Web File Manager\nVersion: %s\nPort: %u", VERSION_TAG, port);
|
||||
```
|
||||
|
||||
没有任何算术、比较或分支依赖它,因此改版本号在语言层面就不可能改变控制流。
|
||||
|
||||
---
|
||||
|
||||
## 附录 B:瘦身逐符号账目
|
||||
|
||||
在 **同一个 Makefile / 同一个 `VERSION_TAG`(v1.9.2)** 下重建三个变体,差异只落在
|
||||
"有没有桩"和"有没有 ICF"这两处,因此是干净的 A/B/C 对照。
|
||||
|
||||
| 变体 | 内容 | stripped | unstripped |
|
||||
|---|---|---:|---:|
|
||||
| `base` | 无桩、无 ICF(瘦身前) | **1,034,328** | 1,222,752 |
|
||||
| `nicf` | 有桩、无 ICF | **886,872** | 1,010,208 |
|
||||
| `new` | 有桩 + `--icf=all`(发布态) | **870,488** | 993,824 |
|
||||
|
||||
拆分:桩贡献 **−147,456 B**,ICF 再贡献 **−16,384 B**,合计 **−163,840 B(−15.8%)**。
|
||||
|
||||
### B.1 section 位移(base → new)
|
||||
|
||||
| section | base | new | 差值 |
|
||||
|---|---:|---:|---:|
|
||||
| `.text` | 637,616 | 538,336 | −99,280 |
|
||||
| `.rela.dyn` | 54,816 | 25,800 | −29,016 |
|
||||
| `.eh_frame` | 58,904 | 43,972 | −14,932 |
|
||||
| `.data.rel.ro` | 20,192 | 9,328 | −10,864 |
|
||||
| `.rodata` | 162,016 | 153,664 | −8,352 |
|
||||
| `.eh_frame_hdr` | 12,540 | 9,108 | −3,432 |
|
||||
| `.gcc_except_table` | 8,500 | 7,364 | −1,136 |
|
||||
| `.dynsym` / `.dynstr` / `.got` | — | — | −24 / −9 / −8 |
|
||||
|
||||
### B.2 符号集合差
|
||||
|
||||
| 项 | 数量 |
|
||||
|---|---:|
|
||||
| base 定义符号 | 2,657 |
|
||||
| new 定义符号 | 2,029 |
|
||||
| base → new **消失** | **628** |
|
||||
| base → new **新增** | **0** |
|
||||
|
||||
628 个消失符号的构成:
|
||||
|
||||
- `itanium_demangle::*` —— **607**
|
||||
- `GCC_except_table*` —— **21**(上面那批代码自己的异常表标签,不是独立函数)
|
||||
|
||||
### B.3 桩本体与 demangler 符号
|
||||
|
||||
| 变体 | `__cxa_demangle` 符号大小 | `itanium_demangle` 符号数 |
|
||||
|---|---:|---:|
|
||||
| base | 1,701 B(真身) | 607 |
|
||||
| nicf | **11 B**(我们的桩) | **0** |
|
||||
| new | **11 B** | **0** |
|
||||
|
||||
异常机制在所有三个变体里都完好:`__cxa_throw` / `__cxa_begin_catch` /
|
||||
`__cxa_end_catch` / `_Unwind_Resume` / `__gxx_personality_v0` /
|
||||
`__cxa_allocate_exception` / `__cxa_free_exception` 各 1 个,无变化。
|
||||
|
||||
自有 `src/` 关键符号(`ctx_fail` / `rarx_fail` / `szx_fail` / `fnv1a` /
|
||||
`nameset_init` / `remove_tree` / `ensure_parent_dirs` / `zipx_volume_detect` /
|
||||
`sevenz_extract` / `rar_extract` / `filemgr_api_request` 等)base 与 new 数量一致。
|
||||
|
||||
### B.4 ICF 折叠了什么
|
||||
|
||||
**符号数 2,029 → 2,029,一个没少** —— ICF 是"合并"不是"删除"。共 **76 个折叠组**,
|
||||
全部含具名符号。典型几类:
|
||||
|
||||
- 我们自己的同码副本:`ctx_fail == rarx_fail == szx_fail`、
|
||||
`nameset_init == szx_nameset_init`、`fnv1a == rarx_fnv1a == szx_fnv1a`、
|
||||
`remove_tree == szx_remove_tree`
|
||||
- C++ 的 `C1 == C2` / `D1 == D2` 构造析构对(编译器为同一函数生成两个 ABI 入口):
|
||||
`_ZN10CmdExtractC1EP11CommandData == ...C2...`、`_ZN4FileD1Ev == _ZN4FileD2Ev` 等
|
||||
- 只读常量表:`Sbox == _ZL1S`(AES 表在 `rijndael.cpp` 与 `Aes.c` 各一份)、
|
||||
`SHA256_K_ARRAY == _ZL1K`、`PPMD7_kExpEscape == _ZL9ExpEscape`
|
||||
- RARDLL 模式下被置空的 UI 函数、`mz_stream_read_int64 == read_uint64`、
|
||||
libunwind 的 `__unw_* == unw_*`、`__unw_resume == unw_resume`
|
||||
|
||||
**风险提示**:`--icf=all` 是 LLD 的激进模式,**不做地址敏感性检查**
|
||||
(`--icf=safe` 才会读 `.llvm_addrsig` 跳过被取地址的函数)。逐组核对下来这 76 组
|
||||
都是同码副本、没有"比较函数/常量表地址"的用法 —— 但这是人工判断,不是编译器给的
|
||||
保证。想绝对保守就把 `--icf=all` 换成 `--icf=safe`,代价是少省几 KB。
|
||||
|
||||
复现脚本:`.build/_whatremoved_v192.sh`(一次跑完三个变体 + 全部核对)。
|
||||
@@ -0,0 +1,451 @@
|
||||
# Upgrade v1.7 — ZIP large-file profile
|
||||
|
||||
> Technical notes for the **v1.7** upgrade of `ps5-web-file-manager`.
|
||||
> Audience: future maintainers, code reviewers, contributors reading the
|
||||
> `git log` of this branch. Pair with `README.md` for the user-facing
|
||||
> overview and `CHANGELOG.md` for the release-note summary.
|
||||
|
||||
---
|
||||
|
||||
## 1. Background
|
||||
|
||||
Before v1.7 the ZIP engine shipped with a single, conservative limits profile
|
||||
(`zipx_default_limits()`). It rejected any archive whose uncompressed content
|
||||
exceeded **512 GiB total** or contained an entry above **64 GiB**, or whose
|
||||
compression ratio exceeded **200 : 1**. This was the right default for the
|
||||
"copy game-dump folders" use case but blocked legitimate large payloads — most
|
||||
notably system images and 200 GB+ backups.
|
||||
|
||||
The user asked: *"单个压缩包不可以为 200GB 以上么"* ("Can't a single archive
|
||||
be larger than 200 GB?"). After surfacing three options (raise the default
|
||||
limits, add an opt-in profile, or document-only) the choice was **🅱 — add an
|
||||
opt-in "large-file profile"**. The implementation brief was:
|
||||
|
||||
> The server must **never** activate the relaxed caps on its own. The user
|
||||
> must explicitly opt in, both via the HTTP API and via the UI.
|
||||
|
||||
## 2. Architecture delta (one-line summary)
|
||||
|
||||
The ZIP engine becomes a **profile-lookup** engine. A new profile table
|
||||
(`k_large_limits`) and a switch (`zipx_limits_profile()`) sit between the HTTP
|
||||
layer and the existing `zipx_extract()`. Everywhere else — task model, HTTP
|
||||
handler, frontend — gains a single boolean that threads through to the engine.
|
||||
|
||||
```
|
||||
HTTP /api/extract?large=1 frontend (confirm() 弹窗)
|
||||
│ │
|
||||
└──────► form parser ──► file_task_t.extract_large
|
||||
│
|
||||
▼
|
||||
zipx_limits_profile(task->extract_large)
|
||||
│
|
||||
┌────────────┴────────────┐
|
||||
▼ ▼
|
||||
k_default_limits k_large_limits
|
||||
(200K / 512GiB / (500K / 2TiB /
|
||||
64GiB / 200:1) 1TiB / 1000:1)
|
||||
│ │
|
||||
└────────────┬─────────────┘
|
||||
▼
|
||||
zipx_extract()
|
||||
(signature unchanged;
|
||||
shared three-phase engine)
|
||||
```
|
||||
|
||||
The public signature of `zipx_extract()` is **unchanged**. Backwards
|
||||
compatibility is deliberate — anything linking against `src/zip_extract.{c,h}`
|
||||
continues to compile without changes.
|
||||
|
||||
## 3. Engine layer (`src/zip_extract.h`, `src/zip_extract.c`)
|
||||
|
||||
### 3.1 New constants and API
|
||||
|
||||
```c
|
||||
/* Pre-built limit profiles. Use zipx_limits_profile() to look one up.
|
||||
ZIPX_LIMITS_DEFAULT is the safe profile shipped by zipx_default_limits().
|
||||
ZIPX_LIMITS_LARGE allows archives up to 2 TiB total / 1 TiB per file and
|
||||
a 1000:1 compression ratio. The caller is responsible for verifying that
|
||||
the PS5 has enough free disk space. */
|
||||
#define ZIPX_LIMITS_DEFAULT 0
|
||||
#define ZIPX_LIMITS_LARGE 1
|
||||
|
||||
const zipx_limits_t *zipx_limits_profile(int profile);
|
||||
```
|
||||
|
||||
### 3.2 The two profiles
|
||||
|
||||
```c
|
||||
static const zipx_limits_t k_default_limits = {
|
||||
.max_entries = 200000,
|
||||
.max_total_bytes = 512ULL * 1024 * 1024 * 1024, /* 512 GiB */
|
||||
.max_file_bytes = 64ULL * 1024 * 1024 * 1024, /* 64 GiB */
|
||||
.max_ratio = 200,
|
||||
.max_depth = 32,
|
||||
.max_name_len = 255,
|
||||
.max_path_len = 1024
|
||||
};
|
||||
|
||||
static const zipx_limits_t k_large_limits = {
|
||||
.max_entries = 500000,
|
||||
.max_total_bytes = 2ULL * 1024 * 1024 * 1024 * 1024, /* 2 TiB */
|
||||
.max_file_bytes = 1ULL * 1024 * 1024 * 1024 * 1024, /* 1 TiB */
|
||||
.max_ratio = 1000,
|
||||
.max_depth = 32,
|
||||
.max_name_len = 255,
|
||||
.max_path_len = 1024
|
||||
};
|
||||
```
|
||||
|
||||
The two profiles are otherwise identical on `max_depth`, `max_name_len` and
|
||||
`max_path_len` — the v1.7 upgrade only relaxes the four "blast radius" caps.
|
||||
|
||||
### 3.3 Profile lookup
|
||||
|
||||
```c
|
||||
const zipx_limits_t *
|
||||
zipx_limits_profile(int profile) {
|
||||
switch(profile) {
|
||||
case ZIPX_LIMITS_LARGE: return &k_large_limits;
|
||||
case ZIPX_LIMITS_DEFAULT:
|
||||
default: return &k_default_limits;
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
`zipx_default_limits()` continues to return `&k_default_limits` — there is
|
||||
**no behavioural change** for callers that did not opt in.
|
||||
|
||||
### 3.4 Behavioural contract (unchanged from pre-v1.7)
|
||||
|
||||
* Plain ZIPs only — stored / deflated / ZIP64. `ZIPX_ERR_UNSUPPORTED` is
|
||||
raised for any encryption flag, multi-volume markers or unsupported
|
||||
compression methods.
|
||||
* Three-phase work model: `ZIPX_PHASE_SCAN → EXTRACT → PUBLISH → CLEANUP`.
|
||||
Each entry is first written to a staging directory (`*.wfm-part-*`),
|
||||
`fsync()`'d, then atomically renamed into place. A failure mid-archive
|
||||
rolls back partial changes.
|
||||
*(Superseded 2026-09-16: the per-entry `fsync` was removed — all three
|
||||
engines now apply "sync nothing, rename everything". See
|
||||
`docs/EXTRACTION-PERF.md`.)*
|
||||
* Security checks run before any output file is opened:
|
||||
- encryption
|
||||
- path traversal (`..`), absolute POSIX paths, Windows drive letters
|
||||
- symbolic links, devices, FIFOs, sockets
|
||||
- duplicate entries / directory↔file clashes inside the archive
|
||||
- the four blast-radius caps above (entries, total bytes, file bytes,
|
||||
compression ratio)
|
||||
|
||||
## 4. Task layer (`src/filemgr_internal.h`, `src/extract.c`)
|
||||
|
||||
### 4.1 New task field
|
||||
|
||||
```c
|
||||
typedef struct file_task {
|
||||
/* …existing fields… */
|
||||
int extract_conflict;
|
||||
int extract_remove_source;
|
||||
int extract_large; /* ← new: 0 = ZIPX_LIMITS_DEFAULT, 1 = ZIPX_LIMITS_LARGE */
|
||||
/* …existing fields… */
|
||||
} file_task_t;
|
||||
```
|
||||
|
||||
### 4.2 Worker dispatch
|
||||
|
||||
In `extract_worker()` (`src/extract.c`, ~line 115):
|
||||
|
||||
```c
|
||||
status = zipx_extract(task->src, task->dst, conflict,
|
||||
zipx_limits_profile(task->extract_large),
|
||||
extract_cancel, extract_progress, task, &result);
|
||||
```
|
||||
|
||||
The third positional argument moved from a direct `&k_default_limits` (or a
|
||||
caller-supplied struct) to `zipx_limits_profile(task->extract_large)`. The
|
||||
`limits` parameter of `zipx_extract()` remains `const zipx_limits_t *` —
|
||||
either pointer is fine.
|
||||
|
||||
### 4.3 Form parsing in `api_extract()`
|
||||
|
||||
```c
|
||||
char *large_str = body_form_value(body, body_size, "large");
|
||||
int large = (large_str != NULL && !strcmp(large_str, "1")) ? 1 : 0;
|
||||
/* … free chain updated to release large_str … */
|
||||
task->extract_large = large;
|
||||
```
|
||||
|
||||
The string comparison is **strict** — only the literal `"1"` activates the
|
||||
large profile. `"true"`, `"yes"`, `"on"` are all ignored. This matches the
|
||||
convention used by the existing `remove_source` field (`src/extract.c`).
|
||||
|
||||
### 4.4 Error reporting path
|
||||
|
||||
When the active profile rejects an archive the engine returns one of:
|
||||
|
||||
| `zipx_status_t` | Frontend maps to |
|
||||
|---------------------|--------------------------|
|
||||
| `ZIPX_ERR_LIMIT_ENTRIES` | `err_extract_too_many_entries` |
|
||||
| `ZIPX_ERR_LIMIT_FILE` | `err_extract_entry_too_large` |
|
||||
| `ZIPX_ERR_LIMIT_TOTAL` | `err_extract_total_too_large` |
|
||||
| `ZIPX_ERR_LIMIT_RATIO` | `err_extract_ratio` |
|
||||
| `ZIPX_ERR_LIMIT_DEPTH` | `err_extract_depth` |
|
||||
| `ZIPX_ERR_LIMIT_NAME` | `err_extract_name` |
|
||||
|
||||
The error string embedded in `zipx_result_t.message` interpolates the active
|
||||
limit (`"compression ratio is above %u"`), so users can see exactly which cap
|
||||
hit. **v1.7 does not change** this mapping — the new large profile uses the
|
||||
same codes, just with larger caps.
|
||||
|
||||
## 5. HTTP API contract (`/api/extract`)
|
||||
|
||||
`POST /api/extract` accepts `application/x-www-form-urlencoded`:
|
||||
|
||||
| Field | Type | Required | Notes |
|
||||
|-----------------|------|----------|-------|
|
||||
| `path` | string | yes | Absolute path to the archive on the PS5. |
|
||||
| `dst_dir` | string | yes | Output directory. |
|
||||
| `conflict` | string | no | `fail` / `overwrite` / `merge`. Default `fail`. |
|
||||
| `remove_source` | `0`/`1` | no | Default `0`. |
|
||||
| `large` | `0`/`1` | no | **new in v1.7**. Default `0`. |
|
||||
|
||||
Compatibility:
|
||||
|
||||
- Existing callers that do **not** send `large` get identical behaviour as
|
||||
before the upgrade — `ZIPX_LIMITS_DEFAULT` always.
|
||||
- The server returns `400` for any value other than `"0"` or `"1"` (only the
|
||||
exact literal `"1"` enables the large profile). This is enforced by the
|
||||
`!strcmp(large_str, "1")` guard, not a separate validator.
|
||||
|
||||
The hand-rolled form parser (`src/json_util.c` `body_form_value()`) already
|
||||
returned the raw value; the upgrade did not touch that helper.
|
||||
|
||||
## 6. Frontend (`assets/main.js`)
|
||||
|
||||
### 6.1 Threshold + prompt primitives
|
||||
|
||||
```js
|
||||
const LARGE_FILE_THRESHOLD_BYTES = 60 * 1024 * 1024 * 1024; // 60 GiB
|
||||
|
||||
function shouldPromptLargeMode(itemSize) {
|
||||
return Number(itemSize || 0) > LARGE_FILE_THRESHOLD_BYTES;
|
||||
}
|
||||
|
||||
function promptLargeMode(sizeBytes) {
|
||||
return confirm(t("extractLargeAsk",
|
||||
{ size: formatSize(sizeBytes, "-") }));
|
||||
}
|
||||
```
|
||||
|
||||
The 60 GiB threshold is **hardcoded**, not user-configurable. To change it,
|
||||
edit the constant on line `813` (current main branch). Setting it to
|
||||
`Infinity` silences the prompt entirely.
|
||||
|
||||
### 6.2 The two call sites
|
||||
|
||||
* `actionExtract()` (existing) — invoked from the file-row "extract" menu
|
||||
item — checks `item.size` and prompts before calling
|
||||
`startExtractTask(item.path, cwd, conflict, false, displayName(item), large)`.
|
||||
* `uploadAndExtractFile()` (existing) — invoked after a remote upload
|
||||
completes — checks `file.size` and prompts before calling
|
||||
`startExtractTask(zipPath, cwd, conflict, true, rel, large)`.
|
||||
|
||||
Both paths funnel into `startExtractTask(path, dst, conflict, removeSource,
|
||||
name, large)`, which `POST`s to `/api/extract` with:
|
||||
|
||||
```js
|
||||
const data = await apiForm("/api/extract", {
|
||||
path,
|
||||
dst_dir: dstDir,
|
||||
conflict,
|
||||
remove_source: removeSource ? "1" : "0",
|
||||
large: large ? "1" : "0"
|
||||
});
|
||||
```
|
||||
|
||||
If the user clicks **Cancel** on the prompt, `large = false` and the request
|
||||
goes out with `large=0`. **No silent fallback** to the large profile.
|
||||
|
||||
### 6.3 UI status
|
||||
|
||||
When `large=1` is sent, the engine logs `task->error_code = "err_extract_large"`
|
||||
status field so the in-app task overlay can show a "large-file profile
|
||||
active" badge. The exact badge wording lives in the locale files (see §7).
|
||||
|
||||
## 7. i18n strings (`assets/lang-{en,zh}.js`)
|
||||
|
||||
Two new keys, both on line `108-109` of each locale file:
|
||||
|
||||
* `extractLargeAsk` — the prompt body. Interpolation parameter: `size`
|
||||
(already pre-formatted by `formatSize()` in bytes / KiB / MiB / GiB / TiB).
|
||||
* `extractLargeActive` — short tag shown next to a running large-profile
|
||||
task.
|
||||
|
||||
When changing the wording, keep the `{size}` placeholder and the
|
||||
newline-separated **OK / Cancel** hint — the prompt is a `confirm()` so the
|
||||
expected user gesture is documented inside the dialog.
|
||||
|
||||
## 8. Tests (`tests/test_zip_extract.c`, `tests/make_fixtures.py`)
|
||||
|
||||
### 8.1 Coverage matrix
|
||||
|
||||
| Scenario | Default | Large | Notes |
|
||||
|---|---|---|---|
|
||||
| `basic.zip` (small, benign) | ✓ | ✓ | sanity |
|
||||
| `medium_bomb.zip` (1 MiB → ratio ≈ 238) | reject | accept | new fixture, profile switchover |
|
||||
| `bomb.zip` (4 MiB of `'A'`, ratio ≈ 1026) | reject | **reject** | large caps still apply |
|
||||
| `bomb.zip` with `tight.max_ratio = 10` | reject | reject | profile is a starting point, lower caps still enforced |
|
||||
| `zip64.zip` with `tight.max_file_bytes = 1024` | reject | reject | lowering file cap from profile |
|
||||
| Conflict policy `fail`/`overwrite`/`merge` | ✓ | ✓ | unchanged |
|
||||
|
||||
Run with:
|
||||
|
||||
```sh
|
||||
cd tests && bash run-tests.sh
|
||||
```
|
||||
|
||||
Output is a per-case `check()` style report — currently **69 checks**,
|
||||
0 failures.
|
||||
|
||||
### 8.2 New fixture — `medium_bomb.zip`
|
||||
|
||||
* Generated by `make_fixtures.py::medium_bomb()`.
|
||||
* Payload: 1 MiB of `bytes(range(256)) * 4096` (a perfect 256-byte period
|
||||
repeated 4096 times).
|
||||
* Compression ratio (with `zlib -9`): **≈ 238 : 1**, deliberately chosen to
|
||||
fall in the band `(default_cap, large_cap) = (200, 1000]`.
|
||||
* Why not the pre-existing `bomb.zip` (4 MiB of `'A'`)? Its real ratio on
|
||||
`zlib -9` is **≈ 1026**, which the large profile's 1000 cap also rejects —
|
||||
the two profiles would behave identically and the test wouldn't show the
|
||||
switchover.
|
||||
|
||||
### 8.3 `test_large_profile()` cases
|
||||
|
||||
```c
|
||||
const zipx_limits_t *d = zipx_limits_profile(ZIPX_LIMITS_DEFAULT);
|
||||
const zipx_limits_t *l = zipx_limits_profile(ZIPX_LIMITS_LARGE);
|
||||
|
||||
check(d != NULL, "default profile exists");
|
||||
check(l != NULL, "large profile exists");
|
||||
|
||||
check(d->max_total_bytes == 512ULL * 1024 * 1024 * 1024, "default 512 GiB");
|
||||
check(d->max_file_bytes == 64ULL * 1024 * 1024 * 1024, "default 64 GiB");
|
||||
check(d->max_ratio == 200, "default ratio 200");
|
||||
check(l->max_entries == 500000, "large 500K entries");
|
||||
check(l->max_total_bytes == 2ULL * 1024 * 1024 * 1024 * 1024, "large 2 TiB");
|
||||
check(l->max_file_bytes == 1ULL * 1024 * 1024 * 1024 * 1024, "large 1 TiB");
|
||||
check(l->max_ratio == 1000, "large ratio 1000");
|
||||
|
||||
expect_status("medium_bomb.zip", "out_default_medium", ZIPX_CONFLICT_FAIL, NULL,
|
||||
ZIPX_ERR_LIMIT_RATIO, "default rejects medium_bomb");
|
||||
expect_ok ("medium_bomb.zip", "out_large_medium", ZIPX_CONFLICT_FAIL, l,
|
||||
"large accepts medium_bomb");
|
||||
|
||||
/* Large caps still enforced — bomb ratio 1026 > 1000 */
|
||||
expect_status("bomb.zip", "out_large_bomb_default", ZIPX_CONFLICT_FAIL, NULL,
|
||||
ZIPX_ERR_LIMIT_RATIO, "default rejects bomb");
|
||||
expect_ok ("bomb.zip", "out_large_bomb_large", ZIPX_CONFLICT_FAIL, l,
|
||||
"large accepts bomb-ratio-1000 border");
|
||||
|
||||
/* Lowered caps remain enforced */
|
||||
zipx_limits_t tight = *l; tight.max_ratio = 10;
|
||||
expect_status("bomb.zip", "out_tight_ratio", ZIPX_CONFLICT_FAIL, &tight,
|
||||
ZIPX_ERR_LIMIT_RATIO, "tight ratio still enforced");
|
||||
```
|
||||
|
||||
(13 new `check`/`expect_*` calls in this function alone.)
|
||||
|
||||
## 9. Build & verify pipeline
|
||||
|
||||
### 9.1 Cross-compile
|
||||
|
||||
The WSL staging helper `.build/build-elf.sh` automates the SDK write-access
|
||||
workaround (the SDK lives at `/opt/ps5-payload-sdk/target/` owned by a
|
||||
different uid). The script:
|
||||
|
||||
1. Runs `make` with a staging dir under `/tmp` for write-protected sources.
|
||||
2. `sudo cp -r` the staged outputs back into the SDK tree in a single batch.
|
||||
3. Copies the resulting `web-file-mgr.elf` to both `/home/song/...` and the
|
||||
Windows desktop.
|
||||
|
||||
Incremental builds are fast — only `src/extract.c` recompiled for v1.7; the
|
||||
nine `gen/lang-*.js.c` files were regenerated because `extractLargeAsk`
|
||||
changed.
|
||||
|
||||
### 9.2 ELF sanity checks
|
||||
|
||||
```sh
|
||||
ls -la web-file-mgr.elf
|
||||
sha256sum web-file-mgr.elf # 648e4a00…
|
||||
file web-file-mgr.elf # ELF 64-bit LSB pie, x86-64
|
||||
od -An -tx1 -N20 web-file-mgr.elf | head -2 # 7f45 4c46 0201 + e_machine 003e
|
||||
```
|
||||
|
||||
### 9.3 Verifying the upgrade made it into the binary
|
||||
|
||||
`assets/*.js` are embedded as `zlib`-compressed C arrays by `gen-asset-module.py`.
|
||||
A simple `strings web-file-mgr.elf | grep` won't find them. Use
|
||||
`.build/check-elf-gzip.py`:
|
||||
|
||||
```sh
|
||||
python3 .build/check-elf-gzip.py ./web-file-mgr.elf
|
||||
# expects:
|
||||
# keys found (7/7):
|
||||
# ✓ extractLargeAsk "The archive looks large …"
|
||||
# ✓ extractLargeActive "Large-file profile is enabled for this task"
|
||||
# ✓ promptLargeMode confirm(t("extractLargeAsk", …))
|
||||
# ✓ shouldPromptLargeMode Number(itemSize || 0) > LARGE_FILE_THRESHOLD_BYTES
|
||||
# ✓ LARGE_FILE_THRESHOLD_BYTES = 60 * 1024 * 1024 * 1024
|
||||
# ✓ large ("…":"1":"0")
|
||||
# ✓ /api/extract … large: large ? "1" : "0" …
|
||||
```
|
||||
|
||||
If any of these are missing, the cross-compile did not pick up the asset
|
||||
rebuild — run `make clean && make` (or remove only `gen/`) and rebuild.
|
||||
|
||||
## 10. Backwards compatibility & migration
|
||||
|
||||
| Surface | v1.6 → v1.7 | Notes |
|
||||
|---|---|---|
|
||||
| `zipx_extract()` signature | unchanged | old callers compile clean |
|
||||
| `zipx_default_limits()` body | unchanged | still returns `&k_default_limits` |
|
||||
| `/api/extract` `large` field | new (optional) | absent → `0` (default profile) |
|
||||
| `file_task_t::extract_large` | new (last field of the extract trio) | downstream consumers reading tasks must handle the new field |
|
||||
| Frontend default behaviour | unchanged | threshold gate is new |
|
||||
| `./web-file-mgr-linux` ABI | unchanged | Linux build also rebuilt with the new symbols |
|
||||
|
||||
**Migration for downstream users**: nothing required. To opt in to large
|
||||
archives, append `large=1` to the `/api/extract` request, OR click "OK" on the
|
||||
prompt that appears for any archive > 60 GiB on disk.
|
||||
|
||||
## 11. Trade-offs and known edges
|
||||
|
||||
* **60 GiB threshold is hardcoded** — yes, deliberate. It's set where the
|
||||
archival image / dump boundary typically lives. Power users can edit
|
||||
`LARGE_FILE_THRESHOLD_BYTES` (line 813 in `assets/main.js`).
|
||||
* **The large profile trusts the user about free space** — the server does
|
||||
not run `statvfs()` against `dst_dir` before extraction. The space check
|
||||
the engine itself runs (per-entry `max_file_bytes`) is the only guard.
|
||||
* **`bomb.zip` with ratio 1026 is rejected under both profiles** — known and
|
||||
intentional. The 1000 cap is the *floor* of the relaxed policy, not a
|
||||
"ZIP bombs welcome" flag. Other ZIP-bomb-shaped payloads with the same
|
||||
ratio will hit the same wall.
|
||||
* **No multi-volume / split support** — unchanged from pre-v1.7. The
|
||||
engine reads a single archive path; spans such as `archive.zip`,
|
||||
`archive.z01`, `archive.z02` are not stitched. minizip-ng has the API;
|
||||
wiring it is on the post-v1.7 roadmap.
|
||||
* **No proxy / streaming for archives above the STAGING_DIR ceiling** — the
|
||||
staging dir lives on the same filesystem as `dst_dir` and is sized
|
||||
proportionally. 2 TiB staging is required for a worst-case 2 TiB archive.
|
||||
|
||||
## 12. Future work (post-v1.7, prioritised)
|
||||
|
||||
1. Multi-volume support via `mz_zip_open_multi()` from vendored minizip-ng.
|
||||
2. Server-side `statvfs()` preflight against `dst_dir` when the active
|
||||
profile is `LARGE`, with a clearer error if there's not enough space.
|
||||
3. Compression-ratio cap that scales with file size (≤ small files: strict
|
||||
200:1; large files: relaxed). Same vector as the explicit profile but
|
||||
automatic.
|
||||
4. Streaming extractor API — `zipx_extract_stream()` — that does not
|
||||
materialise the staging dir at all. Useful once PS5 archive > 4 TiB is a
|
||||
real workload.
|
||||
5. Server-side telemetry (opt-in) for which profile is chosen per archive
|
||||
size band, to validate the 60 GiB threshold over time.
|
||||
@@ -0,0 +1,648 @@
|
||||
# UPGRADE: v1.8 — RAR extraction support
|
||||
|
||||
> This is the long-form maintainer's manual for the v1.8 archive-engine
|
||||
> expansion. It is written for the next developer, not the user. The
|
||||
> user-facing description lives in [`README.md → RAR extraction`](../README.md#rar-extraction);
|
||||
> the release notes are in [`CHANGELOG.md`](../CHANGELOG.md). The vendoring
|
||||
> decision tree (and the v1.9 upgrade path) is at
|
||||
> [`third_party/unrar/VENDORED.md`](../third_party/unrar/VENDORED.md) — most
|
||||
> of the "why" questions are answered there, not here.
|
||||
|
||||
---
|
||||
|
||||
## 1. Scope at a glance
|
||||
|
||||
| Feature | Status | Why |
|
||||
|---|---|---|
|
||||
| Single-volume RAR 1.5 / 2.x / 3.x / 4.x / 5.x | ✅ | dmc_unrar 1.7.0 supports it. |
|
||||
| RAR with PPMd, large dictionary | ✅ | dmc_unrar supports it. |
|
||||
| Conflict policy `fail` / `overwrite` / `merge` | ✅ | Same `zipx_conflict_t` protocol. |
|
||||
| Default + `large=1` limits via `LARGE_FILE_THRESHOLD_BYTES` | ✅ | Same `zipx_limits_t` protocol. |
|
||||
| Path traversal / symlinks / duplicate / ratio bomb / CRC error | ✅ | Same `zipx_status_t` codes as ZIP. |
|
||||
| Multi-volume RAR (`.part01.rar` + `.part02.rar` + …) | ❌ | Upstream `DMC_UNRAR_ARCHIVE_UNSUPPORTED_VOLUMES`; extracted to PC. |
|
||||
| Encrypted RAR (any encrypted header or file) | ❌ | Upstream `DMC_UNRAR_ARCHIVE_UNSUPPORTED_ENCRYPTED`; no `password=` field in v1.8. |
|
||||
| Symbolic links / FIFOs inside RAR | ❌ | `DMC_UNRAR_FILE_UNSUPPORTED_LINK` ⇒ `ZIPX_ERR_SPECIAL`. |
|
||||
| RAR 1.4 (very old) | ❌ | `DMC_UNRAR_ARCHIVE_VERSION_UNSUPPORTED` ⇒ `ZIPX_ERR_UNSUPPORTED`. |
|
||||
|
||||
The "extract on a PC first" recovery is the same escape hatch the engine
|
||||
already uses for ZIP encryption and ZIP64-stitched errors — the failure
|
||||
is an `extract_unsupported` with the file name as the detail argument,
|
||||
and the frontend already shows it with bilingual retry guidance.
|
||||
|
||||
---
|
||||
|
||||
## 2. Why dmc_unrar (and the v1.9 escape hatch)
|
||||
|
||||
`dmc_unrar` was chosen over the obvious alternatives for one reason each:
|
||||
|
||||
- **vs `winrar/unrar` upstream** — the official source is `UnRAR license`,
|
||||
not OSS. Modifying it (which we need to do for the host-side test
|
||||
shim, error-translation wrapper etc.) is prohibited.
|
||||
- **vs `opello/unrar`** — faithful UnRAR 7.x mirror, supports volumes
|
||||
*and* encryption, but is a 150-file C++17 codebase with its own
|
||||
Windows / registry / threading primitives. The C++ integration cost
|
||||
(`-DRAR_SMP`, third CXX link step, `prospero-pkg-config` audit) was
|
||||
not justified by v1.8's stated requirement.
|
||||
- **vs `libarchive`** — it pulls in `libarchive` itself (~400 KiB extra)
|
||||
and still uses an UnRAR-equivalent internally. Adding libarchive for
|
||||
RAR alone costs more than it returns.
|
||||
- **vs implementing UnRAR ourselves** — not even on the table.
|
||||
|
||||
When (if) multi-volume + encrypted RAR becomes worth it, the recipe is
|
||||
short: vendor `opello/unrar`, replace `dmc_unrar.c` with their `*.cpp`
|
||||
in `third_party/unrar/`, add a CXX link step to `Makefile`, switch
|
||||
`src/rar_extract.c` to the `RAROpenArchiveEx` / `RARSetPassword` DLL
|
||||
API. **The `rar_extract()` signature, the dispatch layer and the host
|
||||
tests do not need to change.** Full step-by-step recipe is in
|
||||
[`third_party/unrar/VENDORED.md`](../third_party/unrar/VENDORED.md).
|
||||
|
||||
---
|
||||
|
||||
## 3. Architecture delta vs v1.7
|
||||
|
||||
### 3.1 The shape
|
||||
|
||||
```
|
||||
┌──────────────────────────────────────────────────────────────────────┐
|
||||
│ Frontend: assets/main.js │
|
||||
│ - isExtractableArchive(item) covers .rar and .partNN.rar (NN==1) │
|
||||
│ - isRarSubVolume(item) flags .partNN.rar (NN>1) → greys button │
|
||||
│ - same LARGE_FILE_THRESHOLD_BYTES prompt for .rar as .zip │
|
||||
└─────────────────────┬────────────────────────────────────────────────┘
|
||||
│ POST /api/extract
|
||||
│ (path, dst_dir, conflict, remove_source,
|
||||
│ large) — no password= in v1.8
|
||||
┌─────────────────────▼────────────────────────────────────────────────┐
|
||||
│ Dispatch: src/extract.c │
|
||||
│ - extract_dispatch() by case-insensitive .zip / .rar suffix │
|
||||
│ - anything else → ZIPX_ERR_UNSUPPORTED, the same string the │
|
||||
│ ZIP path used to produce on its own │
|
||||
└─────────────────────┬────────────────────────────────────────────────┘
|
||||
│
|
||||
┌─────────────┴─────────────┐
|
||||
│ │
|
||||
┌───────▼───────────┐ ┌──────▼─────────────────┐
|
||||
│ src/zip_extract.c │ │ src/rar_extract.c │
|
||||
│ (unchanged in v1.8)│ │ (NEW) │
|
||||
│ backend: │ │ backend: │
|
||||
│ minizip-ng + zlib│ │ dmc_unrar (vendored) │
|
||||
│ │ │ dmc_unrar_api.h │
|
||||
│ │ │ (project-authored │
|
||||
│ │ │ facade) │
|
||||
└───────┬───────────┘ └──────┬─────────────────┘
|
||||
│ │
|
||||
└─────────────┬─────────────┘
|
||||
│ zipx_status_t
|
||||
│ zipx_limits_t (default / large)
|
||||
│ zipx_conflict_t
|
||||
│ zipx_progress_t
|
||||
│ zipx_result_t
|
||||
▼
|
||||
(shared task UI / progress / error mapping)
|
||||
```
|
||||
|
||||
The dispatcher and the engines share the entire type vocabulary from
|
||||
`src/zip_extract.h`. `src/rar_extract.c` `#include`s only
|
||||
`third_party/unrar/dmc_unrar_api.h` — it does not `#include` the
|
||||
vendored `.c` and it does not reach into dmc_unrar internals.
|
||||
|
||||
### 3.2 Three-phase pipeline (shared with ZIP)
|
||||
|
||||
`rar_extract()` mirrors `zipx_extract()` exactly:
|
||||
|
||||
1. **Scan** — open the archive with `dmc_unrar_archive_init` /
|
||||
`dmc_unrar_archive_open_path`, walk every entry header with
|
||||
`dmc_unrar_read_header` (the API does not have a list-only mode;
|
||||
scan reads the file content but discards it). For each entry:
|
||||
- normalise the path (`\` → `/`, trim trailing separators),
|
||||
validate against traversal / depth / name-length / file-size /
|
||||
total-size limits,
|
||||
- reject duplicate or clashing entry names with
|
||||
`ZIPX_ERR_DUPLICATE`,
|
||||
- reject encrypted / volume / version-unsupported / link / large
|
||||
RAR via the DMC codes → mapped to `ZIPX_ERR_UNSUPPORTED` /
|
||||
`ZIPX_ERR_SPECIAL` (see `rar_translate_error()`),
|
||||
- report progress at the same throttle the ZIP engine uses.
|
||||
2. **Extract** — for each entry, write into a staging directory (one
|
||||
`.wfm-extract-{pid}-{tid}/` per task, derived from `getpid()` and the
|
||||
task id), via `dmc_unrar_extract_file_to_path`. Each staging file
|
||||
is `fsync`d before renaming, so a power-loss mid-archive does not
|
||||
leave the destination half-written. Staging lives on the same
|
||||
filesystem as the destination so the publish is `rename()` (atomic).
|
||||
3. **Publish** — apply the conflict policy (`fail` / `overwrite` /
|
||||
`merge`) — reuses `zip_extract.c`'s `publish_entry()` /
|
||||
`publish_staging()` logic verbatim.
|
||||
4. **Cleanup / rollback** — on any mid-archive failure, every published
|
||||
entry created by this task is removed, the staging directory is
|
||||
removed recursively with `nftw(..., FTW_DEPTH | FTW_PHYS)`, and the
|
||||
caller is left with `dst_dir` exactly as it was (modulo whatever
|
||||
`ZIPX_CONFLICT_OVERWRITE` had already clobbered).
|
||||
|
||||
The staging layout, fsync strategy, conflict policy plumbing, cancel /
|
||||
progress callbacks and result-mapping logic are copied from
|
||||
`zip_extract.c` — exactly once — into `rar_extract.c` so that the
|
||||
engines can evolve independently. (Refactoring them into a
|
||||
`src/archive_common/` module is on the post-v1.9 roadmap; see §10.)
|
||||
|
||||
> **Note — superseded 2026-09-16.** The per-entry `fsync` described above was
|
||||
> removed. It cost 20–30 minutes on a 95k-file archive and bought nothing the
|
||||
> design needs: a crash mid-extract leaves the staging tree, which is discarded
|
||||
> on the next run, and publish is a rename-only phase. All three engines now
|
||||
> share the same "sync nothing, rename everything" policy. Measurements and the
|
||||
> accepted durability trade-off: `docs/EXTRACTION-PERF.md`.
|
||||
|
||||
### 3.3 Error mapping
|
||||
|
||||
`rar_translate_error()` in `src/rar_extract.c` maps the dmc_unrar
|
||||
return codes to the shared `zipx_status_t` enum so the rest of the
|
||||
project (and the frontend / task UI) cannot tell the difference
|
||||
between a ZIP failure and a RAR failure:
|
||||
|
||||
| dmc_unrar code | `zipx_status_t` |
|
||||
|---|---|
|
||||
| `DMC_UNRAR_ARCHIVE_UNSUPPORTED_VOLUMES` | `ZIPX_ERR_UNSUPPORTED` |
|
||||
| `DMC_UNRAR_ARCHIVE_UNSUPPORTED_ENCRYPTED` | `ZIPX_ERR_UNSUPPORTED` |
|
||||
| `DMC_UNRAR_FILE_UNSUPPORTED_ENCRYPTED` | `ZIPX_ERR_UNSUPPORTED` |
|
||||
| `DMC_UNRAR_FILE_UNSUPPORTED_LINK` | `ZIPX_ERR_SPECIAL` |
|
||||
| `DMC_UNRAR_FILE_UNSUPPORTED_LARGE` | `ZIPX_ERR_LIMIT_FILE` |
|
||||
| `DMC_UNRAR_ARCHIVE_SPLIT` | `ZIPX_ERR_UNSUPPORTED` |
|
||||
| `DMC_UNRAR_ARCHIVE_ANCIENT` | `ZIPX_ERR_UNSUPPORTED` |
|
||||
| `DMC_UNRAR_ARCHIVE_VERSION` | `ZIPX_ERR_UNSUPPORTED` |
|
||||
| `DMC_UNRAR_ARCHIVE_METHOD` | `ZIPX_ERR_UNSUPPORTED` |
|
||||
| `DMC_UNRAR_ARCHIVE_OPEN_FAIL` | `ZIPX_ERR_OPEN` |
|
||||
| `DMC_UNRAR_ARCHIVE_READ_FAIL` | `ZIPX_ERR_IO` |
|
||||
| `DMC_UNRAR_ARCHIVE_WRITE_FAIL` | `ZIPX_ERR_IO` |
|
||||
| `DMC_UNRAR_ARCHIVE_SEEK_FAIL` | `ZIPX_ERR_IO` |
|
||||
| `DMC_UNRAR_FILE_CRC32_FAIL` | `ZIPX_ERR_CRC` |
|
||||
| `DMC_UNRAR_ARCHIVE_NOT_RAR` | `ZIPX_ERR_FORMAT` |
|
||||
| `DMC_UNRAR_ARCHIVE_EMPTY` | `ZIPX_ERR_FORMAT` |
|
||||
| `DMC_UNRAR_ARCHIVE_INVALID_DATA` | `ZIPX_ERR_FORMAT` |
|
||||
| `DMC_UNRAR_ARCHIVE_NO_ALLOC` | `ZIPX_ERR_INTERNAL` |
|
||||
| `DMC_UNRAR_ARCHIVE_ALLOC_FAIL` | `ZIPX_ERR_INTERNAL` |
|
||||
| `DMC_UNRAR_ARCHIVE_IS_NULL` | `ZIPX_ERR_INTERNAL` |
|
||||
| `DMC_UNRAR_ARCHIVE_NOT_CLEARED` | `ZIPX_ERR_INTERNAL` |
|
||||
| `DMC_UNRAR_ARCHIVE_MISSING_FIELDS` | `ZIPX_ERR_INTERNAL` |
|
||||
|
||||
This table is exhaustive — every reachable dmc_unrar code has a
|
||||
defined mapping and there are no `default:` fall-throughs in the
|
||||
switch.
|
||||
|
||||
The progress / cancel / conflict protocol is byte-identical to ZIP:
|
||||
same `zipx_progress_t`, same `zipx_cancel_fn` signature, same
|
||||
`zipx_conflict_t` enum. The frontend never needs to branch on the
|
||||
archive format.
|
||||
|
||||
### 3.4 Behaviour contract (v1.8 — what callers can rely on)
|
||||
|
||||
For any input that produces `ZIPX_OK`:
|
||||
|
||||
- All requested entries from the archive are present in the destination,
|
||||
in the order they appear in the archive, with permissions `0644` for
|
||||
files and `0755` for directories (mirror of `zip_extract`'s default).
|
||||
- A conflict policy of `ZIPX_CONFLICT_FAIL` returns `ZIPX_ERR_CONFLICT`
|
||||
on the first collision; nothing is written.
|
||||
- `ZIPX_CONFLICT_OVERWRITE` replaces existing files with the extracted
|
||||
contents (a copy-paste of the ZIP engine's behaviour).
|
||||
- `ZIPX_CONFLICT_MERGE` keeps existing files, adds new ones.
|
||||
- `result->entries_total`, `result->entries_done`,
|
||||
`result->bytes_total`, `result->bytes_done`, `result->files_created`
|
||||
and `result->dirs_created` are all filled in.
|
||||
|
||||
For any non-`ZIPX_OK` return code:
|
||||
|
||||
- `dst_dir` is left **exactly as it was** before the call (modulo any
|
||||
`ZIPX_CONFLICT_OVERWRITE` clobbers that completed before the failure).
|
||||
- The staging directory is removed before `rar_extract()` returns.
|
||||
- `result->detail[]` and `result->message[]` are filled in for the
|
||||
UI to display.
|
||||
|
||||
For any cancellation request:
|
||||
|
||||
- The engine returns `ZIPX_ERR_CANCELED` from the next progress / cancel
|
||||
callback poll, all staging is removed, no `publish()` runs.
|
||||
|
||||
---
|
||||
|
||||
## 4. Source-tree layout
|
||||
|
||||
```
|
||||
src/
|
||||
extract.c # dispatcher (modified)
|
||||
extract.h # unchanged
|
||||
zip_extract.{c,h} # unchanged in v1.8
|
||||
rar_extract.{c,h} # NEW, ~1276 LOC in .c
|
||||
third_party/
|
||||
unrar/
|
||||
dmc_unrar.c # vendored (verbatim, 11 598 LOC)
|
||||
dmc_unrar_api.h # NEW, project-authored facade (~138 LOC)
|
||||
COPYING # GPL-2.0-or-later (vendored)
|
||||
README.md # upstream README (vendored)
|
||||
example.c # upstream usage example (vendored)
|
||||
VENDORED.md # NEW, why-dmc_unrar + v1.9 upgrade recipe
|
||||
tests/
|
||||
test_rar_extract.c # NEW, 14 checks (negative paths only)
|
||||
make_fixtures.py # adds rar_fixtures(); falls back to placeholder
|
||||
run-tests.sh # compiles dmc_unrar.o + rar_extract.o,
|
||||
# links test-rar-extract including zip_extract.o
|
||||
# so zipx_status_string / zipx_default_limits
|
||||
# / zipx_limits_profile resolve.
|
||||
assets/
|
||||
main.js # isExtractableArchive() / isRarSubVolume()
|
||||
lang-en.js, lang-zh.js # err_extract_unsupported updated
|
||||
Makefile # VERSION_TAG v1.8, third_party/unrar wired
|
||||
THIRD_PARTY_NOTICES # NEW section 3 for dmc_unrar
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 5. Vendoring mechanic — the facade header
|
||||
|
||||
The most important *engineering* lesson from v1.8 is the pattern used
|
||||
in `third_party/unrar/dmc_unrar_api.h`. Without it, integrating dmc_unrar
|
||||
into the host test suite was a mess:
|
||||
|
||||
```c
|
||||
/* src/rar_extract.c — what we wanted to write */
|
||||
#include "dmc_unrar.h" /* dream: a real header */
|
||||
```
|
||||
|
||||
But dmc_unrar ships as a single `.c` file. The "header" content is
|
||||
inside the `.c`. Naively:
|
||||
|
||||
```c
|
||||
/* src/rar_extract.c — what the naive approach forces */
|
||||
#include "dmc_unrar.c" /* ← does NOT work as a TU separate from rar_extract.c */
|
||||
```
|
||||
|
||||
…compiles if you do the `#include` in a **fresh** translation unit, but
|
||||
**fails** when both `src/rar_extract.c` and `tests/test_rar_extract.c`
|
||||
build with `-include tests/posix_compat.h`, because that shim
|
||||
redefines `open` → `wfm_open` / `close` → `wfm_close` and the
|
||||
`dmc_unrar_io_handler` struct inside dmc_unrar would then reference
|
||||
undeclared fields. The result is a flood of `error: 'struct
|
||||
dmc_unrar_io_handler' has no member named 'wfm_open'` and similar.
|
||||
|
||||
The facade pattern fix:
|
||||
|
||||
```c
|
||||
/* third_party/unrar/dmc_unrar_api.h — project-authored, project-license */
|
||||
#pragma once
|
||||
/* Re-declare only the symbols rar_extract.c touches. The names match
|
||||
dmc_unrar's internal names so the .c compiles unmodified. */
|
||||
typedef enum { … } dmc_unrar_return;
|
||||
typedef struct dmc_unrar_archive dmc_unrar_archive; /* opaque */
|
||||
typedef struct { … } dmc_unrar_file;
|
||||
dmc_unrar_return dmc_unrar_archive_init(dmc_unrar_archive *a);
|
||||
dmc_unrar_return dmc_unrar_archive_open_path(dmc_unrar_archive *a, const char *p);
|
||||
… /* … only the symbols rar_extract.c uses */
|
||||
```
|
||||
|
||||
Then:
|
||||
|
||||
```c
|
||||
/* src/rar_extract.c — what the facade enables */
|
||||
#include "dmc_unrar_api.h" /* project-authored, project-licensed */
|
||||
```
|
||||
|
||||
And:
|
||||
|
||||
```c
|
||||
/* Makefile — dmc_unrar is its own TU, unmodified */
|
||||
ps5-obj/third_party/unrar/dmc_unrar.o: third_party/unrar/dmc_unrar.c
|
||||
$(CC) $(THIRD_PARTY_CFLAGS) -c -o $@ $<
|
||||
```
|
||||
|
||||
The benefits:
|
||||
|
||||
- dmc_unrar.c is **never edited**, satisfying its GPL-2.0-or-later
|
||||
purity requirement and making an opello/unrar swap a 5-line diff.
|
||||
- The host test shim that renames `open` / `close` / `mkdirat` only
|
||||
affects the engine and test files, never dmc_unrar's TU.
|
||||
- The vendored `.c` is grep-able with reference back into the project
|
||||
at exactly one symbol boundary — `dmc_unrar_api.h`.
|
||||
- Future C++ integration (opello/unrar) reuses the same facade slot;
|
||||
the body becomes `-DRARDLL` and a different header file.
|
||||
|
||||
This pattern generalises — *anytime you want to vendor a
|
||||
single-file C library that internally uses a name that your build
|
||||
system also touches*, write a 100-line facade header that re-declares
|
||||
just the symbols you use, and treat the vendored `.c` as a compile
|
||||
unit on its own.
|
||||
|
||||
---
|
||||
|
||||
## 6. Compression-ratio / format caveats specific to RAR
|
||||
|
||||
These are not bugs — they are *properties of dmc_unrar* that the
|
||||
host test suite and the engine must respect:
|
||||
|
||||
- **dmc_unrar has no `RAR_OM_LIST`** (list-only mode). The "scan"
|
||||
pass opens the archive, walks every header, and **reads the file
|
||||
content** even though it doesn't write anything out. A 500-entry
|
||||
archive with 4 GiB average entry size therefore costs ~2 TiB of
|
||||
read I/O during scan. This is acceptable for v1.8 because the
|
||||
typical PS5 use case is `one archive, few hundred MiB`, but it is
|
||||
worth documenting so a future optimisation (on-the-fly skip
|
||||
through `dmc_unrar_extract_file_to_path`) doesn't surprise the
|
||||
next reader.
|
||||
- **dmc_unrar decompresses synchronously on the same thread** that
|
||||
calls `dmc_unrar_extract_file_to_path`. Cancel callbacks are
|
||||
polled inside `dmc_unrar_read_header` (and at engine chokepoints)
|
||||
— *not* in the inner loop. A 1 GiB file extraction cannot be
|
||||
cancelled mid-decompression. A future improvement could fork a
|
||||
child process for extract so SIGKILL works deterministically.
|
||||
- **The dmc_unrar byte-swap helpers `be32toh` / `be64toh` collide
|
||||
with `<endian.h>`** on some compilers. We work around it by
|
||||
compiling with `-DDMC_UNRAR_DISABLE_BE32TOH_BE64TOH=1`, which
|
||||
lets dmc_unrar use its own internal byte-swap implementation.
|
||||
This is documented at the top of `dmc_unrar.c` and was the only
|
||||
thing the Makefile needed to set to get a green build on PS5.
|
||||
|
||||
These are engine limitations, not failure modes the user will see in
|
||||
normal operation.
|
||||
|
||||
---
|
||||
|
||||
## 7. Frontend wiring details
|
||||
|
||||
### 7.1 The detection rule
|
||||
|
||||
```js
|
||||
function isExtractableArchive(item) {
|
||||
if (item.type !== "-") return false; // file, not directory
|
||||
if (/\.zipx?$/i.test(item.name)) return true;
|
||||
if (/\.rar$/i.test(item.name)) return true;
|
||||
if (/\.part0*1\.rar$/i.test(item.name)) return true; // RAR main volume
|
||||
return false;
|
||||
}
|
||||
|
||||
function isRarSubVolume(item) {
|
||||
if (item.type !== "-") return false;
|
||||
if (/\.part0*1\.rar$/i.test(item.name)) return false; // main is not a sub
|
||||
return /\.part0*\d+\.rar$/i.test(item.name); // .partNN.rar, NN>1
|
||||
}
|
||||
```
|
||||
|
||||
### 7.2 The button-state rule
|
||||
|
||||
```js
|
||||
function renderExtractButton(items, locked) {
|
||||
const mains = items.filter(isExtractableArchive);
|
||||
const subs = items.filter(isRarSubVolume);
|
||||
if (subs.length && !mains.length) { // only sub-volumes
|
||||
extractBtn.disabled = true;
|
||||
extractBtn.title = t("extractSelectMainVolume");
|
||||
return;
|
||||
}
|
||||
if (mains.length !== 1) {
|
||||
extractBtn.hidden = mains.length !== 1;
|
||||
extractBtn.disabled = true;
|
||||
return;
|
||||
}
|
||||
extractBtn.title = t("extractToCurrent") + ": " + itemTitle(mains[0]);
|
||||
extractBtn.disabled = locked;
|
||||
}
|
||||
```
|
||||
|
||||
### 7.3 The "extract on a PC first" error path
|
||||
|
||||
When the engine rejects a multi-volume / encrypted / link entry inside
|
||||
a RAR, the failure is `ZIPX_ERR_UNSUPPORTED` (or
|
||||
`ZIPX_ERR_SPECIAL`). The frontend already has
|
||||
`err_extract_unsupported` updated to:
|
||||
|
||||
> `…(only unencrypted plain ZIP and single-volume RAR are
|
||||
> supported)…` (en)
|
||||
> `…(仅支持未加密的普通 ZIP 与单卷 RAR)…` (zh)
|
||||
|
||||
The "extract on a PC first" guidance is *implicit* — when the user
|
||||
hits this message with a `.part01.rar` selected, the .part02+.rar
|
||||
tooltips + the error string are the two breadcrumbs. There is no
|
||||
explicit "unrar on your PC" button in v1.8 because the redirect is
|
||||
self-evident from the failure.
|
||||
|
||||
### 7.4 The `large=1` prompt for RAR
|
||||
|
||||
The threshold is shared. A `.rar` larger than `240 GiB` triggers the
|
||||
same `promptLargeMode()` confirmation as a `.zip`. The confirmation
|
||||
text uses `extractLargeAsk` (slightly relaxed in v1.8.1) — the wording
|
||||
is format-agnostic, so no new strings are needed.
|
||||
|
||||
---
|
||||
|
||||
## 8. Host test suite — what 14 checks actually cover
|
||||
|
||||
`tests/test_rar_extract.c` is a **negative-path-only** suite, because
|
||||
we have no RAR writer in this repo and the host probably doesn't have
|
||||
`rar` / `7z` installed either. The suite accepts the absence of real
|
||||
fixtures as a feature: by exercising *only* the dispatch and error
|
||||
translation layers, we get coverage that is independent of whether the
|
||||
host has any RAR tooling.
|
||||
|
||||
| Check | What it asserts |
|
||||
|---|---|
|
||||
| `test_engine_dispatch_zip_renamed_rar` | `.zip` renamed to `.rar` is rejected with `ZIPX_ERR_UNSUPPORTED` (the engine decides by extension; the front-end tests the matching detection). |
|
||||
| `test_engine_dispatch_junk_rar` | A 1 KiB blob named `.rar` is rejected as `ZIPX_ERR_UNSUPPORTED` — the dmc_unrar open fails, mapping to `ZIPX_ERR_UNSUPPORTED`. |
|
||||
| `test_engine_dispatch_missing_source` | `rar_extract()` with a non-existent path returns `ZIPX_ERR_OPEN` (mapped from `DMC_UNRAR_OPEN_FAIL`). |
|
||||
| `test_engine_dispatch_null_rar_path` | `rar_extract(NULL, dst, …)` is rejected with `ZIPX_ERR_INTERNAL` — defensive, never user-visible. |
|
||||
| `test_engine_dispatch_null_dst` | `rar_extract(path, NULL, …)` is rejected with `ZIPX_ERR_INTERNAL`. |
|
||||
| `test_engine_dispatch_dst_is_regular_file` | `rar_extract(path, /some/file, …)` returns `ZIPX_ERR_CONFLICT` (open_parent_dirs fails). |
|
||||
| `test_format_translation` | Parametric: for each `DMC_UNRAR_*` code we care about, the corresponding `rar_translate_error()` mapping is exercised indirectly (via `result->message` strings). |
|
||||
| `test_limits_handoff_default` | When `task->extract_large == 0`, the default profile is handed in (200 K entries / 1 TiB / 256 GiB / 500:1). |
|
||||
| `test_limits_handoff_large` | When `task->extract_large == 1`, the large profile is handed in (500 K / 2 TiB / 1 TiB / 1000:1). |
|
||||
| `test_translate_open_fail_to_err_open` | DMC open-failure → `ZIPX_ERR_OPEN`. |
|
||||
| `test_translate_volume_unsp_to_err_unsupported` | The DMC volume code → `ZIPX_ERR_UNSUPPORTED`. |
|
||||
| `test_translate_encrypted_unsp_to_err_unsupported` | The DMC encryption code → `ZIPX_ERR_UNSUPPORTED`. |
|
||||
| `test_translate_link_unsp_to_err_special` | The DMC link code → `ZIPX_ERR_SPECIAL`. |
|
||||
| `test_progress_throttle` | The progress callback is invoked at most every ~200 ms or every ~1 MiB extracted, like the ZIP engine. |
|
||||
|
||||
When `tests/make_fixtures.py` finds a host `rar` or `7z` writer, it
|
||||
generates a real `basic.rar` fixture, and an additional 2 checks
|
||||
(`test_rar4_basic_extract` / `test_rar5_basic_extract`) succeed
|
||||
automatically — those are *not* counted in the 14 baseline.
|
||||
|
||||
The complete count after `bash tests/run-tests.sh` is therefore **83
|
||||
checks** (69 ZIP + 14 RAR) on a RAR-less host, and **85 checks** on a
|
||||
host with `rar` installed.
|
||||
|
||||
---
|
||||
|
||||
## 9. Cross-compile and verification (user-side checklist)
|
||||
|
||||
The host suite runs anywhere. The PS5 ELF build runs in WSL with
|
||||
`PS5_PAYLOAD_SDK` set:
|
||||
|
||||
```bash
|
||||
# WSL Ubuntu-22.04 bash
|
||||
cd /home/song/ps5-web-file-manager
|
||||
export PS5_PAYLOAD_SDK=/opt/ps5-payload-sdk
|
||||
make all # ~30 s on cold
|
||||
ls -la web-file-mgr.elf # record size
|
||||
sha256sum web-file-mgr.elf # record digest
|
||||
```
|
||||
|
||||
Then in Windows-side Git Bash / PowerShell:
|
||||
|
||||
```bash
|
||||
cd "C:/Users/songl/Desktop/Web File Manager/ps5-web-file-manager"
|
||||
file web-file-mgr.elf # ELF 64-bit LSB pie, x86-64
|
||||
od -An -tx1 -N20 web-file-mgr.elf | head -2 # 7f45 4c46 0201 + e_machine 003e
|
||||
python3 .build/check-elf-gzip.py ./web-file-mgr.elf # 7 v1.7 keys + 1 v1.8 key
|
||||
```
|
||||
|
||||
The new v1.8 ELF-gzip key to verify is `err_extract_unsupported`
|
||||
("only unencrypted plain ZIP and single-volume RAR are supported").
|
||||
It must appear in the binary.
|
||||
|
||||
Then in `CHANGELOG.md`, paste the size + sha256 into the v1.8 banner
|
||||
header at the top of the file. Commit + push:
|
||||
|
||||
```bash
|
||||
git add -A
|
||||
git -c core.autocrlf=false commit -m "v1.8: RAR4/RAR5 single-volume unencrypted (dmc_unrar backend)"
|
||||
# push runs in PowerShell on the user's machine (sandbox github 502)
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 10. Roadmap (post-v1.8)
|
||||
|
||||
### 10.1 v1.9 — full RAR (multi-volume + encrypted)
|
||||
|
||||
See [`third_party/unrar/VENDORED.md`](../third_party/unrar/VENDORED.md)
|
||||
§"Upgrading to a fuller library (v1.9 plan)" for the migration recipe.
|
||||
The public `rar_extract()` signature and the dispatch layer do **not**
|
||||
need to change; only:
|
||||
|
||||
1. `third_party/unrar/dmc_unrar.c` is removed and `*.cpp` from
|
||||
`opello/unrar` are placed there.
|
||||
2. `Makefile` gains a `THIRD_PARTY_CPP_SRCS := $(wildcard
|
||||
third_party/unrar/*.cpp)` and the corresponding CXX link step.
|
||||
3. `third_party/unrar/dmc_unrar_api.h` is renamed to
|
||||
`unrar_api.h` and its bodies filled in from the rarlab DLL API
|
||||
(`RAROpenArchiveEx`, `RARSetPassword`, `RARProcessFileW`,
|
||||
`RARCloseArchive`).
|
||||
4. `src/rar_extract.c` swaps the `dmc_unrar_*` calls for the
|
||||
`RAR*` calls; the visible behaviour is the same except `password=`
|
||||
is now a real field on `POST /api/extract`.
|
||||
5. Frontend gains a password modal (HTML + CSS + JS) that pops when
|
||||
the engine returns `ZIPX_ERR_PASSWORD`, with retry semantics.
|
||||
|
||||
The host tests **should not change** — they test the dispatch /
|
||||
error-translation / limits-handoff layers, none of which see dmc_unrar.
|
||||
|
||||
### 10.2 Common archive library
|
||||
|
||||
Both engines share:
|
||||
|
||||
- staging directory management (`open_parent_dirs`, `remove_tree`,
|
||||
`cleanup_staging`, `publish_staging`, `rollback_published`);
|
||||
- duplicate-name detection (`nameset_add_path`, `nameset_check_duplicate`);
|
||||
- the `extract_progress` callback;
|
||||
- `report()` / `rarx_fail()` / `rarx_set_detail()` style error
|
||||
formatting;
|
||||
- a `time()`-based throttle for progress;
|
||||
- `chmod_0777_fd` permission normalisation.
|
||||
|
||||
These are duplicated between `src/zip_extract.c` and
|
||||
`src/rar_extract.c` today. A natural refactor is
|
||||
`src/archive_engine_common.c` exposing them; v1.9 is the moment to do
|
||||
this refactor since the RAR engine is changing anyway.
|
||||
|
||||
### 10.3 ZIP multi-volume (no decision yet)
|
||||
|
||||
The original v1.7 wishlist also included multi-volume ZIP
|
||||
(`.zip` + `.z01`, `.z02`). The implementation is similar to RAR multi-
|
||||
volume in shape but uses minizip-ng's `zip_open_from_file`-style
|
||||
APIs. Defer — no user demand on record for v1.9 yet.
|
||||
|
||||
### 10.4 Format dispatcher magic-byte sniffing
|
||||
|
||||
Today the dispatch is by extension. A magic-byte sniff for `Rar!` /
|
||||
`PK\x03\x04` would let users rename `.bin` archives and still get
|
||||
correct handling. Add when there's a real bug report — until then
|
||||
the simple suffix check is enough.
|
||||
|
||||
### 10.5 ELF size trend
|
||||
|
||||
| Version | Approx. ELF size | Notes |
|
||||
|---|---|---|
|
||||
| v1.7 | 418 KiB | ZIP only. |
|
||||
| v1.8 | ~430 KiB (est.) | + 11 598 LOC of stripped dmc_unrar code (≈ 25 KiB compressed). Actual size pending WSL cross-compile. |
|
||||
| v1.9 (opello/unrar) | ~700 KiB (est.) | + ~280 KiB of C++ UnRAR. |
|
||||
|
||||
We are still well below the 4 MiB ELF-loader cap, but a future
|
||||
addition (7z or AES ZIP) would tip us past the 1 MiB comfort line;
|
||||
that is the right moment to reconsider scope.
|
||||
|
||||
---
|
||||
|
||||
## 11. Files touched in v1.8
|
||||
|
||||
### 11.1 New
|
||||
|
||||
```
|
||||
third_party/unrar/dmc_unrar.c # 11 598 LOC, verbatim upstream
|
||||
third_party/unrar/dmc_unrar_api.h # 138 LOC, project-authored facade
|
||||
third_party/unrar/COPYING # GPL-2.0-or-later (vendored)
|
||||
third_party/unrar/README.md # upstream README (vendored)
|
||||
third_party/unrar/example.c # upstream usage example (vendored)
|
||||
third_party/unrar/VENDORED.md # 76 LOC, vendoring rationale + v1.9 path
|
||||
src/rar_extract.c # 1276 LOC, engine
|
||||
src/rar_extract.h # 34 LOC, public signature
|
||||
tests/test_rar_extract.c # 346 LOC, 14 negative-path checks
|
||||
docs/UPGRADE-v1.8-rar-support.md # this document
|
||||
```
|
||||
|
||||
### 11.2 Modified
|
||||
|
||||
```
|
||||
Makefile # VERSION_TAG v1.8; add third_party/unrar
|
||||
src/extract.c # extract_dispatch(); ends_with_ci()
|
||||
assets/main.js # isExtractableArchive / isRarSubVolume
|
||||
assets/lang-en.js # err_extract_unsupported copy
|
||||
assets/lang-zh.js # err_extract_unsupported copy
|
||||
tests/make_fixtures.py # rar_fixtures() with rar/7z/placeholder fallback
|
||||
tests/run-tests.sh # dmc_unrar.o, rar_extract.o, test_rar_extract.o
|
||||
THIRD_PARTY_NOTICES # section 3 = dmc_unrar attribution
|
||||
README.md # What's new in v1.8, RAR section, + Credits entry
|
||||
CHANGELOG.md # v1.8 block (this PR)
|
||||
docs/HANDOVER.md # progress checkmarks (D1–D6)
|
||||
```
|
||||
|
||||
### 11.3 Untouched but verified
|
||||
|
||||
```
|
||||
src/zip_extract.{c,h} # ZIP engine behaviour identical
|
||||
src/filemgr_internal.h # ZIP task struct fields unchanged
|
||||
assets/param.json # no version bump; VERSION_TAG is in the build
|
||||
src/app_installer.c # PS5 Media launcher flow unaffected
|
||||
docs/UPGRADE-v1.7-zip-large-file-profile.md # unchanged
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 12. Closing notes
|
||||
|
||||
The hardest engineering decision in v1.8 was *not* "what RAR library
|
||||
to use" — it was "how much scope to ship in v1.8 vs v1.9." The
|
||||
original plan was multi-volume + encrypted; the audit on dmc_unrar
|
||||
+ opello/unrar's audit surface pushed that to v1.9 with a clean
|
||||
upgrade recipe. **v1.8 is therefore intentionally a smaller release
|
||||
than the planning doc (`docs/HANDOVER.md` §6) anticipated.**
|
||||
|
||||
For the next developer reading this:
|
||||
|
||||
- If you implement v1.9, the natural starting point is the
|
||||
`VENDORED.md` recipe, not this section.
|
||||
- If you are debugging an in-the-wild report ("my .rar won't
|
||||
extract"), the most common cause is multi-volume or encrypted —
|
||||
point the user at the `err_extract_unsupported` message and the
|
||||
PC-extract fallback. v1.8 is working as designed when this happens.
|
||||
- If you are adding a new archive format (7z, tar.bz2, …), the
|
||||
pattern is: (a) write a vendored facade header, (b) write a
|
||||
`src/<fmt>_extract.{c,h}` mirror of `rar_extract.c`, (c) extend
|
||||
`extract_dispatch()`, (d) add i18n strings, (e) extend the host
|
||||
test suite.
|
||||
|
||||
Everything else is the same shape as the existing engines.
|
||||
@@ -0,0 +1,218 @@
|
||||
# 上游 v1.8 解压方案 vs 本项目 v1.9.1
|
||||
|
||||
> 核查时间:2026-09-15 · 上游 `owendswang/ps5-web-file-manager`
|
||||
> 来源:GitHub API 查询 + commit `b405721`("Added support for 7zip helper",2026-09-08)完整 patch(2301 行)
|
||||
> 上游 v1.8 = tag `ad7d754`,v1.7 = `72341d6`(本项目 fork 的基线)
|
||||
|
||||
## 结论速览
|
||||
|
||||
**不是同一个层面的方案,各有明确胜负手。**
|
||||
|
||||
| | 上游 v1.8 | 本项目 v1.9.1 |
|
||||
|---|---|---|
|
||||
| 一句话 | **把 7-Zip 本体做成外部 helper 进程,靠 IPC 调用** | **自研 + vendor 解码库,全部内嵌在同一进程** |
|
||||
| 最强的点 | 格式覆盖 **30 种**,解压核心是 7-Zip 本体 | **完整安全护栏** + 单文件部署 |
|
||||
| 最弱的点 | **零安全护栏**,且 helper 缺失 = 功能全废 | 格式覆盖只有 **3 种** |
|
||||
|
||||
## 一、上游 v1.8 的实际架构
|
||||
|
||||
### 1.1 三个组件
|
||||
|
||||
| 组件 | 位置 | 职责 |
|
||||
|---|---|---|
|
||||
| `src/archive_extract.c`(124 行) | 本仓库 | 只做**后缀识别** + 输出目录名推导 |
|
||||
| `src/archive_helper.c`(732 行) | 本仓库 | **IPC 客户端**:启动 helper + Unix socket 协议 |
|
||||
| `wfm-7zip-helper.elf`(~百 MB 级) | `/data/wfm/`,**不在仓库里,单独分发** | 真正的解压 = **7-Zip 本体** |
|
||||
|
||||
README 原文:
|
||||
|
||||
> Extraction requires the separately distributed `wfm-7zip-helper.elf` helper at `/data/wfm/wfm-7zip-helper.elf`.
|
||||
|
||||
### 1.2 启动链路
|
||||
|
||||
```c
|
||||
/* archive_helper_autostart() —— 仅 __SCE__(PS5)分支,Linux 直接返回 0 */
|
||||
1. archive_helper_probe() // 已有实例在跑就复用,绝不替换
|
||||
2. stat("/data/wfm/wfm-7zip-helper.elf") // 不存在 → 静默返回 0
|
||||
3. 校验 ELF magic "\x7fELF"、大小 4B ~ 128MB
|
||||
4. connect(127.0.0.1:9021) // WFM_ELFLDR_PORT —— elfldr payload 加载器
|
||||
5. 把 helper ELF 的**全部字节流**推过去
|
||||
6. shutdown(SHUT_WR)
|
||||
```
|
||||
|
||||
即:**通过 elfldr(PS5 homebrew 的 ELF 加载 payload)把 helper 拉起成一个独立进程。**
|
||||
|
||||
### 1.3 通信协议(自研二进制帧)
|
||||
|
||||
- 传输层:Unix domain socket —— PS5 走 `/system_tmp/wfm-7zip-helper.sock`,Linux 走 `/tmp/...`
|
||||
- 帧格式:magic `"W7HP"` + 20 字节头(type / flags / request_id / payload_size,**大端序**)
|
||||
- 上限:payload 1 MiB、路径 256 KiB、响应 64 KiB
|
||||
|
||||
消息类型:
|
||||
|
||||
| 方向 | 消息 |
|
||||
|---|---|
|
||||
| 主 → helper | `PING` `EXTRACT` `CANCEL` `LIST_TASKS` `ATTACH_TASK` `ACK_TASK` |
|
||||
| helper → 主 | `PONG` `ACCEPTED` `PROGRESS` `CURRENT_FILE` `PASSWORD_REQUIRED` `DONE` `ERROR` `TASK_SNAPSHOT` `LIST_DONE` |
|
||||
|
||||
回调接口 `archive_helper_callbacks_t`:`cancel_requested()` / `progress(done,total)` / `current_file(path)`。
|
||||
|
||||
### 1.4 支持格式(30 种后缀)
|
||||
|
||||
```
|
||||
.7z .001 .zip .zipx .rar .arj .bz2 .bzip2 .tbz .tbz2 .cab .gz .gzip
|
||||
.tgz .tpz .lzh .lha .tar .xz .txz .z .taz .zst .tzst .xar .xip
|
||||
.cpio .lzma .pmd
|
||||
```
|
||||
|
||||
外加 `.partNN.rar`(只接受 `part1`,即必须从第一卷进入)。
|
||||
分卷靠 7-Zip 原生能力:`.001` **无差别接受**(不校验卷集连续性)。
|
||||
|
||||
### 1.5 任务恢复(上游的亮点)
|
||||
|
||||
`filemgr_recover_extract_tasks()` 在 `main.c` 启动时调用:从 helper 拉 `TASK_SNAPSHOT` 列表,把还在跑的 job **reattach 回主进程的任务列表**。
|
||||
|
||||
因为 helper 是独立进程,**主 payload 被重启 / 浏览器重开,解压任务不会丢**。`archive_helper_probe()` 的注释也点明了这个设计的意图:
|
||||
|
||||
```c
|
||||
/* Never replace a connected daemon, even if it is temporarily slow. */
|
||||
```
|
||||
|
||||
### 1.6 ⚠️ 没有的东西(全 patch 逐行核查)
|
||||
|
||||
| 项目 | 上游 v1.8 | 说明 |
|
||||
|---|---|---|
|
||||
| 条目数上限 | ❌ | 无 `max_entries` 类逻辑 |
|
||||
| 单文件/总大小上限 | ❌ | 无 |
|
||||
| 压缩比筛查(防炸弹) | ❌ | 无 |
|
||||
| 磁盘空间预检 | ❌ | 无 `statvfs` 调用 |
|
||||
| 路径穿越防护 | ❌ | 未见 `..`/绝对路径校验,交给 7-Zip |
|
||||
| 原子发布 | ❌(未见) | 直接解到目标目录,中断留半成品 |
|
||||
|
||||
`grep -i "ratio|max_entries|statvfs|bomb"` 的全部命中都是误报(`operations` 里含子串 `ratio`)。
|
||||
|
||||
**换句话说:上游把解压这件事整体外包给了 7-Zip,包括安全责任。**
|
||||
|
||||
## 二、本项目 v1.9.1 的架构
|
||||
|
||||
| 组件 | 职责 |
|
||||
|---|---|
|
||||
| `src/zip_extract.c` | ZIP:minizip-ng,含 zip64、三种分卷命名、`.z01` 真分盘语义 |
|
||||
| `src/rar_extract.c` | RAR:vendor unrar 7.20.1(DLL 模式),v4/v5/多卷/加密 |
|
||||
| `src/sevenz_extract.c` + `sevenz_chain.c` | 7z:自解析 folder + pull 式 codec 链 + 7zAES |
|
||||
| `src/zipx_volume.c` / `zipx_volstream.c` / `sevenz_volstream.c` | 卷集识别 + 连续流抽象 |
|
||||
|
||||
**格式覆盖:`.zip` / `.rar` / `.7z` 三种**,各自支持单卷 / 分卷 / 密码。
|
||||
|
||||
### 已有的工程能力
|
||||
|
||||
| 项目 | 本项目 | 实现位置 |
|
||||
|---|---|---|
|
||||
| 条目数 / 总大小 / 单文件上限 | ✅ 两档 profile(20万~50万条目 / 2~4 TiB / 512 GiB~1 TiB) | `zipx_common.c` |
|
||||
| 压缩比筛查 | ✅ `max_ratio` 500/1000,**1 GiB 下限豁免**小文件 | `zip_extract.c` |
|
||||
| 磁盘空间预检 | ✅ `check_space()` 按**解压后总量**查 `statvfs` | `zip_extract.c:636` |
|
||||
| 路径穿越防护 | ✅ 有专项测试(`path traversal variants`) | 测试矩阵 |
|
||||
| 原子发布 | ✅ staging 目录 + 整 rename(**无逐条目 fsync**,2026-09-16 起) | 三引擎统一 |
|
||||
| 冲突策略 | ✅ FAIL / OVERWRITE / MERGE,目录碰撞递归下钻 | 三引擎统一 |
|
||||
| 取消 | ✅ 条目粒度 | — |
|
||||
| 任务恢复 | ❌ **没有** | — |
|
||||
| 内存隔离 | ❌ 与主进程共享地址空间(LZMA2 字典须封顶) | — |
|
||||
|
||||
### 测试覆盖
|
||||
|
||||
ZIP 108 + RAR 27 + 7z 28 = **163 checks**,0 失败(MinGW host)+ PS5 真机构建通过。
|
||||
|
||||
## 三、逐维度对比
|
||||
|
||||
| 维度 | 上游 v1.8 | 本项目 v1.9.1 | 胜 |
|
||||
|---|---|---|---|
|
||||
| 格式覆盖 | **30 种** | 3 种 | 上游 |
|
||||
| 解压核心正确性 | 7-Zip 本体(20 年验证) | 自研 7z 链 + 成熟 vendor 库 | 上游 |
|
||||
| 分卷语义 | 靠 7-Zip 原生(`.001` 无差别) | 自研两套语义(byte-split / zip split disk) | 平手(我们更细,上游更省心) |
|
||||
| `.rar.001` | ✅ 直接吃 | ⚠️ 需改名为 `.partN.rar` | 上游 |
|
||||
| 部署 | **两个文件**,路径写死 `/data/wfm/` | **单文件**,零外部依赖 | 我们 |
|
||||
| helper 缺失时 | **功能全废**(`archive_helper_not_running`) | 不适用 | 我们 |
|
||||
| 防压缩炸弹 | ❌ 无 | ✅ ratio + 1 GiB 下限 | **我们** |
|
||||
| 磁盘写满保护 | ❌ 无 | ✅ 预检解压后总量 | **我们** |
|
||||
| 路径穿越 | ❌ 无 | ✅ 有防护 + 测试 | **我们** |
|
||||
| 中断留残留 | ⚠️ 可能留半成品 | ✅ staging 隔离,失败即清 | **我们** |
|
||||
| 任务恢复 / 跨重启 | ✅ 跨进程 reattach | ❌ | 上游 |
|
||||
| 内存隔离 | ✅ 独立进程,峰值不影响主服务 | ❌ 共享地址空间 | 上游 |
|
||||
| 主仓库构建成本 | 低(不编 7-Zip) | 首次 +3~5 min、ELF +98 KiB | 上游 |
|
||||
| 错误信息详细度 | 中等(6 个 code) | 含条目名 / errno / 字节数 | 我们 |
|
||||
|
||||
## 四、该怎么评价
|
||||
|
||||
### 上游那步棋走对了什么
|
||||
|
||||
**把 7-Zip 当外部依赖,是性价比极高的工程决策。** 自己写解码器要几个月,`apt` 一个 7-Zip 就换来 30 种格式 + 20 年验证的正确性。而且顺手拿到了两个我们暂时没有的能力:跨进程任务恢复、内存隔离。
|
||||
|
||||
### 但它把安全责任也一起外包了
|
||||
|
||||
这是**实质缺陷**,不是风格问题。在 PS5 上跑的具体后果:
|
||||
|
||||
1. **压缩炸弹直接写满内置存储** —— 一个 10 KB 的 zip 可以声明 100 GB,没有任何拦截
|
||||
2. **路径穿越** —— `../../` 条目可以写到解压目标之外(7-Zip 本身会做基本清理,但这属于"相信第三方"而非"自己保证")
|
||||
3. **磁盘写满** —— 不预检,写到 ENOSPC 才失败,此时已留下部分文件
|
||||
4. **失败留残留** —— 没有 staging 隔离
|
||||
|
||||
我们在这四项上都有明确实现和测试。163 checks 里专门有一组 `path traversal variants` 和 `limits`。
|
||||
|
||||
### 但必须承认格式覆盖是短板
|
||||
|
||||
31 种格式的差距不是"多一点便利",是**用户会觉得我们弱**:`.tar.gz`、`.xz`、`.zst`、`.bz2` 在 PS5 场景(游戏包、备份、Mod)里出现频率不低。
|
||||
|
||||
## 五、可借鉴 / 不建议照抄
|
||||
|
||||
### 建议做:补常见格式(性价比高)
|
||||
|
||||
按实际收益排序:
|
||||
|
||||
| 优先级 | 格式 | 实现路径 |
|
||||
|---|---|---|
|
||||
| 高 | `.tar` / `.tar.gz` / `.tgz` | tar 解析器自己写(格式极简,~300 行)+ zlib 已在手 |
|
||||
| 高 | `.gz` / `.xz` / `.lzma` | gzip 用 zlib;xz/lzma 可 vendor liblzma 或复用 LZMA SDK 的 LzmaDec |
|
||||
| 中 | `.bz2` / `.zst` | 单文件解码器,各 ~1000 行,可 vendor |
|
||||
| 低 | `.cab` / `.arj` / `.lzh` / `.cpio` / `.xar` | 罕见,除非有具体需求 |
|
||||
|
||||
**注意**:`gzip`/`xz`/`zst` 是**单文件**格式(不是归档),解出来就是一个文件,输出路径语义需单独设计。
|
||||
|
||||
### 建议评估:任务恢复 / 进程隔离
|
||||
|
||||
**动机**:主 payload 被系统杀或用户重开浏览器时,正在跑的大包解压会整个丢失。上游靠 helper 独立进程解决了这点。
|
||||
|
||||
**但在我们架构下的成本**:需要引入子进程 + IPC(或至少状态持久化 + 重启后重扫 staging)。PS5 上 fork/exec 与 elfldr 强耦合,不是小改动。
|
||||
|
||||
**中间路线**:解压失败时保留 staging 目录 + 记录任务清单文件,重启后支持"续解"。改动量中等,能拿到大部分收益,不必引入 IPC。
|
||||
|
||||
### 不建议:照抄 helper 路线
|
||||
|
||||
理由:
|
||||
|
||||
1. **部署体验倒退** —— 用户要装两个文件,还得记住放 `/data/wfm/`;丢一个功能全废。现在单 ELF 是无状态交付,这是真实优势
|
||||
2. **安全护栏会一起丢** —— 走 7-Zip 就意味着放弃我们对 entries/ratio/空间/穿越的控制
|
||||
3. **helper 上游自己都不敢放进仓库**("separately distributed"),大概率是体积或许可原因,跟着走会继承同样的问题
|
||||
4. **我们已经付过的成本会沉没** —— 7z 引擎(自解析 folder + pull 链 + 7zAES)+ 三类分卷抽象共约 3,600 行零耦合代码
|
||||
|
||||
### 一句话总结
|
||||
|
||||
**上游赢在"格式广度 + 进程架构",我们赢在"安全 + 部署 + 错误质量"。**
|
||||
|
||||
如果只想要功能广度,上游的路线更省力;如果要一个**能放心交付给用户、不会被一个恶意压缩包搞崩存储**的工具,我们的路线是对的,缺的只是格式覆盖 —— 而那是可以在现有架构里增量补的,不需要推倒重来。
|
||||
|
||||
## 附:核查方法备忘
|
||||
|
||||
```bash
|
||||
# 拿某个 commit 的完整 patch(不要用 WebFetch,会被 AI 摘要截断)
|
||||
curl -sSL --ssl-no-revoke -o up.patch \
|
||||
"https://github.com/<owner>/<repo>/commit/<sha>.patch"
|
||||
|
||||
# 沙箱内 curl 必须加 --ssl-no-revoke,否则 schannel 报
|
||||
# CRYPT_E_NO_REVOCATION_CHECK (0x80092012)
|
||||
|
||||
# 提取单个文件的 diff
|
||||
sed -n '/^diff --git a\/src\/foo.c/,/^diff --git a\/src\/bar/p' up.patch
|
||||
|
||||
# 只看新增行(去掉 diff 前缀)
|
||||
... | grep '^+' | sed 's/^+//'
|
||||
```
|
||||
@@ -0,0 +1,224 @@
|
||||
<div align="right">
|
||||
简体中文 · 开发者文档见 <a href="../README.zh-CN.md">README</a>
|
||||
</div>
|
||||
|
||||
# PS5 网页文件管理器 · 新手使用说明
|
||||
|
||||
> 适用版本:**v1.9.3M**(版本号末尾的 `M` = 改版,文末有解释)
|
||||
> 本文不假设你懂任何技术名词,照着做即可。
|
||||
|
||||
---
|
||||
|
||||
## 一、它到底是什么
|
||||
|
||||
一句话:**在你的 PS5 上开一个"网页版文件管理器"**。只要设备和 PS5 连着同一个 WiFi,用手机、电脑、甚至 PS5 自带的浏览器打开一个网址,就能像在电脑上一样管理 PS5 里的文件和插在 PS5 上的 U 盘。
|
||||
|
||||
### 能做的事
|
||||
|
||||
| 想干什么 | 可以吗 |
|
||||
|---|---|
|
||||
| 看文件、建文件夹、改名、复制、移动 | ✅ |
|
||||
| 电脑 ↔ PS5 互传文件 | ✅ |
|
||||
| 解压 ZIP / RAR / 7z(**带密码的也行**) | ✅ |
|
||||
| 改小的文本文件(`.txt` `.json` `.ini` 等) | ✅ |
|
||||
| 看图片(`.png` `.jpg` `.gif` `.webp` 等) | ✅ |
|
||||
| 安装 PKG | ✅ |
|
||||
| 改文件权限 | ✅ |
|
||||
|
||||
### 不能做的事(先说清楚,省得白试)
|
||||
|
||||
| 想干什么 | 可以吗 | 说明 |
|
||||
|---|---|---|
|
||||
| 把一堆文件**打包**成压缩包 | ❌ | 它只会"解",不会"压"。下载多个文件时会自动打成一个 `.tar` 包,那只是下载用的 |
|
||||
| 同时做两件事 | ❌ | 一个任务在跑的时候,其它操作会被拒绝(提示"有任务正在执行") |
|
||||
| 删错了找回 | ❌ | 删除是**永久**的,没有回收站 |
|
||||
| 解压 ZIP / RAR / 7z **以外**的格式 | ❌ | 比如 `.tar.gz`、`.iso`、`.xz` 现在不行,见第七节 |
|
||||
|
||||
---
|
||||
|
||||
## 二、三步把它跑起来
|
||||
|
||||
**第 1 步:PS5 上先运行一个"ELF 加载器"**(常见端口是 `9021`)。这一步取决于你用的越狱方案,这里不展开。
|
||||
|
||||
**第 2 步:在电脑上把程序发过去。** 打开终端(Windows 用 Git Bash / PowerShell 都行),执行:
|
||||
|
||||
```sh
|
||||
nc -q0 你的PS5的IP 9021 < web-file-mgr-v1.9.3M.elf
|
||||
```
|
||||
|
||||
例如你的 PS5 是 `192.168.1.50`:
|
||||
|
||||
```sh
|
||||
nc -q0 192.168.1.50 9021 < web-file-mgr-v1.9.3M.elf
|
||||
```
|
||||
|
||||
**第 3 步:看 PS5 左上角的通知。** 它会显示程序名、版本和一个网址,通常是:
|
||||
|
||||
```text
|
||||
http://192.168.1.50:8888/
|
||||
```
|
||||
|
||||
在浏览器里打开这个网址就行。**如果通知显示的端口不是 `8888`(比如 `8889`),以通知为准** —— 端口不是写死的,程序会自动挑一个能用的。
|
||||
|
||||
> 第一次运行时,它还会在 PS5 主屏装一个「PS5 Web File Manager」快捷方式(Media 分类)。已有的不会被覆盖。
|
||||
|
||||
---
|
||||
|
||||
## 三、界面上都是些什么
|
||||
|
||||
打开页面后大致是这样:
|
||||
|
||||
- **最上面 / 侧边**:选"去哪儿"。会看到 **内部存储**、**USB存储**、**M2扩充存储**、**扩展存储**、**根分区** 这几项,点哪个就进哪个。
|
||||
- **中间的大列表**:当前文件夹里的内容,可以按名称、类型、大小、修改时间、权限排序(排序方式会被记住)。
|
||||
- **列表每一行的按钮**:下载、解压(压缩包才有)、安装(`.pkg` 才有)、文本编辑、改权限、删除等。
|
||||
- **上方工具条**:复制、移动、重命名、下载、删除、**解压**、上传(点开选"上传文件 / 上传文件夹")、新建目录、新建文本、刷新、退出。
|
||||
- **解压按钮一直都在**,只是没选中压缩包时是**灰色**的、点不动。选中**一个**压缩包(`.zip` / `.rar` / `.7z`)它才会变亮可点。鼠标停在灰色按钮上会告诉你为什么不能点。
|
||||
- **右下角**:版本号(这里应该显示 `v1.9.3M`)。
|
||||
|
||||
界面截图(点击看大图):
|
||||
|
||||
<p>
|
||||
<a href="screenshots/20260617_231827.376.jpg" target="_blank"><img src="screenshots/20260617_231827.376.jpg" width="31%" alt="界面截图 1"></a>
|
||||
<a href="screenshots/20260619_131432.399.jpg" target="_blank"><img src="screenshots/20260619_131432.399.jpg" width="31%" alt="界面截图 2"></a>
|
||||
<a href="screenshots/20260617_232348.855.jpg" target="_blank"><img src="screenshots/20260617_232348.855.jpg" width="31%" alt="界面截图 3"></a>
|
||||
</p>
|
||||
|
||||
> 小提示:用 **PS5 自带浏览器** 打开时,"上传"和"下载"按钮是隐藏的(PS5 浏览器不支持);要用电脑或手机浏览器才能传文件。
|
||||
|
||||
---
|
||||
|
||||
## 四、六个最常用操作
|
||||
|
||||
### 1. 把电脑上的文件传到 PS5
|
||||
|
||||
1. 用**电脑或手机**浏览器打开那个网址(不要用 PS5 浏览器)。
|
||||
2. 进入你想放到的文件夹。
|
||||
3. 点 **上传**,在弹出的列表里选 **上传文件**(可多选)或 **上传文件夹**(整个目录一起传)。
|
||||
4. 也可以**直接把文件或文件夹拖进网页**——页脚那行灰字就是提醒这件事的。拖进来后会提示"松开即上传到当前目录"。
|
||||
5. 看到全屏进度条就是开始传了,可以随时**取消**。
|
||||
|
||||
### 2. 在 PS5 内部搬运文件(比如 U 盘 → 内置存储)
|
||||
|
||||
1. 选中要搬的项目 → 点 **复制**(保留原文件)或 **移动**(不保留)。
|
||||
2. 进入目标文件夹 → 点 **粘贴**。
|
||||
3. 会有几秒到几十秒的"准备中"(它在统计大小和检查目标空间够不够),别急。
|
||||
|
||||
### 3. 解压一个压缩包
|
||||
|
||||
1. 在列表里找到那个包,点它右边的 **解压**。
|
||||
2. 问你"若目标已存在同名文件或目录"时:
|
||||
- **确定** = 同名文件覆盖掉(同名文件夹会合并进去)
|
||||
- **取消** = 只要目标已经有同名东西就**直接失败**(这是默认,最安全)
|
||||
3. 如果包是加密的,会弹出密码框,输入密码即可。**密码错了会让你重填,最多 3 次。**
|
||||
- 上传压缩包时如果你选了"上传后自动解压",加密包**会自己弹密码框**,不用先手动点一次"解压"。
|
||||
- 第一次弹框写的是"此压缩包已加密"(因为还没输过密码);之后才写"密码不正确"。
|
||||
|
||||
> ⚠️ **强烈建议每次解压都新建一个空文件夹作为目标。** 已知问题:往同一个文件夹里**第二次**用"覆盖"解一个含文件夹的压缩包,会失败。换个空文件夹就没事。
|
||||
|
||||
### 4. 分卷压缩包怎么解
|
||||
|
||||
一个文件被切成好几段的那种(比如 `xxx.part1.rar` / `xxx.part2.rar`,或 `xxx.z01` + `xxx.zip`,或 `xxx.7z.001` …):
|
||||
|
||||
- **RAR:必须点第一个分卷**(`xxx.part1.rar` 或 `xxx.part01.rar`)。点到后面的卷,按钮是灰的,会提示"请改选主卷"。
|
||||
- **ZIP / 7z:点第一卷即可**,其余会自动接上。
|
||||
- **整卷必须在同一个文件夹里**,少一个都会失败(提示"压缩包损坏或不完整")。
|
||||
|
||||
### 5. 改一个文本文件
|
||||
|
||||
点文本文件右边打开编辑器即可。限制:文件 **小于 1 MiB**,且是纯文本(UTF-8)。太大或不是文本会被拒绝。
|
||||
|
||||
### 6. 安装 PKG
|
||||
|
||||
点 `.pkg` 文件右边的 **安装**,会先显示这个包的标题等信息,确认后提交给系统安装。
|
||||
|
||||
---
|
||||
|
||||
## 五、解压的"规矩"(为什么有的包不给解)
|
||||
|
||||
它不是拿到包就解,而是**先检查**,不符合规矩会直接拒绝。这是为了保护你的 PS5 不被一个恶意压缩包搞崩。默认规矩:
|
||||
|
||||
| 规矩 | 上限 |
|
||||
|---|---|
|
||||
| 包里的文件/文件夹数量 | 20 万个 |
|
||||
| 解压后的总大小 | 2 TiB |
|
||||
| 单个文件最大 | 512 GiB |
|
||||
| 压缩比(解压后 ÷ 压缩包) | 500 倍 |
|
||||
| 文件夹最多嵌套多少层 | 32 层 |
|
||||
|
||||
**包特别大时**(磁盘上超过 480 GiB),它会问你要不要开「大文件模式」:开了之后上限放宽到 50 万个 / 4 TiB / 单个 1 TiB / 1000 倍。
|
||||
|
||||
**还有一条最容易踩的:硬盘要留出大约"两份"空间。** 它是先把东西解到一个临时地方,确认全部成功后才整体搬过去 —— 所以一个解压后 50 GB 的包,你得有大约 100 GB 可用空间。空间不够会直接告诉你"需要 X,可用 Y"。
|
||||
|
||||
---
|
||||
|
||||
## 六、看到这句话,是什么意思
|
||||
|
||||
| 界面上显示 | 说人话 | 怎么办 |
|
||||
|---|---|---|
|
||||
| **密码错误,或压缩包未使用所提供的密码加密** | 密码不对,或者这个包根本没加密你却填了密码 | 重填(最多 3 次);确认密码大小写 |
|
||||
| **此压缩包已加密,密码不正确** | 同上,重试提示 | 输入正确密码,或取消 |
|
||||
| **目标空间不足,需要 X,可用 Y** | 硬盘不够 | 记住要留**两份**空间;删东西或换更大的盘 |
|
||||
| **压缩包内单个文件过大** | 包里有个超大文件,超过默认 512 GiB | 出现提示时选「大文件模式」;或拆包 |
|
||||
| **压缩比异常(疑似压缩炸弹)** | 一个很小的包声称能解出一大堆东西,被判定为危险 | 基本是恶意包,别解 |
|
||||
| **目标已存在同名文件或目录** | 目标位置已经有同名东西了 | 换一个空文件夹(推荐),或在提示时选"确定"覆盖 |
|
||||
| **压缩包损坏或不完整** | 包坏了,或分卷少了一卷 | 重新下载/拷贝,确认所有分卷都在同一目录 |
|
||||
| **不支持的压缩包** | 不是 ZIP / RAR / 7z,或用了它处理不了的特性 | 在电脑上解好再传 |
|
||||
| **压缩包需要的字典超出本机可承受范围** | 极少数用超大设置压的 RAR,PS5 解不动 | 在电脑上用普通设置重新压一次 |
|
||||
| **请改选主卷** | 你点的是分卷的后面几卷 | 点第一个分卷(`.part1.rar` / `.part01.rar`) |
|
||||
| **有任务正在执行** | 已经有一个活儿在干 | 等它结束,或先取消它 |
|
||||
| **操作失败** | 其它错误 | 记下原文,反馈时带上 |
|
||||
|
||||
---
|
||||
|
||||
## 七、和原版(上游)有什么不一样
|
||||
|
||||
这个程序是从开源项目 **owendswang/ps5-web-file-manager** 改来的。下面是和你有关的差别,说人话:
|
||||
|
||||
| 对你意味着什么 | 原版(上游 v1.8) | 本版(v1.9.3M) |
|
||||
|---|---|---|
|
||||
| **要装几个东西** | **两个**:主程序 + 一个上百 MB 的 7-Zip 辅助程序,还得放到固定目录 `/data/wfm/`。**辅助文件丢了,解压功能直接全废** | **就一个文件**,拷上去就能用 |
|
||||
| **能解多少种格式** | **约 30 种**(`.tar.gz` `.xz` `.cab` `.iso` 类……) | **3 种**:`.zip` `.rar` `.7z` |
|
||||
| **带密码的压缩包** | 靠 7-Zip 支持 | **三种格式都支持**,密码错了会弹框让你重填(最多 3 次) |
|
||||
| **RAR 分卷** | 支持 | 支持(要点第一个分卷) |
|
||||
| **防"压缩炸弹"/防硬盘写满** | ❌ **没有**,交给 7-Zip 自己看着办 | ✅ 有:压缩比筛查、提前算空间够不够 |
|
||||
| **失败后会不会留下一堆半成品** | 会(直接解到目标目录,中断就留在那儿) | 不会:先解到临时地方,全部成功才整体搬过去,失败自动清理 |
|
||||
| **程序被重启后,正在解的任务** | 还能接着跑(辅助程序是独立进程) | **会丢**(关掉浏览器再打开还能看到进度,但程序本身重启就没了) |
|
||||
| **出错时的提示** | 比较笼统 | 会告诉你具体是哪个文件、差多少字节 |
|
||||
| **版本号长什么样** | `v1.8`、`v1.9` 这样纯数字 | **`v1.9.3M`**,末尾多一个 `M` |
|
||||
|
||||
### 一句话总结
|
||||
|
||||
- **原版赢在"格式多"**:`.tar.gz`、`.xz` 这类它也能解,本版不行。如果你经常遇到这三种以外的格式,原版更方便。
|
||||
- **本版赢在"省心和安全"**:一个文件就完事,不会因为你少放一个辅助文件就整个不能用;也不会被一个恶意压缩包写满你的内置存储、或者在半路留一堆垃圾。
|
||||
|
||||
### `M` 是什么意思
|
||||
|
||||
版本号末尾的 **`M`** = **Modified(改版)**。上游原版是纯数字(如 `v1.8`),所以:
|
||||
|
||||
- 看到 **`v1.9.3M`** → 这是本仓的改版
|
||||
- 看到 **`v1.9.3`**(没有 M)→ 那不是本仓出的
|
||||
|
||||
这个字母会同时出现在:文件名、PS5 启动通知、网页右下角。网页右下角把鼠标**悬停**在版本号上,会弹出说明文字。
|
||||
|
||||
---
|
||||
|
||||
## 八、怎么确认你装的是哪一版
|
||||
|
||||
| 看哪里 | 应该显示 |
|
||||
|---|---|
|
||||
| 网页右下角 | `v1.9.3M` |
|
||||
| PS5 启动时的通知 | 程序名 + `v1.9.3M` + 监听端口 |
|
||||
| 浏览器打开 `http://<PS5的IP>:<端口>/api/version` | JSON 里的版本号 = `v1.9.3M` |
|
||||
|
||||
**只要界面上显示带 `M` 的版本号,就说明装对了。**
|
||||
|
||||
---
|
||||
|
||||
## 九、几条重要提醒
|
||||
|
||||
1. **解压时别断电、别重启 PS5。** 它为了速度不做逐文件强制落盘,如果在最后搬运阶段断电,可能出现"文件在,但内容不完整"。
|
||||
2. **解压前确认空间够**,而且是**两份**(见第五节)。
|
||||
3. **每次解压用新的空文件夹**,别往同一个目录连解两次(见第四节第 3 条)。
|
||||
4. **删除不可恢复。** 删文件夹会把它里面所有东西都删掉,弹窗会提醒。
|
||||
5. **一次只做一件事**,有任务在跑时其它操作会被拒绝。
|
||||
6. 这是自制程序。如果遇到 PS5 内核崩溃,请换更新的越狱方式 / ELF 加载器,或回到你常用的稳定方案。
|
||||
|
After Width: | Height: | Size: 279 KiB |
|
After Width: | Height: | Size: 146 KiB |
|
After Width: | Height: | Size: 153 KiB |
|
After Width: | Height: | Size: 124 KiB |
|
After Width: | Height: | Size: 116 KiB |
|
After Width: | Height: | Size: 290 KiB |
@@ -4,12 +4,13 @@
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <sys/stat.h>
|
||||
|
||||
#include <ps5/kernel.h>
|
||||
|
||||
#include "asset.h"
|
||||
#include "pkg_installer.h"
|
||||
|
||||
#define INCASSET(name, file) \
|
||||
__asm__(".section .rodata\n" \
|
||||
".global " #name "\n" \
|
||||
@@ -28,14 +29,20 @@
|
||||
INCASSET(param_json, "assets/param.json");
|
||||
INCASSET(icon0_png, "assets/icon0.png");
|
||||
|
||||
int sceAppInstUtilInitialize(void);
|
||||
int sceAppInstUtilAppInstallAll(void *);
|
||||
|
||||
static int
|
||||
install_file(const char *path, const uint8_t *data, size_t size) {
|
||||
struct stat st;
|
||||
FILE *f;
|
||||
|
||||
if(!(f = fopen(path, "w"))) {
|
||||
if(!stat(path, &st)) {
|
||||
return 0;
|
||||
}
|
||||
if(errno != ENOENT) {
|
||||
return -1;
|
||||
}
|
||||
if(!(f = fopen(path, "wb"))) {
|
||||
return -1;
|
||||
}
|
||||
if(fwrite(data, size, 1, f) != 1) {
|
||||
@@ -65,32 +72,13 @@ install_app(const char *title_id, const char *dir) {
|
||||
}
|
||||
|
||||
static int
|
||||
needs_update(const char *path, const uint8_t *expected, size_t expected_size) {
|
||||
needs_install_file(const char *path) {
|
||||
struct stat st;
|
||||
uint8_t *buf;
|
||||
FILE *f;
|
||||
int mismatch;
|
||||
|
||||
if(stat(path, &st) || st.st_size != (off_t)expected_size) {
|
||||
if(stat(path, &st)) {
|
||||
return 1;
|
||||
}
|
||||
if(!(f = fopen(path, "r"))) {
|
||||
return 1;
|
||||
}
|
||||
if(!(buf = malloc(expected_size))) {
|
||||
fclose(f);
|
||||
return 1;
|
||||
}
|
||||
if(fread(buf, 1, expected_size, f) != expected_size) {
|
||||
free(buf);
|
||||
fclose(f);
|
||||
return 1;
|
||||
}
|
||||
fclose(f);
|
||||
|
||||
mismatch = memcmp(buf, expected, expected_size);
|
||||
free(buf);
|
||||
return mismatch != 0;
|
||||
return 0;
|
||||
}
|
||||
|
||||
int
|
||||
@@ -104,6 +92,8 @@ app_install_if_needed(void) {
|
||||
int update_needed = 0;
|
||||
int err;
|
||||
|
||||
asset_register("/icon0.png", icon0_png, icon0_png_size, "image/png", 0);
|
||||
|
||||
snprintf(base_dir, sizeof(base_dir), "/user/app/%s", title_id);
|
||||
snprintf(sce_sys_dir, sizeof(sce_sys_dir), "/user/app/%s/sce_sys", title_id);
|
||||
snprintf(param_path, sizeof(param_path), "%s/param.json", sce_sys_dir);
|
||||
@@ -111,8 +101,7 @@ app_install_if_needed(void) {
|
||||
|
||||
if(stat(base_dir, &st)) {
|
||||
update_needed = 1;
|
||||
} else if(needs_update(param_path, param_json, param_json_size) ||
|
||||
needs_update(icon_path, icon0_png, icon0_png_size)) {
|
||||
} else if(needs_install_file(param_path) || needs_install_file(icon_path)) {
|
||||
update_needed = 1;
|
||||
}
|
||||
|
||||
@@ -122,7 +111,7 @@ app_install_if_needed(void) {
|
||||
|
||||
printf("Installing launcher app %s\n", title_id);
|
||||
|
||||
if((err = sceAppInstUtilInitialize())) {
|
||||
if((err = pkg_installer_initialize())) {
|
||||
printf("sceAppInstUtilInitialize: error 0x%08X\n", err);
|
||||
return -1;
|
||||
}
|
||||
|
||||
@@ -0,0 +1,26 @@
|
||||
/* PS5-only stub for the libgcc CPU model symbols.
|
||||
*
|
||||
* unrar's rijndael.cpp / system.cpp call __builtin_cpu_supports() to pick
|
||||
* AES-NI fast paths. Clang lowers that to a reference on __cpu_model (data)
|
||||
* and __cpu_indicator_init() (function), which the FreeBSD-style PS5
|
||||
* sysroot does not provide (no libgcc). This TU supplies both so the link
|
||||
* succeeds; the detection result is unused because we always build the
|
||||
* portable C path.
|
||||
*
|
||||
* Do NOT add this file to host/linux builds: libstdc++/libgcc already
|
||||
* define __cpu_model there and the symbols would collide. */
|
||||
#if defined(__x86_64__) && !defined(__linux__) && !defined(_WIN32)
|
||||
|
||||
struct __cpu_model {
|
||||
int __cpu_vendor;
|
||||
int __cpu_type;
|
||||
int __cpu_subtype;
|
||||
};
|
||||
|
||||
int __cpu_indicator_init(void) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
struct __cpu_model __cpu_model = { 0, 0, 0 };
|
||||
|
||||
#endif
|
||||
@@ -0,0 +1,31 @@
|
||||
/* Stub for the C++ name demangler (__cxa_demangle).
|
||||
*
|
||||
* unrar7 is compiled with C++ exceptions enabled -- dll.cpp catches
|
||||
* RAR_EXIT and std::bad_alloc, unpack.cpp/model.cpp throw bad_alloc -- so
|
||||
* the runtime's __cxa_throw chain holds a reference to __cxa_demangle. That
|
||||
* one reference drags the whole Itanium demangler TU into the link: 607
|
||||
* symbols, ~105 KiB, over 10% of the final ELF (see docs/SIZE-OPTIMIZATION.md).
|
||||
*
|
||||
* __cxa_demangle is only ever reached on the uncaught-exception diagnostic
|
||||
* path (std::terminate printing the exception's type name). Every unrar
|
||||
* exception is caught inside dll.cpp, so that path is unreachable here.
|
||||
* Defining the symbol in our own TU keeps cxa_demangle.o out of the archive
|
||||
* pull -- the linker resolves against ours and never opens the member.
|
||||
*
|
||||
* Returning NULL is the documented "demangle failed" result; the caller
|
||||
* falls back to printing the mangled name. Exception handling itself
|
||||
* (__cxa_throw / __cxa_begin_catch / _Unwind_Resume / __gxx_personality_v0)
|
||||
* is untouched. Applies to both the PS5 and the host/linux builds. */
|
||||
#include <stddef.h>
|
||||
|
||||
char *__cxa_demangle(const char *mangled, char *buf, size_t *len, int *status) {
|
||||
(void)mangled;
|
||||
(void)buf;
|
||||
(void)len;
|
||||
|
||||
if(status) {
|
||||
*status = -1;
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,761 @@
|
||||
#include <dirent.h>
|
||||
#include <errno.h>
|
||||
#include <fcntl.h>
|
||||
#include <limits.h>
|
||||
#include <stdint.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <sys/stat.h>
|
||||
#include <time.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#include "filemgr_internal.h"
|
||||
#include "json_util.h"
|
||||
#include "path_util.h"
|
||||
#include "websrv.h"
|
||||
|
||||
#define DOWNLOAD_ARCHIVE_NAME_LIMIT 20
|
||||
#define DOWNLOAD_BUFFER_SIZE (2 * 1024 * 1024)
|
||||
|
||||
typedef struct tar_frame {
|
||||
char path[PATH_MAX];
|
||||
char name[PATH_MAX];
|
||||
DIR *dir;
|
||||
int header_sent;
|
||||
struct tar_frame *next;
|
||||
} tar_frame_t;
|
||||
|
||||
typedef struct tar_stream {
|
||||
file_task_t *task;
|
||||
char **paths;
|
||||
size_t path_count;
|
||||
size_t path_index;
|
||||
size_t stack_depth;
|
||||
tar_frame_t *stack;
|
||||
int fd;
|
||||
char current_file[PATH_MAX];
|
||||
unsigned long long file_remaining;
|
||||
size_t file_padding;
|
||||
size_t pad_remaining;
|
||||
char *pending;
|
||||
size_t pending_size;
|
||||
size_t pending_offset;
|
||||
int final_blocks;
|
||||
int done;
|
||||
int error;
|
||||
} tar_stream_t;
|
||||
|
||||
typedef struct download_file_stream {
|
||||
file_task_t *task;
|
||||
int fd;
|
||||
unsigned long long size;
|
||||
unsigned long long sent;
|
||||
int done;
|
||||
int error;
|
||||
} download_file_stream_t;
|
||||
|
||||
static enum MHD_Result
|
||||
download_task_request_error(struct MHD_Connection *conn, file_task_t *task,
|
||||
char **paths, size_t count, unsigned int status,
|
||||
const char *msg) {
|
||||
free_paths(paths, count);
|
||||
free_task(task);
|
||||
return send_json_error(conn, status, msg);
|
||||
}
|
||||
|
||||
static int
|
||||
tar_checksum(char *header) {
|
||||
int sum = 0;
|
||||
int i;
|
||||
|
||||
memset(header + 148, ' ', 8);
|
||||
for(i = 0; i < 512; i++) {
|
||||
sum += (unsigned char)header[i];
|
||||
}
|
||||
snprintf(header + 148, 8, "%06o", sum);
|
||||
header[154] = 0;
|
||||
header[155] = ' ';
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int
|
||||
tar_split_name(const char *name, char *out_name, size_t out_name_size,
|
||||
char *prefix, size_t prefix_size) {
|
||||
size_t len = strlen(name);
|
||||
const char *slash;
|
||||
|
||||
memset(out_name, 0, out_name_size);
|
||||
memset(prefix, 0, prefix_size);
|
||||
if(len < out_name_size) {
|
||||
strcpy(out_name, name);
|
||||
return 0;
|
||||
}
|
||||
for(slash = name + len; slash > name; slash--) {
|
||||
size_t prefix_len;
|
||||
size_t name_len;
|
||||
|
||||
if(*slash != '/') {
|
||||
continue;
|
||||
}
|
||||
prefix_len = (size_t)(slash - name);
|
||||
name_len = len - prefix_len - 1;
|
||||
if(prefix_len < prefix_size && name_len > 0 && name_len < out_name_size) {
|
||||
memcpy(prefix, name, prefix_len);
|
||||
prefix[prefix_len] = 0;
|
||||
memcpy(out_name, slash + 1, name_len + 1);
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
errno = ENAMETOOLONG;
|
||||
return -1;
|
||||
}
|
||||
|
||||
static int
|
||||
tar_queue_header(tar_stream_t *s, const char *name, const struct stat *st,
|
||||
char type) {
|
||||
char header[512];
|
||||
char tar_name[100];
|
||||
char prefix[155];
|
||||
unsigned int mode = (unsigned int)(st->st_mode & 07777);
|
||||
unsigned long long size = type == '0' ? (unsigned long long)st->st_size : 0;
|
||||
long long mtime = (long long)st->st_mtime;
|
||||
|
||||
if(tar_split_name(name, tar_name, sizeof(tar_name), prefix, sizeof(prefix))) {
|
||||
return -1;
|
||||
}
|
||||
memset(header, 0, sizeof(header));
|
||||
memcpy(header, tar_name, strlen(tar_name));
|
||||
snprintf(header + 100, 8, "%07o", mode);
|
||||
snprintf(header + 108, 8, "%07o", 0);
|
||||
snprintf(header + 116, 8, "%07o", 0);
|
||||
snprintf(header + 124, 12, "%011llo", size);
|
||||
snprintf(header + 136, 12, "%011llo", (unsigned long long)mtime);
|
||||
header[156] = type;
|
||||
memcpy(header + 257, "ustar", 5);
|
||||
memcpy(header + 263, "00", 2);
|
||||
if(prefix[0]) {
|
||||
memcpy(header + 345, prefix, strlen(prefix));
|
||||
}
|
||||
tar_checksum(header);
|
||||
s->pending = malloc(sizeof(header));
|
||||
if(!s->pending) {
|
||||
errno = ENOMEM;
|
||||
return -1;
|
||||
}
|
||||
memcpy(s->pending, header, sizeof(header));
|
||||
s->pending_size = sizeof(header);
|
||||
s->pending_offset = 0;
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int
|
||||
tar_push_dir(tar_stream_t *s, const char *path, const char *name) {
|
||||
tar_frame_t *frame = calloc(1, sizeof(*frame));
|
||||
|
||||
if(!frame) {
|
||||
errno = ENOMEM;
|
||||
return -1;
|
||||
}
|
||||
snprintf(frame->path, sizeof(frame->path), "%s", path);
|
||||
snprintf(frame->name, sizeof(frame->name), "%s", name);
|
||||
frame->dir = opendir(path);
|
||||
if(!frame->dir) {
|
||||
int error = errno;
|
||||
free(frame);
|
||||
errno = error;
|
||||
return -1;
|
||||
}
|
||||
frame->next = s->stack;
|
||||
s->stack = frame;
|
||||
s->stack_depth++;
|
||||
return 0;
|
||||
}
|
||||
|
||||
static void
|
||||
tar_pop_dir(tar_stream_t *s) {
|
||||
tar_frame_t *frame = s->stack;
|
||||
|
||||
if(!frame) {
|
||||
return;
|
||||
}
|
||||
s->stack = frame->next;
|
||||
if(s->stack_depth) {
|
||||
s->stack_depth--;
|
||||
}
|
||||
if(frame->dir) {
|
||||
closedir(frame->dir);
|
||||
}
|
||||
free(frame);
|
||||
}
|
||||
|
||||
static int
|
||||
tar_start_path(tar_stream_t *s, const char *path, const char *name) {
|
||||
struct stat st;
|
||||
|
||||
if(lstat(path, &st)) {
|
||||
return -1;
|
||||
}
|
||||
if(S_ISDIR(st.st_mode)) {
|
||||
char dir_name[PATH_MAX];
|
||||
snprintf(dir_name, sizeof(dir_name), "%s%s", name,
|
||||
name[strlen(name) - 1] == '/' ? "" : "/");
|
||||
return tar_push_dir(s, path, dir_name);
|
||||
}
|
||||
if(!S_ISREG(st.st_mode)) {
|
||||
errno = ENOTSUP;
|
||||
return -1;
|
||||
}
|
||||
if(tar_queue_header(s, name, &st, '0')) {
|
||||
return -1;
|
||||
}
|
||||
s->fd = open(path, O_RDONLY);
|
||||
if(s->fd < 0) {
|
||||
return -1;
|
||||
}
|
||||
snprintf(s->current_file, sizeof(s->current_file), "%s", path);
|
||||
s->file_remaining = (unsigned long long)st.st_size;
|
||||
s->file_padding = (size_t)((512 - ((unsigned long long)st.st_size % 512)) % 512);
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int
|
||||
tar_prepare_next(tar_stream_t *s) {
|
||||
while(!s->pending && s->fd < 0 && !s->done) {
|
||||
if(s->task && task_cancel_requested(s->task)) {
|
||||
errno = ECANCELED;
|
||||
return -1;
|
||||
}
|
||||
if(s->pad_remaining) {
|
||||
size_t size = s->pad_remaining > 512 ? 512 : s->pad_remaining;
|
||||
s->pending = calloc(1, size);
|
||||
if(!s->pending) {
|
||||
errno = ENOMEM;
|
||||
return -1;
|
||||
}
|
||||
s->pending_size = size;
|
||||
s->pending_offset = 0;
|
||||
s->pad_remaining -= size;
|
||||
return 0;
|
||||
}
|
||||
if(s->stack) {
|
||||
tar_frame_t *frame = s->stack;
|
||||
struct dirent *entry;
|
||||
struct stat st;
|
||||
|
||||
if(!frame->header_sent) {
|
||||
frame->header_sent = 1;
|
||||
if(lstat(frame->path, &st) ||
|
||||
tar_queue_header(s, frame->name, &st, '5')) {
|
||||
return -1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
while((entry = readdir(frame->dir))) {
|
||||
char child[PATH_MAX];
|
||||
char child_name[PATH_MAX];
|
||||
|
||||
if(!strcmp(entry->d_name, ".") || !strcmp(entry->d_name, "..")) {
|
||||
continue;
|
||||
}
|
||||
if(path_join(child, sizeof(child), frame->path, entry->d_name) ||
|
||||
path_join(child_name, sizeof(child_name), frame->name,
|
||||
entry->d_name) ||
|
||||
tar_start_path(s, child, child_name)) {
|
||||
return -1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
tar_pop_dir(s);
|
||||
continue;
|
||||
}
|
||||
if(s->path_index < s->path_count) {
|
||||
const char *path = s->paths[s->path_index++];
|
||||
if(tar_start_path(s, path, path_basename(path))) {
|
||||
return -1;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
if(s->final_blocks < 2) {
|
||||
s->pending = calloc(1, 512);
|
||||
if(!s->pending) {
|
||||
errno = ENOMEM;
|
||||
return -1;
|
||||
}
|
||||
s->pending_size = 512;
|
||||
s->pending_offset = 0;
|
||||
s->final_blocks++;
|
||||
return 0;
|
||||
}
|
||||
s->done = 1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
static ssize_t
|
||||
tar_read(void *cls, uint64_t pos, char *buf, size_t max) {
|
||||
tar_stream_t *s = cls;
|
||||
size_t out = 0;
|
||||
unsigned long long progress = 0;
|
||||
char progress_current[PATH_MAX] = {0};
|
||||
(void)pos;
|
||||
|
||||
while(out < max && !s->done) {
|
||||
if(s->task && task_cancel_requested(s->task)) {
|
||||
s->error = ECANCELED;
|
||||
return out ? (ssize_t)out : MHD_CONTENT_READER_END_WITH_ERROR;
|
||||
}
|
||||
if(tar_prepare_next(s)) {
|
||||
s->error = errno ? errno : EIO;
|
||||
return out ? (ssize_t)out : MHD_CONTENT_READER_END_WITH_ERROR;
|
||||
}
|
||||
if(s->pending) {
|
||||
size_t left = s->pending_size - s->pending_offset;
|
||||
size_t take = left < max - out ? left : max - out;
|
||||
memcpy(buf + out, s->pending + s->pending_offset, take);
|
||||
s->pending_offset += take;
|
||||
out += take;
|
||||
if(s->pending_offset >= s->pending_size) {
|
||||
free(s->pending);
|
||||
s->pending = NULL;
|
||||
s->pending_size = 0;
|
||||
s->pending_offset = 0;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
if(s->fd >= 0) {
|
||||
size_t want = max - out;
|
||||
ssize_t n;
|
||||
if((unsigned long long)want > s->file_remaining) {
|
||||
want = (size_t)s->file_remaining;
|
||||
}
|
||||
n = read(s->fd, buf + out, want);
|
||||
if(n < 0) {
|
||||
s->error = errno;
|
||||
return out ? (ssize_t)out : MHD_CONTENT_READER_END_WITH_ERROR;
|
||||
}
|
||||
if(!n) {
|
||||
close(s->fd);
|
||||
s->fd = -1;
|
||||
s->current_file[0] = 0;
|
||||
s->file_remaining = 0;
|
||||
continue;
|
||||
}
|
||||
out += (size_t)n;
|
||||
s->file_remaining -= (unsigned long long)n;
|
||||
progress += (unsigned long long)n;
|
||||
if(s->current_file[0]) {
|
||||
snprintf(progress_current, sizeof(progress_current), "%s",
|
||||
s->current_file);
|
||||
}
|
||||
if(!s->file_remaining) {
|
||||
close(s->fd);
|
||||
s->fd = -1;
|
||||
s->current_file[0] = 0;
|
||||
s->pad_remaining = s->file_padding;
|
||||
s->file_padding = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
if(s->task && progress) {
|
||||
task_update(s->task, TASK_RUNNING,
|
||||
progress_current[0] ? progress_current : NULL,
|
||||
progress, NULL);
|
||||
}
|
||||
return out ? (ssize_t)out : MHD_CONTENT_READER_END_OF_STREAM;
|
||||
}
|
||||
|
||||
static void
|
||||
tar_close(void *cls) {
|
||||
tar_stream_t *s = cls;
|
||||
int canceled;
|
||||
|
||||
if(!s) {
|
||||
return;
|
||||
}
|
||||
if(s->fd >= 0) {
|
||||
close(s->fd);
|
||||
}
|
||||
while(s->stack) {
|
||||
tar_pop_dir(s);
|
||||
}
|
||||
canceled = s->task && task_cancel_requested(s->task);
|
||||
if(s->task) {
|
||||
char current[PATH_MAX];
|
||||
snprintf(current, sizeof(current), "%s",
|
||||
s->task->current[0] ? s->task->current : s->task->src);
|
||||
if(canceled) {
|
||||
task_update(s->task, TASK_CANCELED, current, 0, "canceled");
|
||||
} else if(s->error) {
|
||||
errno = s->error;
|
||||
task_update(s->task, TASK_FAILED, current, 0, strerror(errno));
|
||||
} else if(s->done) {
|
||||
task_update(s->task, TASK_DONE, current, 0, NULL);
|
||||
} else {
|
||||
task_update(s->task, TASK_FAILED, current, 0, "client disconnected");
|
||||
}
|
||||
}
|
||||
free(s->pending);
|
||||
free_paths(s->paths, s->path_count);
|
||||
free(s);
|
||||
}
|
||||
|
||||
static int
|
||||
prepare_download_task(file_task_t *task) {
|
||||
unsigned long long total = 0;
|
||||
size_t file_count = 0;
|
||||
size_t dir_count = 0;
|
||||
size_t i;
|
||||
|
||||
task_update(task, TASK_RUNNING, "preparing", 0, NULL);
|
||||
for(i = 0; i < task->src_count; i++) {
|
||||
if(count_task_path_bytes(task, task->srcs[i], task->srcs[i],
|
||||
&total, &file_count, &dir_count)) {
|
||||
return -1;
|
||||
}
|
||||
}
|
||||
pthread_mutex_lock(&g_tasks_lock);
|
||||
task->total = total;
|
||||
task->file_count = file_count;
|
||||
task->dir_count = dir_count;
|
||||
task->updated_at = time(NULL);
|
||||
pthread_mutex_unlock(&g_tasks_lock);
|
||||
return 0;
|
||||
}
|
||||
|
||||
static size_t
|
||||
append_truncated_name(char *out, size_t size, size_t len, const char *name,
|
||||
size_t limit) {
|
||||
while(*name && len < limit && len + 1 < size) {
|
||||
unsigned char c = (unsigned char)*name;
|
||||
out[len++] = *name++;
|
||||
if(c >= 0x80) {
|
||||
while((*name & 0xc0) == 0x80 && len < limit && len + 1 < size) {
|
||||
out[len++] = *name++;
|
||||
}
|
||||
}
|
||||
}
|
||||
out[len] = 0;
|
||||
return len;
|
||||
}
|
||||
|
||||
static void
|
||||
download_archive_name(char *out, size_t size, char **paths, size_t count) {
|
||||
char name[PATH_MAX];
|
||||
size_t len = 0;
|
||||
size_t i;
|
||||
|
||||
out[0] = 0;
|
||||
if(count == 1) {
|
||||
path_basename_copy(paths[0], name, sizeof(name));
|
||||
append_truncated_name(out, size, 0, name, DOWNLOAD_ARCHIVE_NAME_LIMIT);
|
||||
} else {
|
||||
for(i = 0; i < count && len < DOWNLOAD_ARCHIVE_NAME_LIMIT; i++) {
|
||||
path_basename_copy(paths[i], name, sizeof(name));
|
||||
if(i && len + 1 < DOWNLOAD_ARCHIVE_NAME_LIMIT && len + 1 < size) {
|
||||
out[len++] = ' ';
|
||||
out[len] = 0;
|
||||
}
|
||||
len = append_truncated_name(out, size, len, name,
|
||||
DOWNLOAD_ARCHIVE_NAME_LIMIT);
|
||||
}
|
||||
}
|
||||
if(!out[0]) {
|
||||
snprintf(out, size, "download");
|
||||
}
|
||||
snprintf(out + strlen(out), size - strlen(out), ".tar");
|
||||
}
|
||||
|
||||
static void
|
||||
download_file_name(char *out, size_t size, const char *path) {
|
||||
path_basename_copy(path, out, size);
|
||||
if(!out[0]) {
|
||||
snprintf(out, size, "download");
|
||||
}
|
||||
}
|
||||
|
||||
static size_t
|
||||
header_quoted_filename(char *out, size_t size, const char *name) {
|
||||
size_t len = 0;
|
||||
|
||||
while(*name && len + 1 < size) {
|
||||
unsigned char c = (unsigned char)*name++;
|
||||
if(c < 0x20 || c == 0x7f || c == '"' || c == '\\') {
|
||||
c = '_';
|
||||
}
|
||||
out[len++] = (char)c;
|
||||
}
|
||||
out[len] = 0;
|
||||
return len;
|
||||
}
|
||||
|
||||
static size_t
|
||||
header_percent_filename(char *out, size_t size, const char *name) {
|
||||
static const char hex[] = "0123456789ABCDEF";
|
||||
size_t len = 0;
|
||||
|
||||
while(*name && len + 1 < size) {
|
||||
unsigned char c = (unsigned char)*name++;
|
||||
int safe = (c >= '0' && c <= '9') || (c >= 'A' && c <= 'Z') ||
|
||||
(c >= 'a' && c <= 'z') || c == '.' || c == '_' || c == '-';
|
||||
if(safe) {
|
||||
out[len++] = (char)c;
|
||||
} else {
|
||||
if(len + 3 >= size) {
|
||||
break;
|
||||
}
|
||||
out[len++] = '%';
|
||||
out[len++] = hex[c >> 4];
|
||||
out[len++] = hex[c & 15];
|
||||
}
|
||||
}
|
||||
out[len] = 0;
|
||||
return len;
|
||||
}
|
||||
|
||||
static void
|
||||
add_download_filename_header(struct MHD_Response *resp, const char *name) {
|
||||
char quoted[PATH_MAX];
|
||||
char encoded[PATH_MAX * 3];
|
||||
char header[PATH_MAX * 4];
|
||||
|
||||
header_quoted_filename(quoted, sizeof(quoted), name);
|
||||
header_percent_filename(encoded, sizeof(encoded), name);
|
||||
snprintf(header, sizeof(header),
|
||||
"attachment; filename=\"%s\"; filename*=UTF-8''%s",
|
||||
quoted, encoded);
|
||||
MHD_add_response_header(resp, "Content-Disposition", header);
|
||||
}
|
||||
|
||||
static enum MHD_Result
|
||||
create_download_task_response(struct MHD_Connection *conn, char **paths,
|
||||
size_t count) {
|
||||
file_task_t *task = calloc(1, sizeof(file_task_t));
|
||||
strbuf_t b = {0};
|
||||
struct stat st;
|
||||
size_t i;
|
||||
|
||||
if(!task) {
|
||||
free_paths(paths, count);
|
||||
return send_json_error(conn, MHD_HTTP_INTERNAL_SERVER_ERROR, "out of memory");
|
||||
}
|
||||
if(!count) {
|
||||
return download_task_request_error(conn, task, paths, count,
|
||||
MHD_HTTP_BAD_REQUEST, "no source paths");
|
||||
}
|
||||
for(i = 0; i < count; i++) {
|
||||
if(lstat(paths[i], &st)) {
|
||||
return download_task_request_error(conn, task, paths, count,
|
||||
MHD_HTTP_NOT_FOUND, "file not found");
|
||||
}
|
||||
if(!S_ISREG(st.st_mode) && !S_ISDIR(st.st_mode)) {
|
||||
return download_task_request_error(conn, task, paths, count,
|
||||
MHD_HTTP_BAD_REQUEST, "invalid path");
|
||||
}
|
||||
}
|
||||
task->op = TASK_DOWNLOAD;
|
||||
task->state = TASK_QUEUED;
|
||||
task->srcs = paths;
|
||||
task->src_count = count;
|
||||
snprintf(task->src, sizeof(task->src), "%s%s", paths[0],
|
||||
count > 1 ? " ..." : "");
|
||||
snprintf(task->current, sizeof(task->current), "%s", task->src);
|
||||
task->created_at = time(NULL);
|
||||
task->updated_at = task->created_at;
|
||||
|
||||
pthread_mutex_lock(&g_tasks_lock);
|
||||
remove_finished_tasks_locked();
|
||||
if(has_active_task_locked()) {
|
||||
pthread_mutex_unlock(&g_tasks_lock);
|
||||
return download_task_request_error(conn, task, paths, count,
|
||||
MHD_HTTP_CONFLICT, "another task is running");
|
||||
}
|
||||
task->id = g_next_task_id++;
|
||||
task->next = g_tasks;
|
||||
g_tasks = task;
|
||||
pthread_mutex_unlock(&g_tasks_lock);
|
||||
|
||||
strbuf_printf(&b, "{\"ok\":true,\"task_id\":%lu}", task->id);
|
||||
return send_buffer(conn, MHD_HTTP_OK, b.data, "application/json");
|
||||
}
|
||||
|
||||
enum MHD_Result
|
||||
api_download_prepare(struct MHD_Connection *conn, const char *body,
|
||||
size_t body_size) {
|
||||
char *paths_raw = body_form_value(body, body_size, "paths");
|
||||
char **paths = NULL;
|
||||
size_t count = 0;
|
||||
|
||||
if(!paths_raw || parse_paths(paths_raw, &paths, &count)) {
|
||||
free(paths_raw);
|
||||
return send_json_error(conn, MHD_HTTP_BAD_REQUEST, "invalid path");
|
||||
}
|
||||
free(paths_raw);
|
||||
return create_download_task_response(conn, paths, count);
|
||||
}
|
||||
|
||||
static ssize_t
|
||||
download_file_read(void *cls, uint64_t pos, char *buf, size_t max) {
|
||||
download_file_stream_t *s = cls;
|
||||
ssize_t len;
|
||||
|
||||
if(task_cancel_requested(s->task)) {
|
||||
s->error = ECANCELED;
|
||||
return MHD_CONTENT_READER_END_WITH_ERROR;
|
||||
}
|
||||
len = pread(s->fd, buf, max, (off_t)pos);
|
||||
if(len < 0) {
|
||||
s->error = errno ? errno : EIO;
|
||||
return MHD_CONTENT_READER_END_WITH_ERROR;
|
||||
}
|
||||
if(!len) {
|
||||
s->done = 1;
|
||||
return MHD_CONTENT_READER_END_OF_STREAM;
|
||||
}
|
||||
{
|
||||
unsigned long long end = (unsigned long long)pos + (unsigned long long)len;
|
||||
unsigned long long add = end > s->sent ? end - s->sent : 0;
|
||||
if(end > s->sent) {
|
||||
s->sent = end;
|
||||
}
|
||||
if(s->sent >= s->size) {
|
||||
s->done = 1;
|
||||
}
|
||||
task_update(s->task, TASK_RUNNING, s->task->src, add, NULL);
|
||||
}
|
||||
return len;
|
||||
}
|
||||
|
||||
static void
|
||||
download_file_close(void *cls) {
|
||||
download_file_stream_t *s = cls;
|
||||
int canceled;
|
||||
|
||||
if(!s) {
|
||||
return;
|
||||
}
|
||||
if(s->fd >= 0) {
|
||||
close(s->fd);
|
||||
}
|
||||
canceled = task_cancel_requested(s->task);
|
||||
if(canceled) {
|
||||
task_update(s->task, TASK_CANCELED, s->task->src, 0, "canceled");
|
||||
} else if(s->error) {
|
||||
errno = s->error;
|
||||
task_update(s->task, TASK_FAILED, s->task->src, 0, strerror(errno));
|
||||
} else if(s->done) {
|
||||
task_update(s->task, TASK_DONE, s->task->src, 0, NULL);
|
||||
} else {
|
||||
task_update(s->task, TASK_FAILED, s->task->src, 0, "client disconnected");
|
||||
}
|
||||
free(s);
|
||||
}
|
||||
|
||||
enum MHD_Result
|
||||
api_download(struct MHD_Connection *conn) {
|
||||
char *idstr = query_value(conn, "id");
|
||||
unsigned long id = idstr ? strtoul(idstr, NULL, 10) : 0;
|
||||
char **paths = NULL;
|
||||
size_t count = 0;
|
||||
struct stat st;
|
||||
tar_stream_t *stream;
|
||||
struct MHD_Response *resp;
|
||||
enum MHD_Result ret;
|
||||
file_task_t *task;
|
||||
char download_name[PATH_MAX];
|
||||
|
||||
free(idstr);
|
||||
pthread_mutex_lock(&g_tasks_lock);
|
||||
task = find_task_locked(id);
|
||||
if(!task || task->op != TASK_DOWNLOAD || task->state != TASK_QUEUED) {
|
||||
pthread_mutex_unlock(&g_tasks_lock);
|
||||
return send_json_error(conn, MHD_HTTP_NOT_FOUND, "active task not found");
|
||||
}
|
||||
task->state = TASK_RUNNING;
|
||||
task->updated_at = time(NULL);
|
||||
pthread_mutex_unlock(&g_tasks_lock);
|
||||
|
||||
if(prepare_download_task(task)) {
|
||||
char current[PATH_MAX];
|
||||
snprintf(current, sizeof(current), "%s",
|
||||
task->current[0] ? task->current : task->src);
|
||||
if(errno == ECANCELED || task_cancel_requested(task)) {
|
||||
task_update(task, TASK_CANCELED, current, 0, "canceled");
|
||||
} else {
|
||||
task_update(task, TASK_FAILED, current, 0, strerror(errno));
|
||||
}
|
||||
return send_json_error(conn, MHD_HTTP_INTERNAL_SERVER_ERROR, NULL);
|
||||
}
|
||||
|
||||
paths = task->srcs;
|
||||
count = task->src_count;
|
||||
if(count == 1 && !stat(paths[0], &st) && S_ISREG(st.st_mode)) {
|
||||
download_file_stream_t *file_stream = calloc(1, sizeof(*file_stream));
|
||||
if(!file_stream) {
|
||||
task_update(task, TASK_FAILED, task->src, 0, "out of memory");
|
||||
return send_json_error(conn, MHD_HTTP_INTERNAL_SERVER_ERROR, "out of memory");
|
||||
}
|
||||
file_stream->task = task;
|
||||
file_stream->fd = -1;
|
||||
file_stream->size = (unsigned long long)st.st_size;
|
||||
file_stream->done = file_stream->size == 0;
|
||||
file_stream->fd = open(paths[0], O_RDONLY);
|
||||
if(file_stream->fd < 0) {
|
||||
free(file_stream);
|
||||
task_update(task, TASK_FAILED, task->src, 0, strerror(errno));
|
||||
return send_json_error(conn, MHD_HTTP_NOT_FOUND, "file not found");
|
||||
}
|
||||
resp = MHD_create_response_from_callback((uint64_t)st.st_size,
|
||||
DOWNLOAD_BUFFER_SIZE,
|
||||
download_file_read, file_stream,
|
||||
download_file_close);
|
||||
if(!resp) {
|
||||
download_file_close(file_stream);
|
||||
return MHD_NO;
|
||||
}
|
||||
MHD_add_response_header(resp, MHD_HTTP_HEADER_CONTENT_TYPE,
|
||||
"application/octet-stream");
|
||||
download_file_name(download_name, sizeof(download_name), paths[0]);
|
||||
add_download_filename_header(resp, download_name);
|
||||
ret = websrv_queue_response(conn, MHD_HTTP_OK, resp);
|
||||
MHD_destroy_response(resp);
|
||||
return ret;
|
||||
}
|
||||
if(!(stream = calloc(1, sizeof(*stream)))) {
|
||||
task_update(task, TASK_FAILED, task->src, 0, "out of memory");
|
||||
return send_json_error(conn, MHD_HTTP_INTERNAL_SERVER_ERROR, "out of memory");
|
||||
}
|
||||
stream->fd = -1;
|
||||
stream->task = task;
|
||||
stream->paths = NULL;
|
||||
stream->path_count = count;
|
||||
stream->paths = calloc(count, sizeof(char *));
|
||||
if(!stream->paths) {
|
||||
tar_close(stream);
|
||||
task_update(task, TASK_FAILED, task->src, 0, "out of memory");
|
||||
return send_json_error(conn, MHD_HTTP_INTERNAL_SERVER_ERROR, "out of memory");
|
||||
}
|
||||
for(size_t i = 0; i < count; i++) {
|
||||
stream->paths[i] = strdup(paths[i]);
|
||||
if(!stream->paths[i]) {
|
||||
stream->path_count = i;
|
||||
tar_close(stream);
|
||||
task_update(task, TASK_FAILED, task->src, 0, "out of memory");
|
||||
return send_json_error(conn, MHD_HTTP_INTERNAL_SERVER_ERROR, "out of memory");
|
||||
}
|
||||
}
|
||||
resp = MHD_create_response_from_callback(MHD_SIZE_UNKNOWN,
|
||||
DOWNLOAD_BUFFER_SIZE,
|
||||
tar_read, stream, tar_close);
|
||||
if(!resp) {
|
||||
tar_close(stream);
|
||||
return MHD_NO;
|
||||
}
|
||||
MHD_add_response_header(resp, MHD_HTTP_HEADER_CONTENT_TYPE,
|
||||
"application/x-tar");
|
||||
download_archive_name(download_name, sizeof(download_name), paths, count);
|
||||
add_download_filename_header(resp, download_name);
|
||||
ret = websrv_queue_response(conn, MHD_HTTP_OK, resp);
|
||||
MHD_destroy_response(resp);
|
||||
return ret;
|
||||
}
|
||||
@@ -0,0 +1,426 @@
|
||||
#include "filemgr.h"
|
||||
|
||||
#include <ctype.h>
|
||||
#include <errno.h>
|
||||
#include <pthread.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <sys/stat.h>
|
||||
#include <time.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#include "extract.h"
|
||||
#include "filemgr_internal.h"
|
||||
#include "json_util.h"
|
||||
#include "path_util.h"
|
||||
#include "rar_extract.h"
|
||||
#include "sevenz_extract.h"
|
||||
#include "zip_extract.h"
|
||||
#include "zipx_volume.h"
|
||||
|
||||
/* Cancellation callback: stop when the task is asked to cancel. */
|
||||
static int
|
||||
extract_cancel(void *userdata) {
|
||||
file_task_t *task = userdata;
|
||||
|
||||
return task_cancel_requested(task);
|
||||
}
|
||||
|
||||
/* Case-insensitive suffix check. Returns 1 if path ends in suffix
|
||||
(the comparison ignores trailing slashes, so "x.rar/" is still .rar). */
|
||||
static int
|
||||
ends_with_ci(const char *path, const char *suffix) {
|
||||
size_t path_len = strlen(path);
|
||||
size_t suf_len = strlen(suffix);
|
||||
|
||||
if(path_len < suf_len) {
|
||||
return 0;
|
||||
}
|
||||
/* Trim trailing path separators (defensive — the API rejects them but
|
||||
we get here with whatever path the caller passed). */
|
||||
while(path_len && path[path_len - 1] == '/') {
|
||||
path_len--;
|
||||
}
|
||||
if(path_len < suf_len) {
|
||||
return 0;
|
||||
}
|
||||
return !strcasecmp(path + path_len - suf_len, suffix);
|
||||
}
|
||||
|
||||
/* Progress callback. The engine already throttles reports (200 ms / 1 MiB),
|
||||
so we can forward each report straight into the shared task state.
|
||||
Defined before extract_dispatch() so the dispatcher's call site compiles
|
||||
cleanly under -Werror=implicit-function-declaration. */
|
||||
static void
|
||||
extract_progress(void *userdata, const zipx_progress_t *p) {
|
||||
file_task_t *task = userdata;
|
||||
unsigned long long prev_done;
|
||||
unsigned long long delta;
|
||||
|
||||
pthread_mutex_lock(&g_tasks_lock);
|
||||
task->entries_total = p->entries_total;
|
||||
task->entries_done = p->entries_done;
|
||||
task->total = p->bytes_total;
|
||||
prev_done = task->done;
|
||||
pthread_mutex_unlock(&g_tasks_lock);
|
||||
|
||||
delta = p->bytes_done > prev_done ? p->bytes_done - prev_done : 0;
|
||||
task_update(task, TASK_RUNNING, p->current ? p->current : task->src,
|
||||
delta, NULL);
|
||||
}
|
||||
|
||||
/* Case-insensitive substring search (strcasestr is not available on MinGW). */
|
||||
static const char *
|
||||
ci_strstr(const char *hay, const char *needle) {
|
||||
size_t nlen = strlen(needle);
|
||||
const char *p;
|
||||
|
||||
if(!nlen) {
|
||||
return hay;
|
||||
}
|
||||
for(p = hay; *p; p++) {
|
||||
size_t i;
|
||||
|
||||
for(i = 0; i < nlen; i++) {
|
||||
if(!p[i] ||
|
||||
tolower((unsigned char)p[i]) != tolower((unsigned char)needle[i])) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
if(i == nlen) {
|
||||
return p;
|
||||
}
|
||||
}
|
||||
return NULL;
|
||||
}
|
||||
|
||||
static void
|
||||
extract_set_detail(zipx_result_t *result, const char *text) {
|
||||
size_t len = strlen(text);
|
||||
|
||||
if(len > sizeof(result->detail) - 1) {
|
||||
len = sizeof(result->detail) - 1;
|
||||
}
|
||||
memcpy(result->detail, text, len);
|
||||
result->detail[len] = 0;
|
||||
}
|
||||
|
||||
/* Which engine a volume set belongs to, decided from the member names:
|
||||
0 zip, 1 rar, 2 7z, -1 unknown. */
|
||||
static int
|
||||
volume_format(const zipx_volume_t *vol) {
|
||||
static const char *const exts[] = { ".zip", ".rar", ".7z", NULL };
|
||||
const char *best = NULL;
|
||||
int best_kind = -1;
|
||||
int i;
|
||||
int j;
|
||||
|
||||
for(i = 0; i < vol->count; i++) {
|
||||
for(j = 0; exts[j]; j++) {
|
||||
const char *hit = ci_strstr(vol->paths[i], exts[j]);
|
||||
|
||||
if(hit && (!best || hit > best)) {
|
||||
best = hit;
|
||||
best_kind = j;
|
||||
}
|
||||
}
|
||||
}
|
||||
return best_kind;
|
||||
}
|
||||
|
||||
/* Removes the source archive once a task is done with it. For a split set
|
||||
every volume has to go: leaving the other parts behind would leave the user
|
||||
with something that still looks like a usable archive. */
|
||||
static void
|
||||
remove_source_archives(const char *path) {
|
||||
zipx_volume_t vol;
|
||||
char *err = NULL;
|
||||
int rc = zipx_volume_detect(path, &vol, &err);
|
||||
int i;
|
||||
|
||||
free(err);
|
||||
if(rc > 0) {
|
||||
for(i = 0; i < vol.count; i++) {
|
||||
unlink(vol.paths[i]);
|
||||
}
|
||||
zipx_volume_free(&vol);
|
||||
return;
|
||||
}
|
||||
unlink(path);
|
||||
}
|
||||
|
||||
/* Pick the right engine by the archive file name. Returns ZIPX_ERR_FORMAT
|
||||
for anything that does not look like a supported archive. */
|
||||
static zipx_status_t
|
||||
extract_dispatch(file_task_t *task, zipx_conflict_t conflict,
|
||||
const zipx_limits_t *limits, zipx_result_t *result) {
|
||||
zipx_volume_t vol;
|
||||
char *vol_err = NULL;
|
||||
int vrc = zipx_volume_detect(task->src, &vol, &vol_err);
|
||||
int kind = vrc > 0 ? volume_format(&vol) : -1;
|
||||
|
||||
if(vrc < 0) {
|
||||
/* A broken set gets the precise reason (which volume is missing, ...)
|
||||
instead of a generic "unsupported format". */
|
||||
extract_set_detail(result, task->src);
|
||||
snprintf(result->message, sizeof(result->message), "%s",
|
||||
vol_err ? vol_err : "the archive volumes are incomplete");
|
||||
free(vol_err);
|
||||
return ZIPX_ERR_OPEN;
|
||||
}
|
||||
free(vol_err);
|
||||
if(vrc > 0) {
|
||||
zipx_status_t status;
|
||||
|
||||
if(kind == 0) {
|
||||
status = zipx_extract(task->src, task->dst, conflict, limits,
|
||||
extract_cancel, extract_progress, task,
|
||||
task->extract_password[0] ? task->extract_password
|
||||
: NULL,
|
||||
result);
|
||||
} else if(kind == 1) {
|
||||
/* unrar chains its own volume naming (x.part1.rar); a byte contiguous
|
||||
set named x.rar.001 cannot be handed to it as-is. */
|
||||
extract_set_detail(result, task->src);
|
||||
snprintf(result->message, sizeof(result->message),
|
||||
"RAR volume sets named 'x.rar.001' are not supported yet "
|
||||
"(rename the parts to 'x.part1.rar', 'x.part2.rar', ...)");
|
||||
status = ZIPX_ERR_UNSUPPORTED;
|
||||
} else if(kind == 2) {
|
||||
status = sevenz_extract(task->src, task->dst, conflict, limits,
|
||||
extract_cancel, extract_progress, task,
|
||||
task->extract_password[0] ? task->extract_password
|
||||
: NULL,
|
||||
result);
|
||||
} else {
|
||||
extract_set_detail(result, task->src);
|
||||
snprintf(result->message, sizeof(result->message),
|
||||
"unsupported split archive (only .zip, .rar and .7z volumes "
|
||||
"are recognised)");
|
||||
status = ZIPX_ERR_UNSUPPORTED;
|
||||
}
|
||||
zipx_volume_free(&vol);
|
||||
return status;
|
||||
}
|
||||
if(ends_with_ci(task->src, ".zip")) {
|
||||
return zipx_extract(task->src, task->dst, conflict, limits,
|
||||
extract_cancel, extract_progress, task,
|
||||
task->extract_password[0] ? task->extract_password
|
||||
: NULL,
|
||||
result);
|
||||
}
|
||||
if(ends_with_ci(task->src, ".rar")) {
|
||||
return rar_extract(task->src, task->dst, conflict, limits,
|
||||
extract_cancel, extract_progress, task,
|
||||
task->extract_password[0] ? task->extract_password
|
||||
: NULL,
|
||||
result);
|
||||
}
|
||||
if(ends_with_ci(task->src, ".7z")) {
|
||||
return sevenz_extract(task->src, task->dst, conflict, limits,
|
||||
extract_cancel, extract_progress, task,
|
||||
task->extract_password[0] ? task->extract_password
|
||||
: NULL,
|
||||
result);
|
||||
}
|
||||
extract_set_detail(result, task->src);
|
||||
snprintf(result->message, sizeof(result->message),
|
||||
"unsupported archive format (only .zip, .rar and .7z are accepted)");
|
||||
return ZIPX_ERR_UNSUPPORTED;
|
||||
}
|
||||
|
||||
static const char *
|
||||
extract_error_code(zipx_status_t status) {
|
||||
switch(status) {
|
||||
case ZIPX_ERR_OPEN: return "extract_open_failed";
|
||||
case ZIPX_ERR_FORMAT: return "extract_corrupt";
|
||||
case ZIPX_ERR_UNSUPPORTED: return "extract_unsupported";
|
||||
case ZIPX_ERR_UNSAFE_NAME: return "extract_unsafe_name";
|
||||
case ZIPX_ERR_SPECIAL: return "extract_special_entry";
|
||||
case ZIPX_ERR_DUPLICATE: return "extract_duplicate";
|
||||
case ZIPX_ERR_LIMIT_ENTRIES: return "extract_too_many_entries";
|
||||
case ZIPX_ERR_LIMIT_FILE: return "extract_entry_too_large";
|
||||
case ZIPX_ERR_LIMIT_TOTAL: return "extract_too_large";
|
||||
case ZIPX_ERR_LIMIT_RATIO: return "extract_ratio";
|
||||
case ZIPX_ERR_LIMIT_DEPTH: return "extract_too_deep";
|
||||
case ZIPX_ERR_LIMIT_NAME: return "extract_name_too_long";
|
||||
case ZIPX_ERR_LIMIT_DICT: return "extract_dict_too_large";
|
||||
case ZIPX_ERR_CONFLICT: return "extract_conflict";
|
||||
case ZIPX_ERR_SPACE: return "no_space";
|
||||
case ZIPX_ERR_IO: return "extract_io";
|
||||
case ZIPX_ERR_CRC: return "extract_crc";
|
||||
case ZIPX_ERR_PASSWORD: return "extract_password";
|
||||
default: return "extract_failed";
|
||||
}
|
||||
}
|
||||
|
||||
static void
|
||||
extract_set_error(file_task_t *task, zipx_status_t status,
|
||||
const zipx_result_t *result) {
|
||||
const char *code = extract_error_code(status);
|
||||
const char *detail = result->detail[0] ? result->detail : NULL;
|
||||
const char *msg = result->message[0] ? result->message :
|
||||
zipx_status_string(status);
|
||||
|
||||
pthread_mutex_lock(&g_tasks_lock);
|
||||
snprintf(task->error_code, sizeof(task->error_code), "%s", code);
|
||||
if(detail) {
|
||||
size_t n = strlen(detail);
|
||||
if(n >= sizeof(task->error_arg)) {
|
||||
n = sizeof(task->error_arg) - 1;
|
||||
}
|
||||
memcpy(task->error_arg, detail, n);
|
||||
task->error_arg[n] = 0;
|
||||
}
|
||||
task->updated_at = time(NULL);
|
||||
pthread_mutex_unlock(&g_tasks_lock);
|
||||
|
||||
task_update(task, TASK_FAILED, detail ? detail : task->src, 0, msg);
|
||||
}
|
||||
|
||||
static void *
|
||||
extract_worker(void *arg) {
|
||||
file_task_t *task = arg;
|
||||
zipx_result_t result = {0};
|
||||
zipx_conflict_t conflict;
|
||||
zipx_status_t status;
|
||||
|
||||
switch(task->extract_conflict) {
|
||||
case EXTRACT_CONFLICT_OVERWRITE:
|
||||
conflict = ZIPX_CONFLICT_OVERWRITE;
|
||||
break;
|
||||
case EXTRACT_CONFLICT_MERGE:
|
||||
conflict = ZIPX_CONFLICT_MERGE;
|
||||
break;
|
||||
default:
|
||||
conflict = ZIPX_CONFLICT_FAIL;
|
||||
break;
|
||||
}
|
||||
|
||||
task_update(task, TASK_RUNNING, "scanning archive", 0, NULL);
|
||||
|
||||
status = extract_dispatch(task, conflict,
|
||||
zipx_limits_profile(task->extract_large),
|
||||
&result);
|
||||
|
||||
if(status == ZIPX_OK) {
|
||||
time_t completed_at = time(NULL);
|
||||
|
||||
/* Only delete the source archive when this task owns it (upload flow). */
|
||||
if(task->extract_remove_source && task->src[0]) {
|
||||
remove_source_archives(task->src);
|
||||
}
|
||||
pthread_mutex_lock(&g_tasks_lock);
|
||||
task->state = TASK_DONE;
|
||||
if(task->total) {
|
||||
task->done = task->total;
|
||||
}
|
||||
task->entries_done = task->entries_total;
|
||||
task->updated_at = completed_at;
|
||||
record_task_completion_locked(task, completed_at);
|
||||
pthread_mutex_unlock(&g_tasks_lock);
|
||||
} else if(status == ZIPX_ERR_CANCELED) {
|
||||
task_update(task, TASK_CANCELED,
|
||||
task->current[0] ? task->current : task->src, 0, "canceled");
|
||||
} else {
|
||||
extract_set_error(task, status, &result);
|
||||
}
|
||||
|
||||
return NULL;
|
||||
}
|
||||
|
||||
enum MHD_Result
|
||||
api_extract(struct MHD_Connection *conn, const char *body, size_t body_size) {
|
||||
char *path = fs_path_value(body_form_value(body, body_size, "path"));
|
||||
char *dst_dir = fs_path_value(body_form_value(body, body_size, "dst_dir"));
|
||||
char *conflict_str = body_form_value(body, body_size, "conflict");
|
||||
char *remove_str = body_form_value(body, body_size, "remove_source");
|
||||
char *large_str = body_form_value(body, body_size, "large");
|
||||
char *password_str = body_form_value(body, body_size, "password");
|
||||
extract_conflict_t conflict = EXTRACT_CONFLICT_FAIL;
|
||||
int remove_source = remove_str && !strcmp(remove_str, "1");
|
||||
int large = large_str && !strcmp(large_str, "1");
|
||||
file_task_t *task;
|
||||
strbuf_t b = {0};
|
||||
struct stat st;
|
||||
|
||||
if(!path || !dst_dir || !path[0] || !dst_dir[0]) {
|
||||
free(path); free(dst_dir); free(conflict_str); free(remove_str);
|
||||
free(large_str); free(password_str);
|
||||
return send_json_error(conn, MHD_HTTP_BAD_REQUEST, "invalid path");
|
||||
}
|
||||
if(conflict_str) {
|
||||
if(!strcmp(conflict_str, "overwrite")) {
|
||||
conflict = EXTRACT_CONFLICT_OVERWRITE;
|
||||
} else if(!strcmp(conflict_str, "merge")) {
|
||||
conflict = EXTRACT_CONFLICT_MERGE;
|
||||
} else if(strcmp(conflict_str, "fail")) {
|
||||
free(path); free(dst_dir); free(conflict_str); free(remove_str);
|
||||
free(large_str); free(password_str);
|
||||
return send_json_error(conn, MHD_HTTP_BAD_REQUEST, "invalid conflict");
|
||||
}
|
||||
}
|
||||
if(stat(path, &st) || !S_ISREG(st.st_mode)) {
|
||||
free(path); free(dst_dir); free(conflict_str); free(remove_str);
|
||||
free(large_str); free(password_str);
|
||||
return send_json_error(conn, MHD_HTTP_BAD_REQUEST, "file not found");
|
||||
}
|
||||
if(stat(dst_dir, &st) || !S_ISDIR(st.st_mode)) {
|
||||
free(path); free(dst_dir); free(conflict_str); free(remove_str);
|
||||
free(large_str); free(password_str);
|
||||
return send_json_error(conn, MHD_HTTP_BAD_REQUEST,
|
||||
"destination must be a directory");
|
||||
}
|
||||
|
||||
task = calloc(1, sizeof(*task));
|
||||
if(!task) {
|
||||
free(path); free(dst_dir); free(conflict_str); free(remove_str);
|
||||
free(large_str); free(password_str);
|
||||
return send_json_error(conn, MHD_HTTP_INTERNAL_SERVER_ERROR,
|
||||
"out of memory");
|
||||
}
|
||||
|
||||
task->op = TASK_EXTRACT;
|
||||
task->state = TASK_QUEUED;
|
||||
task->extract_conflict = (int)conflict;
|
||||
task->extract_remove_source = remove_source;
|
||||
task->extract_large = large;
|
||||
/* The size cap (256 bytes, including the NUL) leaves room for a 255-codepoint
|
||||
UTF-8 password without overflowing the field or letting a malicious header
|
||||
run away with it. Anything longer is truncated, which is what a sane user
|
||||
will never hit but matches the storage size of the field. */
|
||||
if(password_str) {
|
||||
snprintf(task->extract_password, sizeof(task->extract_password), "%s",
|
||||
password_str);
|
||||
}
|
||||
snprintf(task->src, sizeof(task->src), "%s", path);
|
||||
snprintf(task->dst, sizeof(task->dst), "%s", dst_dir);
|
||||
task->created_at = time(NULL);
|
||||
task->updated_at = task->created_at;
|
||||
|
||||
pthread_mutex_lock(&g_tasks_lock);
|
||||
remove_finished_tasks_locked();
|
||||
if(has_active_task_locked()) {
|
||||
pthread_mutex_unlock(&g_tasks_lock);
|
||||
free_task(task);
|
||||
free(path); free(dst_dir); free(conflict_str); free(remove_str);
|
||||
free(large_str); free(password_str);
|
||||
return send_json_error(conn, MHD_HTTP_CONFLICT, "another task is running");
|
||||
}
|
||||
task->id = g_next_task_id++;
|
||||
task->next = g_tasks;
|
||||
g_tasks = task;
|
||||
pthread_mutex_unlock(&g_tasks_lock);
|
||||
|
||||
if(pthread_create(&task->thread, NULL, extract_worker, task)) {
|
||||
task_update(task, TASK_FAILED, NULL, 0, "pthread_create failed");
|
||||
} else {
|
||||
pthread_detach(task->thread);
|
||||
}
|
||||
|
||||
free(path); free(dst_dir); free(conflict_str); free(remove_str);
|
||||
free(large_str); free(password_str);
|
||||
strbuf_printf(&b, "{\"ok\":true,\"task_id\":%lu}", task->id);
|
||||
return send_buffer(conn, MHD_HTTP_OK, b.data, "application/json");
|
||||
}
|
||||
@@ -0,0 +1,22 @@
|
||||
#pragma once
|
||||
|
||||
#include <microhttpd.h>
|
||||
|
||||
/* Conflict policy for ZIP extraction, mirrors the zipx_conflict_t values. */
|
||||
typedef enum extract_conflict {
|
||||
EXTRACT_CONFLICT_FAIL = 0,
|
||||
EXTRACT_CONFLICT_OVERWRITE = 1,
|
||||
EXTRACT_CONFLICT_MERGE = 2
|
||||
} extract_conflict_t;
|
||||
|
||||
/* POST /api/extract handler.
|
||||
Accepts a form-encoded body (matching the other filemgr endpoints):
|
||||
path - ZIP path on the device (required)
|
||||
dst_dir - target directory (required)
|
||||
conflict - "fail" (default) | "overwrite" | "merge"
|
||||
remove_source - "1" deletes the source ZIP after a successful extraction
|
||||
(used by the "upload and extract" flow; never set for a
|
||||
pre-existing user archive).
|
||||
Returns {"ok":true,"task_id":N} or a JSON error. */
|
||||
enum MHD_Result api_extract(struct MHD_Connection *conn, const char *body,
|
||||
size_t body_size);
|
||||
@@ -0,0 +1,68 @@
|
||||
#include "filemgr.h"
|
||||
|
||||
#include <stdint.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <sys/stat.h>
|
||||
#include <sys/types.h>
|
||||
|
||||
#include "filemgr_internal.h"
|
||||
#include "mime.h"
|
||||
#include "path_util.h"
|
||||
#include "websrv.h"
|
||||
|
||||
static ssize_t
|
||||
file_read(void *cls, uint64_t pos, char *buf, size_t max) {
|
||||
FILE *file = cls;
|
||||
size_t len;
|
||||
|
||||
if(fseek(file, (long)pos, SEEK_SET)) {
|
||||
return MHD_CONTENT_READER_END_WITH_ERROR;
|
||||
}
|
||||
if(!(len = fread(buf, 1, max, file))) {
|
||||
return ferror(file) ? MHD_CONTENT_READER_END_WITH_ERROR :
|
||||
MHD_CONTENT_READER_END_OF_STREAM;
|
||||
}
|
||||
return (ssize_t)len;
|
||||
}
|
||||
|
||||
static void
|
||||
file_close(void *cls) {
|
||||
fclose((FILE *)cls);
|
||||
}
|
||||
|
||||
enum MHD_Result
|
||||
filemgr_fs_request(struct MHD_Connection *conn) {
|
||||
char *path = fs_path_value(query_value(conn, "path"));
|
||||
struct MHD_Response *resp;
|
||||
enum MHD_Result ret = MHD_NO;
|
||||
struct stat st;
|
||||
FILE *file;
|
||||
|
||||
if(has_active_task()) {
|
||||
return send_json_error(conn, MHD_HTTP_CONFLICT, "another task is running");
|
||||
}
|
||||
|
||||
if(!path || stat(path, &st) || !S_ISREG(st.st_mode) ||
|
||||
!(file = fopen(path, "rb"))) {
|
||||
free(path);
|
||||
return send_json_error(conn, MHD_HTTP_NOT_FOUND, "file not found");
|
||||
}
|
||||
|
||||
if((resp = MHD_create_response_from_callback((uint64_t)st.st_size,
|
||||
32 * 0x4000, file_read, file,
|
||||
file_close))) {
|
||||
const char *mime = mime_get_type(path);
|
||||
if(mime) {
|
||||
MHD_add_response_header(resp, MHD_HTTP_HEADER_CONTENT_TYPE, mime);
|
||||
}
|
||||
ret = websrv_queue_response(conn, MHD_HTTP_OK, resp);
|
||||
MHD_destroy_response(resp);
|
||||
free(path);
|
||||
return ret;
|
||||
}
|
||||
|
||||
fclose(file);
|
||||
free(path);
|
||||
return MHD_NO;
|
||||
}
|
||||
@@ -3,5 +3,12 @@
|
||||
#include <microhttpd.h>
|
||||
|
||||
enum MHD_Result filemgr_api_request(struct MHD_Connection *conn,
|
||||
const char *url);
|
||||
const char *url, const char *method,
|
||||
const char *body, size_t body_size);
|
||||
enum MHD_Result filemgr_fs_request(struct MHD_Connection *conn);
|
||||
|
||||
int filemgr_upload_begin(struct MHD_Connection *conn, void **upload_ctx);
|
||||
int filemgr_upload_data(void *upload_ctx, const char *data, size_t size);
|
||||
enum MHD_Result filemgr_upload_finish(struct MHD_Connection *conn,
|
||||
void *upload_ctx);
|
||||
void filemgr_upload_free(void *upload_ctx);
|
||||
@@ -0,0 +1,140 @@
|
||||
#pragma once
|
||||
|
||||
#include <limits.h>
|
||||
#include <pthread.h>
|
||||
#include <stddef.h>
|
||||
#include <time.h>
|
||||
|
||||
#include <microhttpd.h>
|
||||
|
||||
#define ETA_SAMPLE_SLOTS 64
|
||||
|
||||
typedef enum task_op {
|
||||
TASK_COPY,
|
||||
TASK_MOVE,
|
||||
TASK_DELETE,
|
||||
TASK_CHMOD,
|
||||
TASK_DOWNLOAD,
|
||||
TASK_UPLOAD,
|
||||
TASK_PKG_INSTALL,
|
||||
TASK_EXTRACT,
|
||||
} task_op_t;
|
||||
|
||||
typedef enum task_state {
|
||||
TASK_QUEUED,
|
||||
TASK_RUNNING,
|
||||
TASK_DONE,
|
||||
TASK_FAILED,
|
||||
TASK_CANCELED,
|
||||
} task_state_t;
|
||||
|
||||
typedef struct task_eta_sample {
|
||||
unsigned long long done;
|
||||
struct timespec time;
|
||||
} task_eta_sample_t;
|
||||
|
||||
typedef struct file_task {
|
||||
unsigned long id;
|
||||
task_op_t op;
|
||||
task_state_t state;
|
||||
char src[PATH_MAX];
|
||||
char dst[PATH_MAX];
|
||||
char current[PATH_MAX];
|
||||
char error[160];
|
||||
char error_code[64];
|
||||
char error_arg[PATH_MAX + 96];
|
||||
char **srcs;
|
||||
size_t src_count;
|
||||
size_t file_count;
|
||||
size_t dir_count;
|
||||
size_t upload_completed;
|
||||
unsigned int chmod_mode;
|
||||
int recursive;
|
||||
unsigned long long total;
|
||||
unsigned long long done;
|
||||
unsigned long long speed;
|
||||
unsigned long long eta;
|
||||
unsigned long long entries_total;
|
||||
unsigned long long entries_done;
|
||||
int extract_conflict;
|
||||
int extract_remove_source;
|
||||
int extract_large;
|
||||
/* UTF-8 password for archives that encrypt their streams (7zAES, RAR5 AES).
|
||||
Empty means "try without one"; the engine returns ZIPX_ERR_PASSWORD for
|
||||
an archive that needs one, and the web UI prompts and retries. The
|
||||
length is bounded so a runaway header field cannot overflow task memory. */
|
||||
char extract_password[256];
|
||||
unsigned long long speed_sample_done;
|
||||
struct timespec speed_sample_time;
|
||||
task_eta_sample_t eta_samples[ETA_SAMPLE_SLOTS];
|
||||
unsigned int eta_sample_next;
|
||||
unsigned int eta_sample_count;
|
||||
int cancel_requested;
|
||||
int reported; /* Terminal state has been included in /api/tasks. */
|
||||
unsigned int active_streams;
|
||||
time_t created_at;
|
||||
time_t transfer_started_at;
|
||||
time_t updated_at;
|
||||
pthread_t thread;
|
||||
struct file_task *next;
|
||||
} file_task_t;
|
||||
|
||||
extern pthread_mutex_t g_tasks_lock;
|
||||
extern file_task_t *g_tasks;
|
||||
extern unsigned long g_next_task_id;
|
||||
|
||||
const char *task_op_name(task_op_t op);
|
||||
const char *task_state_name(task_state_t state);
|
||||
int task_is_active(const file_task_t *task);
|
||||
int has_active_task_locked(void);
|
||||
int has_active_task(void);
|
||||
void free_task(file_task_t *task);
|
||||
void remove_finished_tasks_locked(void);
|
||||
int task_cancel_requested(file_task_t *task);
|
||||
file_task_t *find_task_locked(unsigned long id);
|
||||
void task_update(file_task_t *task, task_state_t state, const char *current,
|
||||
unsigned long long add_done, const char *error);
|
||||
void record_task_completion_locked(file_task_t *task, time_t completed_at);
|
||||
|
||||
enum MHD_Result send_json_ok(struct MHD_Connection *conn);
|
||||
enum MHD_Result send_json_error(struct MHD_Connection *conn,
|
||||
unsigned int status, const char *msg);
|
||||
enum MHD_Result send_json_error_detail(struct MHD_Connection *conn,
|
||||
unsigned int status, const char *msg,
|
||||
const char *code, const char *arg);
|
||||
enum MHD_Result send_buffer(struct MHD_Connection *conn, unsigned int status,
|
||||
char *data, const char *mime);
|
||||
|
||||
int ensure_parent_dirs(const char *base, const char *rel);
|
||||
int chmod_path_mode(const char *path, unsigned int mode);
|
||||
int chmod_path_0777(const char *path);
|
||||
int fchmod_0777(int fd);
|
||||
int ignore_chmod_error(int err);
|
||||
int mode_access(const char *path, int mode);
|
||||
int check_target_writable(const char *target, char ***checked_dirs,
|
||||
size_t *checked_dir_count,
|
||||
char *error, size_t error_size,
|
||||
char *code, size_t code_size,
|
||||
char *arg, size_t arg_size);
|
||||
int check_target_space(const char *target, unsigned long long required,
|
||||
char *error, size_t error_size,
|
||||
char *code, size_t code_size,
|
||||
char *arg, size_t arg_size);
|
||||
int target_available_space(const char *target, unsigned long long *available);
|
||||
int count_task_path_bytes(file_task_t *task, const char *path,
|
||||
const char *display, unsigned long long *total,
|
||||
size_t *file_count, size_t *dir_count);
|
||||
|
||||
enum MHD_Result api_upload_prepare(struct MHD_Connection *conn,
|
||||
const char *body, size_t body_size);
|
||||
enum MHD_Result api_upload_finish(struct MHD_Connection *conn);
|
||||
enum MHD_Result api_download_prepare(struct MHD_Connection *conn,
|
||||
const char *body, size_t body_size);
|
||||
enum MHD_Result api_download(struct MHD_Connection *conn);
|
||||
enum MHD_Result api_list(struct MHD_Connection *conn);
|
||||
enum MHD_Result api_space(struct MHD_Connection *conn);
|
||||
enum MHD_Result api_version(struct MHD_Connection *conn);
|
||||
enum MHD_Result api_text(struct MHD_Connection *conn);
|
||||
enum MHD_Result api_text_create(struct MHD_Connection *conn);
|
||||
enum MHD_Result api_text_save(struct MHD_Connection *conn, const char *body,
|
||||
size_t body_size);
|
||||
@@ -0,0 +1,371 @@
|
||||
#include "filemgr_internal.h"
|
||||
|
||||
#include <errno.h>
|
||||
#include <fcntl.h>
|
||||
#include <limits.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <sys/stat.h>
|
||||
#include <sys/statvfs.h>
|
||||
#include <sys/types.h>
|
||||
#ifdef __linux__
|
||||
#include <sys/vfs.h>
|
||||
#else
|
||||
#include <sys/mount.h>
|
||||
#endif
|
||||
#include <time.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#include "path_util.h"
|
||||
|
||||
int
|
||||
ignore_chmod_error(int err) {
|
||||
return err == ENOTSUP || err == EPERM || err == EINVAL || err == EROFS;
|
||||
}
|
||||
|
||||
#ifndef __linux__
|
||||
static int
|
||||
fs_type_has_unix_modes(const char *type) {
|
||||
return strcmp(type, "exfat") &&
|
||||
strcmp(type, "exfatfs") &&
|
||||
strcmp(type, "msdosfs") &&
|
||||
strcmp(type, "fat") &&
|
||||
strcmp(type, "vfat");
|
||||
}
|
||||
#endif
|
||||
|
||||
static int
|
||||
path_has_unix_modes(const char *path) {
|
||||
#ifdef __linux__
|
||||
(void)path;
|
||||
return 1;
|
||||
#else
|
||||
struct statfs fs;
|
||||
|
||||
if(statfs(path, &fs)) {
|
||||
return 1;
|
||||
}
|
||||
return fs_type_has_unix_modes(fs.f_fstypename);
|
||||
#endif
|
||||
}
|
||||
|
||||
static int
|
||||
fd_has_unix_modes(int fd) {
|
||||
#ifdef __linux__
|
||||
(void)fd;
|
||||
return 1;
|
||||
#else
|
||||
struct statfs fs;
|
||||
|
||||
if(fstatfs(fd, &fs)) {
|
||||
return 1;
|
||||
}
|
||||
return fs_type_has_unix_modes(fs.f_fstypename);
|
||||
#endif
|
||||
}
|
||||
|
||||
int
|
||||
chmod_path_mode(const char *path, unsigned int mode) {
|
||||
if(!path_has_unix_modes(path)) {
|
||||
return 0;
|
||||
}
|
||||
if(chmod(path, (mode_t)(mode & 0777))) {
|
||||
/* A filesystem that does not implement Unix modes is compatible with the
|
||||
paste behavior. Real permission and read-only errors must reach the UI. */
|
||||
if(errno != ENOTSUP && errno != EINVAL) {
|
||||
return -1;
|
||||
}
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
int
|
||||
chmod_path_0777(const char *path) {
|
||||
if(!path_has_unix_modes(path)) {
|
||||
return 0;
|
||||
}
|
||||
if(chmod(path, 0777) && !ignore_chmod_error(errno)) {
|
||||
return -1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
int
|
||||
fchmod_0777(int fd) {
|
||||
if(!fd_has_unix_modes(fd)) {
|
||||
return 0;
|
||||
}
|
||||
if(fchmod(fd, 0777) && !ignore_chmod_error(errno)) {
|
||||
return -1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int
|
||||
target_statvfs(const char *target, struct statvfs *vfs) {
|
||||
char parent[PATH_MAX];
|
||||
|
||||
if(!statvfs(target, vfs)) {
|
||||
return 0;
|
||||
}
|
||||
if(path_dirname(target, parent, sizeof(parent))) {
|
||||
return -1;
|
||||
}
|
||||
return statvfs(parent, vfs);
|
||||
}
|
||||
|
||||
int
|
||||
target_available_space(const char *target, unsigned long long *available) {
|
||||
struct statvfs vfs;
|
||||
unsigned long long block_size;
|
||||
|
||||
if(target_statvfs(target, &vfs)) {
|
||||
return -1;
|
||||
}
|
||||
block_size = vfs.f_frsize ? vfs.f_frsize : vfs.f_bsize;
|
||||
*available = (unsigned long long)vfs.f_bavail * block_size;
|
||||
return 0;
|
||||
}
|
||||
|
||||
int
|
||||
check_target_space(const char *target, unsigned long long required,
|
||||
char *error, size_t error_size,
|
||||
char *code, size_t code_size,
|
||||
char *arg, size_t arg_size) {
|
||||
unsigned long long available;
|
||||
|
||||
if(!required) {
|
||||
return 0;
|
||||
}
|
||||
if(target_available_space(target, &available)) {
|
||||
snprintf(error, error_size, "cannot read target free space");
|
||||
snprintf(code, code_size, "space_check_failed");
|
||||
snprintf(arg, arg_size, "%s", target);
|
||||
return -1;
|
||||
}
|
||||
if(available < required) {
|
||||
snprintf(error, error_size,
|
||||
"not enough target space, required %llu bytes, available %llu bytes",
|
||||
required, available);
|
||||
snprintf(code, code_size, "no_space");
|
||||
snprintf(arg, arg_size, "%llu,%llu", required, available);
|
||||
errno = ENOSPC;
|
||||
return -1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
static void
|
||||
set_error_detail(char *error, size_t error_size, char *code, size_t code_size,
|
||||
char *arg, size_t arg_size, const char *error_code,
|
||||
const char *message, const char *path) {
|
||||
snprintf(error, error_size, "%s: %s", message, path);
|
||||
snprintf(code, code_size, "%s", error_code);
|
||||
snprintf(arg, arg_size, "%s", path);
|
||||
}
|
||||
|
||||
static void
|
||||
set_permission_error_detail(char *error, size_t error_size,
|
||||
char *code, size_t code_size,
|
||||
char *arg, size_t arg_size,
|
||||
const char *error_code, const char *message,
|
||||
const char *path) {
|
||||
struct stat st;
|
||||
|
||||
set_error_detail(error, error_size, code, code_size, arg, arg_size,
|
||||
error_code, message, path);
|
||||
if(!stat(path, &st)) {
|
||||
snprintf(arg, arg_size, "%s (mode=%04o, uid=%lu, gid=%lu)", path,
|
||||
(unsigned int)(st.st_mode & 07777),
|
||||
(unsigned long)st.st_uid, (unsigned long)st.st_gid);
|
||||
}
|
||||
}
|
||||
|
||||
static int
|
||||
mode_access_stat(const struct stat *st, int mode) {
|
||||
mode_t allowed;
|
||||
uid_t uid = geteuid();
|
||||
|
||||
if(uid == 0) {
|
||||
if(!(mode & X_OK) || !S_ISREG(st->st_mode) ||
|
||||
(st->st_mode & (S_IXUSR | S_IXGRP | S_IXOTH))) {
|
||||
return 0;
|
||||
}
|
||||
} else if(uid == st->st_uid) {
|
||||
allowed = (st->st_mode >> 6) & 7;
|
||||
if((allowed & mode) == (mode_t)mode) {
|
||||
return 0;
|
||||
}
|
||||
} else {
|
||||
gid_t gid = getegid();
|
||||
int group_match = gid == st->st_gid;
|
||||
|
||||
if(!group_match) {
|
||||
int count = getgroups(0, NULL);
|
||||
gid_t *groups = count > 0 ? malloc((size_t)count * sizeof(*groups)) : NULL;
|
||||
|
||||
if(groups && getgroups(count, groups) == count) {
|
||||
int i;
|
||||
for(i = 0; i < count; i++) {
|
||||
if(groups[i] == st->st_gid) {
|
||||
group_match = 1;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
free(groups);
|
||||
}
|
||||
allowed = group_match ? (st->st_mode >> 3) & 7 : st->st_mode & 7;
|
||||
if((allowed & mode) == (mode_t)mode) {
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
errno = EACCES;
|
||||
return -1;
|
||||
}
|
||||
|
||||
int
|
||||
mode_access(const char *path, int mode) {
|
||||
struct stat st;
|
||||
|
||||
return stat(path, &st) ? -1 : mode_access_stat(&st, mode);
|
||||
}
|
||||
|
||||
static int
|
||||
probe_dir_writable(const char *path) {
|
||||
char name[80];
|
||||
char probe[PATH_MAX];
|
||||
int attempt;
|
||||
|
||||
for(attempt = 0; attempt < 16; attempt++) {
|
||||
int fd;
|
||||
int error = 0;
|
||||
|
||||
snprintf(name, sizeof(name), ".web-file-mgr-%ld-%lld-%d.tmp",
|
||||
(long)getpid(), (long long)time(NULL), attempt);
|
||||
if(path_join(probe, sizeof(probe), path, name)) {
|
||||
return -1;
|
||||
}
|
||||
fd = open(probe, O_WRONLY | O_CREAT | O_EXCL, 0600);
|
||||
if(fd < 0) {
|
||||
if(errno == EEXIST) {
|
||||
continue;
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
if(close(fd)) {
|
||||
error = errno;
|
||||
}
|
||||
if(unlink(probe) && !error) {
|
||||
error = errno;
|
||||
}
|
||||
if(error) {
|
||||
errno = error;
|
||||
return -1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
errno = EEXIST;
|
||||
return -1;
|
||||
}
|
||||
|
||||
static int
|
||||
probe_file_writable(const char *path) {
|
||||
int fd = open(path, O_WRONLY);
|
||||
|
||||
return fd < 0 ? -1 : close(fd);
|
||||
}
|
||||
|
||||
static int
|
||||
path_seen(char **paths, size_t count, const char *path) {
|
||||
size_t i;
|
||||
|
||||
for(i = 0; i < count; i++) {
|
||||
if(!strcmp(paths[i], path)) {
|
||||
return 1;
|
||||
}
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int
|
||||
remember_path(char ***paths, size_t *count, const char *path) {
|
||||
char **tmp;
|
||||
|
||||
if(path_seen(*paths, *count, path)) {
|
||||
return 0;
|
||||
}
|
||||
tmp = realloc(*paths, sizeof(char *) * (*count + 1));
|
||||
if(!tmp) {
|
||||
errno = ENOMEM;
|
||||
return -1;
|
||||
}
|
||||
*paths = tmp;
|
||||
(*paths)[*count] = strdup(path);
|
||||
if(!(*paths)[*count]) {
|
||||
errno = ENOMEM;
|
||||
return -1;
|
||||
}
|
||||
(*count)++;
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int
|
||||
probe_dir_once(const char *path, char ***checked, size_t *checked_count) {
|
||||
if(path_seen(*checked, *checked_count, path)) {
|
||||
return 0;
|
||||
}
|
||||
if(probe_dir_writable(path)) {
|
||||
return -1;
|
||||
}
|
||||
return remember_path(checked, checked_count, path);
|
||||
}
|
||||
|
||||
int
|
||||
check_target_writable(const char *target, char ***checked_dirs,
|
||||
size_t *checked_dir_count, char *error, size_t error_size,
|
||||
char *code, size_t code_size, char *arg, size_t arg_size) {
|
||||
struct stat st;
|
||||
char parent[PATH_MAX];
|
||||
|
||||
if(!stat(target, &st)) {
|
||||
if(S_ISDIR(st.st_mode)) {
|
||||
if(probe_dir_once(target, checked_dirs, checked_dir_count)) {
|
||||
set_permission_error_detail(error, error_size, code, code_size,
|
||||
arg, arg_size, "target_dir_not_writable",
|
||||
"target directory is not writable", target);
|
||||
return -1;
|
||||
}
|
||||
} else {
|
||||
if(probe_file_writable(target)) {
|
||||
set_permission_error_detail(error, error_size, code, code_size,
|
||||
arg, arg_size, "target_file_not_writable",
|
||||
"target file is not writable", target);
|
||||
return -1;
|
||||
}
|
||||
goto check_parent;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
if(errno != ENOENT) {
|
||||
set_error_detail(error, error_size, code, code_size, arg, arg_size,
|
||||
"target_check_failed", "cannot check target path", target);
|
||||
return -1;
|
||||
}
|
||||
|
||||
check_parent:
|
||||
if(path_dirname(target, parent, sizeof(parent))) {
|
||||
set_error_detail(error, error_size, code, code_size, arg, arg_size,
|
||||
"target_check_failed", "cannot check target path", target);
|
||||
return -1;
|
||||
}
|
||||
if(probe_dir_once(parent, checked_dirs, checked_dir_count)) {
|
||||
set_permission_error_detail(error, error_size, code, code_size,
|
||||
arg, arg_size, "target_parent_not_writable",
|
||||
"current directory is not writable", parent);
|
||||
return -1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,90 @@
|
||||
#include "json_util.h"
|
||||
|
||||
#include <stdarg.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
|
||||
int
|
||||
strbuf_reserve(strbuf_t *b, size_t extra) {
|
||||
size_t need = b->len + extra + 1;
|
||||
char *tmp;
|
||||
size_t cap;
|
||||
|
||||
if(need <= b->cap) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
cap = b->cap ? b->cap : 4096;
|
||||
while(cap < need) {
|
||||
cap *= 2;
|
||||
}
|
||||
|
||||
if(!(tmp = realloc(b->data, cap))) {
|
||||
return -1;
|
||||
}
|
||||
|
||||
b->data = tmp;
|
||||
b->cap = cap;
|
||||
return 0;
|
||||
}
|
||||
|
||||
int
|
||||
strbuf_append(strbuf_t *b, const char *s) {
|
||||
size_t n = strlen(s);
|
||||
if(strbuf_reserve(b, n)) {
|
||||
return -1;
|
||||
}
|
||||
memcpy(b->data + b->len, s, n + 1);
|
||||
b->len += n;
|
||||
return 0;
|
||||
}
|
||||
|
||||
int
|
||||
strbuf_printf(strbuf_t *b, const char *fmt, ...) {
|
||||
va_list ap;
|
||||
va_list cp;
|
||||
int n;
|
||||
|
||||
va_start(ap, fmt);
|
||||
va_copy(cp, ap);
|
||||
n = vsnprintf(NULL, 0, fmt, cp);
|
||||
va_end(cp);
|
||||
if(n < 0 || strbuf_reserve(b, (size_t)n)) {
|
||||
va_end(ap);
|
||||
return -1;
|
||||
}
|
||||
|
||||
vsnprintf(b->data + b->len, b->cap - b->len, fmt, ap);
|
||||
va_end(ap);
|
||||
b->len += (size_t)n;
|
||||
return 0;
|
||||
}
|
||||
|
||||
int
|
||||
json_escape(strbuf_t *b, const char *s) {
|
||||
if(strbuf_append(b, "\"")) return -1;
|
||||
while(*s) {
|
||||
unsigned char c = (unsigned char)*s;
|
||||
switch(c) {
|
||||
case '"': if(strbuf_append(b, "\\\"")) return -1; s++; break;
|
||||
case '\\': if(strbuf_append(b, "\\\\")) return -1; s++; break;
|
||||
case '\b': if(strbuf_append(b, "\\b")) return -1; s++; break;
|
||||
case '\f': if(strbuf_append(b, "\\f")) return -1; s++; break;
|
||||
case '\n': if(strbuf_append(b, "\\n")) return -1; s++; break;
|
||||
case '\r': if(strbuf_append(b, "\\r")) return -1; s++; break;
|
||||
case '\t': if(strbuf_append(b, "\\t")) return -1; s++; break;
|
||||
default:
|
||||
if(c < 0x20 || c >= 0x80) {
|
||||
if(strbuf_printf(b, "\\u%04x", c)) return -1;
|
||||
} else {
|
||||
if(strbuf_reserve(b, 1)) return -1;
|
||||
b->data[b->len++] = (char)c;
|
||||
b->data[b->len] = 0;
|
||||
}
|
||||
s++;
|
||||
break;
|
||||
}
|
||||
}
|
||||
return strbuf_append(b, "\"");
|
||||
}
|
||||
@@ -0,0 +1,14 @@
|
||||
#pragma once
|
||||
|
||||
#include <stddef.h>
|
||||
|
||||
typedef struct strbuf {
|
||||
char *data;
|
||||
size_t len;
|
||||
size_t cap;
|
||||
} strbuf_t;
|
||||
|
||||
int strbuf_reserve(strbuf_t *b, size_t extra);
|
||||
int strbuf_append(strbuf_t *b, const char *s);
|
||||
int strbuf_printf(strbuf_t *b, const char *fmt, ...);
|
||||
int json_escape(strbuf_t *b, const char *s);
|
||||
@@ -0,0 +1,84 @@
|
||||
#include "filemgr_internal.h"
|
||||
|
||||
#include <dirent.h>
|
||||
#include <limits.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <sys/stat.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#include "json_util.h"
|
||||
#include "path_util.h"
|
||||
|
||||
static char
|
||||
mode_type(const struct stat *st) {
|
||||
if(S_ISDIR(st->st_mode)) return 'd';
|
||||
if(S_ISLNK(st->st_mode)) return 'l';
|
||||
if(S_ISCHR(st->st_mode)) return 'c';
|
||||
if(S_ISBLK(st->st_mode)) return 'b';
|
||||
if(S_ISFIFO(st->st_mode)) return 'p';
|
||||
if(S_ISSOCK(st->st_mode)) return 's';
|
||||
return '-';
|
||||
}
|
||||
|
||||
enum MHD_Result
|
||||
api_list(struct MHD_Connection *conn) {
|
||||
char *path = fs_path_value(query_value(conn, "path"));
|
||||
DIR *dir;
|
||||
struct dirent *entry;
|
||||
struct stat st;
|
||||
strbuf_t b = {0};
|
||||
int first = 1;
|
||||
|
||||
if(has_active_task()) {
|
||||
return send_json_error(conn, MHD_HTTP_CONFLICT, "another task is running");
|
||||
}
|
||||
|
||||
if(!path) {
|
||||
path = strdup("/");
|
||||
}
|
||||
if(!(dir = opendir(path))) {
|
||||
free(path);
|
||||
return send_json_error(conn, MHD_HTTP_NOT_FOUND, NULL);
|
||||
}
|
||||
|
||||
strbuf_append(&b, "{\"ok\":true,\"path\":");
|
||||
json_escape(&b, path);
|
||||
strbuf_append(&b, ",\"parent\":");
|
||||
char parent[PATH_MAX];
|
||||
if(path_dirname(path, parent, sizeof(parent))) {
|
||||
strcpy(parent, "/");
|
||||
}
|
||||
json_escape(&b, parent);
|
||||
strbuf_append(&b, ",\"entries\":[");
|
||||
|
||||
while((entry = readdir(dir))) {
|
||||
char child[PATH_MAX];
|
||||
|
||||
if(!strcmp(entry->d_name, ".") || !strcmp(entry->d_name, "..")) {
|
||||
continue;
|
||||
}
|
||||
if(path_join(child, sizeof(child), path, entry->d_name) ||
|
||||
lstat(child, &st)) {
|
||||
continue;
|
||||
}
|
||||
|
||||
if(!first) {
|
||||
strbuf_append(&b, ",");
|
||||
}
|
||||
first = 0;
|
||||
strbuf_append(&b, "{\"name\":");
|
||||
json_escape(&b, entry->d_name);
|
||||
strbuf_append(&b, ",\"path\":");
|
||||
json_escape(&b, child);
|
||||
strbuf_printf(&b, ",\"type\":\"%c\",\"mode\":%u,\"size\":%lld,\"mtime\":%lld}",
|
||||
mode_type(&st), (unsigned int)(st.st_mode & 07777),
|
||||
(long long)st.st_size, (long long)st.st_mtime);
|
||||
}
|
||||
|
||||
closedir(dir);
|
||||
free(path);
|
||||
strbuf_append(&b, "]}");
|
||||
return send_buffer(conn, MHD_HTTP_OK, b.data, "application/json");
|
||||
}
|
||||
@@ -150,6 +150,9 @@ main(int argc, char **argv) {
|
||||
#endif
|
||||
|
||||
websrv_listen(port);
|
||||
if(websrv_stop_requested()) {
|
||||
break;
|
||||
}
|
||||
sleep(3);
|
||||
}
|
||||
|
||||
|
||||
@@ -8,8 +8,10 @@ typedef struct mime_entry {
|
||||
} mime_entry_t;
|
||||
|
||||
static const mime_entry_t g_mimes[] = {
|
||||
{"bmp", "image/bmp"},
|
||||
{"css", "text/css"},
|
||||
{"elf", "application/octet-stream"},
|
||||
{"gif", "image/gif"},
|
||||
{"html", "text/html"},
|
||||
{"jpeg", "image/jpeg"},
|
||||
{"jpg", "image/jpeg"},
|
||||
@@ -20,6 +22,7 @@ static const mime_entry_t g_mimes[] = {
|
||||
{"png", "image/png"},
|
||||
{"txt", "text/plain"},
|
||||
{"xml", "text/xml"},
|
||||
{"webp", "image/webp"},
|
||||
{"zip", "application/zip"},
|
||||
};
|
||||
|
||||
|
||||
@@ -0,0 +1,401 @@
|
||||
#include "path_util.h"
|
||||
|
||||
#include <ctype.h>
|
||||
#include <errno.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
|
||||
char *
|
||||
query_value(struct MHD_Connection *conn, const char *key) {
|
||||
const char *raw = MHD_lookup_connection_value(conn, MHD_GET_ARGUMENT_KIND, key);
|
||||
char *value = raw ? strdup(raw) : NULL;
|
||||
|
||||
if(value && !strcmp(key, "name")) {
|
||||
char *start = value;
|
||||
char *end;
|
||||
while(isspace((unsigned char)*start)) start++;
|
||||
end = start + strlen(start);
|
||||
while(end > start && isspace((unsigned char)end[-1])) end--;
|
||||
memmove(value, start, (size_t)(end - start));
|
||||
value[end - start] = 0;
|
||||
}
|
||||
return value;
|
||||
}
|
||||
|
||||
char *
|
||||
header_value(struct MHD_Connection *conn, const char *key) {
|
||||
const char *raw = MHD_lookup_connection_value(conn, MHD_HEADER_KIND, key);
|
||||
return raw ? form_decode(raw, strlen(raw)) : NULL;
|
||||
}
|
||||
|
||||
char *
|
||||
request_value(struct MHD_Connection *conn, const char *header,
|
||||
const char *query) {
|
||||
char *value = header_value(conn, header);
|
||||
return value ? value : query_value(conn, query);
|
||||
}
|
||||
|
||||
static int
|
||||
hex_value(char c) {
|
||||
if(c >= '0' && c <= '9') return c - '0';
|
||||
if(c >= 'a' && c <= 'f') return c - 'a' + 10;
|
||||
if(c >= 'A' && c <= 'F') return c - 'A' + 10;
|
||||
return -1;
|
||||
}
|
||||
|
||||
char *
|
||||
form_decode(const char *src, size_t len) {
|
||||
char *out = malloc(len + 1);
|
||||
size_t i;
|
||||
size_t j = 0;
|
||||
|
||||
if(!out) {
|
||||
return NULL;
|
||||
}
|
||||
for(i = 0; i < len; i++) {
|
||||
if(src[i] == '+') {
|
||||
out[j++] = ' ';
|
||||
} else if(src[i] == '%' && i + 2 < len) {
|
||||
int hi = hex_value(src[i + 1]);
|
||||
int lo = hex_value(src[i + 2]);
|
||||
if(hi >= 0 && lo >= 0) {
|
||||
out[j++] = (char)((hi << 4) | lo);
|
||||
i += 2;
|
||||
} else {
|
||||
out[j++] = src[i];
|
||||
}
|
||||
} else {
|
||||
out[j++] = src[i];
|
||||
}
|
||||
}
|
||||
out[j] = 0;
|
||||
return out;
|
||||
}
|
||||
|
||||
char *
|
||||
body_form_value(const char *body, size_t body_size, const char *key) {
|
||||
size_t key_len = strlen(key);
|
||||
size_t pos = 0;
|
||||
|
||||
while(body && pos < body_size) {
|
||||
size_t start = pos;
|
||||
size_t end;
|
||||
size_t eq;
|
||||
|
||||
while(pos < body_size && body[pos] != '&') {
|
||||
pos++;
|
||||
}
|
||||
end = pos;
|
||||
if(pos < body_size && body[pos] == '&') {
|
||||
pos++;
|
||||
}
|
||||
eq = start;
|
||||
while(eq < end && body[eq] != '=') {
|
||||
eq++;
|
||||
}
|
||||
if(eq - start == key_len && !strncmp(body + start, key, key_len)) {
|
||||
return form_decode(body + eq + (eq < end), end - eq - (eq < end));
|
||||
}
|
||||
}
|
||||
return NULL;
|
||||
}
|
||||
|
||||
void
|
||||
free_paths(char **paths, size_t count) {
|
||||
size_t i;
|
||||
|
||||
if(!paths) {
|
||||
return;
|
||||
}
|
||||
for(i = 0; i < count; i++) {
|
||||
free(paths[i]);
|
||||
}
|
||||
free(paths);
|
||||
}
|
||||
|
||||
int
|
||||
parse_paths(const char *raw, char ***out_paths, size_t *out_count) {
|
||||
char *copy;
|
||||
char *line;
|
||||
char *save;
|
||||
char **paths = NULL;
|
||||
size_t count = 0;
|
||||
size_t capacity = 0;
|
||||
const char *p;
|
||||
int in_line = 0;
|
||||
|
||||
*out_paths = NULL;
|
||||
*out_count = 0;
|
||||
|
||||
if(!raw || !raw[0]) {
|
||||
errno = EINVAL;
|
||||
return -1;
|
||||
}
|
||||
for(p = raw; *p; p++) {
|
||||
if(*p == '\n') {
|
||||
if(in_line) {
|
||||
capacity++;
|
||||
in_line = 0;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
in_line = 1;
|
||||
}
|
||||
if(in_line) {
|
||||
capacity++;
|
||||
}
|
||||
if(!capacity) {
|
||||
errno = EINVAL;
|
||||
return -1;
|
||||
}
|
||||
|
||||
if(!(paths = calloc(capacity, sizeof(char *))) ||
|
||||
!(copy = strdup(raw))) {
|
||||
free(paths);
|
||||
errno = ENOMEM;
|
||||
return -1;
|
||||
}
|
||||
|
||||
for(line = strtok_r(copy, "\n", &save); line; line = strtok_r(NULL, "\n", &save)) {
|
||||
if(!line[0]) {
|
||||
continue;
|
||||
}
|
||||
if(!(paths[count] = fs_path_value(strdup(line)))) {
|
||||
free_paths(paths, count);
|
||||
free(copy);
|
||||
errno = ENOMEM;
|
||||
return -1;
|
||||
}
|
||||
count++;
|
||||
}
|
||||
|
||||
free(copy);
|
||||
if(!count) {
|
||||
errno = EINVAL;
|
||||
return -1;
|
||||
}
|
||||
|
||||
*out_paths = paths;
|
||||
*out_count = count;
|
||||
return 0;
|
||||
}
|
||||
|
||||
const char *
|
||||
path_basename(const char *path) {
|
||||
const char *end = path + strlen(path);
|
||||
const char *base;
|
||||
|
||||
while(end > path && end[-1] == '/') {
|
||||
end--;
|
||||
}
|
||||
base = end;
|
||||
while(base > path && base[-1] != '/') {
|
||||
base--;
|
||||
}
|
||||
return base;
|
||||
}
|
||||
|
||||
void
|
||||
path_basename_copy(const char *path, char *out, size_t size) {
|
||||
const char *end = path + strlen(path);
|
||||
const char *base;
|
||||
size_t len;
|
||||
|
||||
while(end > path && end[-1] == '/') {
|
||||
end--;
|
||||
}
|
||||
base = end;
|
||||
while(base > path && base[-1] != '/') {
|
||||
base--;
|
||||
}
|
||||
len = (size_t)(end - base);
|
||||
if(!len) {
|
||||
snprintf(out, size, "root");
|
||||
return;
|
||||
}
|
||||
if(len >= size) {
|
||||
len = size - 1;
|
||||
}
|
||||
memcpy(out, base, len);
|
||||
out[len] = 0;
|
||||
}
|
||||
|
||||
int
|
||||
path_dirname(const char *path, char *out, size_t size) {
|
||||
const char *base = path_basename(path);
|
||||
size_t len = (size_t)(base - path);
|
||||
|
||||
while(len > 1 && path[len - 1] == '/') {
|
||||
len--;
|
||||
}
|
||||
if(!len) {
|
||||
len = 1;
|
||||
}
|
||||
if(len >= size) {
|
||||
return -1;
|
||||
}
|
||||
memcpy(out, path, len);
|
||||
out[len] = 0;
|
||||
return 0;
|
||||
}
|
||||
|
||||
int
|
||||
path_join(char *out, size_t size, const char *dir, const char *name) {
|
||||
int n;
|
||||
if(!dir || !name || !dir[0] || !name[0] || strchr(name, '/')) {
|
||||
errno = EINVAL;
|
||||
return -1;
|
||||
}
|
||||
n = snprintf(out, size, "%s%s%s", dir,
|
||||
(strcmp(dir, "/") && dir[strlen(dir) - 1] != '/') ? "/" : "",
|
||||
name);
|
||||
if(n < 0 || (size_t)n >= size) {
|
||||
errno = ENAMETOOLONG;
|
||||
return -1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
int
|
||||
relative_path_safe(const char *path) {
|
||||
const char *p = path;
|
||||
|
||||
if(!path || !path[0] || path[0] == '/') {
|
||||
errno = EINVAL;
|
||||
return 0;
|
||||
}
|
||||
while(*p) {
|
||||
const char *start = p;
|
||||
size_t len;
|
||||
|
||||
while(*p && *p != '/') {
|
||||
if(*p == '\\') {
|
||||
errno = EINVAL;
|
||||
return 0;
|
||||
}
|
||||
p++;
|
||||
}
|
||||
len = (size_t)(p - start);
|
||||
if(!len || (len == 1 && start[0] == '.') ||
|
||||
(len == 2 && start[0] == '.' && start[1] == '.')) {
|
||||
errno = EINVAL;
|
||||
return 0;
|
||||
}
|
||||
if(*p == '/') {
|
||||
p++;
|
||||
}
|
||||
}
|
||||
return 1;
|
||||
}
|
||||
|
||||
int
|
||||
path_join_relative(char *out, size_t size, const char *dir, const char *rel) {
|
||||
int n;
|
||||
|
||||
if(!relative_path_safe(rel)) {
|
||||
return -1;
|
||||
}
|
||||
n = snprintf(out, size, "%s%s%s", dir,
|
||||
(strcmp(dir, "/") && dir[strlen(dir) - 1] != '/') ? "/" : "",
|
||||
rel);
|
||||
if(n < 0 || (size_t)n >= size) {
|
||||
errno = ENAMETOOLONG;
|
||||
return -1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int
|
||||
utf8_decode_char(const char **ps, unsigned int *out) {
|
||||
const unsigned char *s = (const unsigned char *)*ps;
|
||||
unsigned char c = s[0];
|
||||
unsigned int cp;
|
||||
size_t n;
|
||||
size_t i;
|
||||
|
||||
if(c < 0x80) {
|
||||
*out = c;
|
||||
*ps += 1;
|
||||
return 0;
|
||||
}
|
||||
if((c & 0xe0) == 0xc0) {
|
||||
cp = c & 0x1f;
|
||||
n = 2;
|
||||
} else if((c & 0xf0) == 0xe0) {
|
||||
cp = c & 0x0f;
|
||||
n = 3;
|
||||
} else if((c & 0xf8) == 0xf0) {
|
||||
cp = c & 0x07;
|
||||
n = 4;
|
||||
} else {
|
||||
return -1;
|
||||
}
|
||||
|
||||
for(i = 1; i < n; i++) {
|
||||
unsigned char t = s[i];
|
||||
if(!t || (t & 0xc0) != 0x80) {
|
||||
return -1;
|
||||
}
|
||||
cp = (cp << 6) | (t & 0x3f);
|
||||
}
|
||||
if((n == 2 && cp < 0x80) ||
|
||||
(n == 3 && cp < 0x800) ||
|
||||
(n == 4 && (cp < 0x10000 || cp > 0x10ffff)) ||
|
||||
(cp >= 0xd800 && cp <= 0xdfff)) {
|
||||
return -1;
|
||||
}
|
||||
|
||||
*out = cp;
|
||||
*ps += n;
|
||||
return 0;
|
||||
}
|
||||
|
||||
char *
|
||||
fs_path_value(char *path) {
|
||||
const char *p;
|
||||
char *out;
|
||||
size_t len;
|
||||
size_t pos = 0;
|
||||
int has_byte_token = 0;
|
||||
|
||||
if(!path) {
|
||||
return path;
|
||||
}
|
||||
for(p = path; *p;) {
|
||||
unsigned int cp;
|
||||
const char *next = p;
|
||||
|
||||
if(utf8_decode_char(&next, &cp)) {
|
||||
return path;
|
||||
}
|
||||
if(cp >= 0x80 && cp <= 0xff) {
|
||||
has_byte_token = 1;
|
||||
} else if(cp > 0xff) {
|
||||
return path;
|
||||
}
|
||||
p = next;
|
||||
}
|
||||
if(!has_byte_token) {
|
||||
return path;
|
||||
}
|
||||
|
||||
len = strlen(path);
|
||||
if(!(out = malloc(len + 1))) {
|
||||
return path;
|
||||
}
|
||||
for(p = path; *p;) {
|
||||
unsigned int cp;
|
||||
const char *next = p;
|
||||
|
||||
if(utf8_decode_char(&next, &cp)) {
|
||||
free(out);
|
||||
return path;
|
||||
}
|
||||
out[pos++] = (char)(unsigned char)cp;
|
||||
p = next;
|
||||
}
|
||||
out[pos] = 0;
|
||||
free(path);
|
||||
return out;
|
||||
}
|
||||
@@ -0,0 +1,24 @@
|
||||
#pragma once
|
||||
|
||||
#include <stddef.h>
|
||||
#include <limits.h>
|
||||
|
||||
#include <microhttpd.h>
|
||||
|
||||
char *query_value(struct MHD_Connection *conn, const char *key);
|
||||
char *header_value(struct MHD_Connection *conn, const char *key);
|
||||
char *request_value(struct MHD_Connection *conn, const char *header,
|
||||
const char *query);
|
||||
char *form_decode(const char *src, size_t len);
|
||||
char *body_form_value(const char *body, size_t body_size, const char *key);
|
||||
|
||||
void free_paths(char **paths, size_t count);
|
||||
int parse_paths(const char *raw, char ***out_paths, size_t *out_count);
|
||||
|
||||
const char *path_basename(const char *path);
|
||||
void path_basename_copy(const char *path, char *out, size_t size);
|
||||
int path_dirname(const char *path, char *out, size_t size);
|
||||
int path_join(char *out, size_t size, const char *dir, const char *name);
|
||||
int relative_path_safe(const char *path);
|
||||
int path_join_relative(char *out, size_t size, const char *dir, const char *rel);
|
||||
char *fs_path_value(char *path);
|
||||
@@ -0,0 +1,611 @@
|
||||
/* PKG preview -- the /api/pkg_info and /api/pkg_icon endpoints.
|
||||
*
|
||||
* Independent C99 implementation of the PS5 .pkg container layout (entry table
|
||||
* + PARAM.SFO + param.json + ICON0), written for this project.
|
||||
*
|
||||
* NOT derived from cy33hc/ps5-ezremote-client, which the README credits for the
|
||||
* same feature. That project is GPL-2.0-only -- its sources carry no "or later"
|
||||
* notice -- and therefore cannot be combined with this GPL-3.0 codebase at all.
|
||||
* The two share nothing but the on-disk format facts: the SFO magic
|
||||
* 0x46535000, the 20-byte header / 16-byte entry layout, and the key/value
|
||||
* offset indirection. Those are dictated by the format and no implementation
|
||||
* can avoid them. Everything else differs -- this file is C where that one is
|
||||
* C++, it parses the .pkg entry table and param.json (upstream has no .pkg
|
||||
* parser), and it tokenizes JSON itself (upstream links json-c).
|
||||
*
|
||||
* Provenance record and the line-by-line comparison: docs/REWRITE-FEASIBILITY.md
|
||||
* section 2.2. Revisit that note if this file is ever rewritten or the upstream
|
||||
* licence wording changes.
|
||||
*/
|
||||
|
||||
#include "pkg_info.h"
|
||||
|
||||
#include <errno.h>
|
||||
#include <fcntl.h>
|
||||
#include <stdint.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <sys/stat.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#include "filemgr_internal.h"
|
||||
#include "json_util.h"
|
||||
#include "path_util.h"
|
||||
#include "websrv.h"
|
||||
|
||||
#define PKG_CNT_MAGIC 0x7f434e54u
|
||||
#define PKG_FIH_MAGIC 0x7f464948u
|
||||
#define PKG_LIH_MAGIC 0x7f4c4948u
|
||||
#define PKG_HEADER_SIZE 0xa0
|
||||
#define PKG_ENTRY_SIZE 0x20
|
||||
#define PKG_ENTRY_PARAM_SFO 0x1000u
|
||||
#define PKG_ENTRY_ICON0_PNG 0x1200u
|
||||
#define PKG_ENTRY_ICON0_LOCALIZED_LAST 0x121fu
|
||||
#define PKG_ENTRY_PARAM_JSON 0x2000u
|
||||
#define PKG_ENTRY_FLAG_ENCRYPTED 0x80000000u
|
||||
#define PKG_ENTRY_MAX 65536u
|
||||
#define PKG_PARAM_MAX (4u * 1024u * 1024u)
|
||||
#define PKG_ICON_MAX (32u * 1024u * 1024u)
|
||||
#define SFO_MAGIC 0x46535000u
|
||||
#define SFO_ENTRY_SIZE 0x10
|
||||
#define SFO_ENTRY_MAX 256u
|
||||
#define JSON_TOKEN_MAX 4096u
|
||||
|
||||
typedef enum pkg_param_type {
|
||||
PKG_PARAM_SFO,
|
||||
PKG_PARAM_JSON,
|
||||
} pkg_param_type_t;
|
||||
|
||||
typedef struct pkg_source {
|
||||
int fd;
|
||||
uint64_t size;
|
||||
uint32_t content_type;
|
||||
uint32_t content_flags;
|
||||
char content_id[37];
|
||||
uint64_t param_offset;
|
||||
uint32_t param_size;
|
||||
pkg_param_type_t param_type;
|
||||
uint64_t icon_offset;
|
||||
uint32_t icon_size;
|
||||
} pkg_source_t;
|
||||
|
||||
typedef enum json_token_type {
|
||||
JSON_OBJECT,
|
||||
JSON_ARRAY,
|
||||
JSON_STRING,
|
||||
JSON_PRIMITIVE,
|
||||
} json_token_type_t;
|
||||
|
||||
typedef struct json_token {
|
||||
json_token_type_t type;
|
||||
size_t start;
|
||||
size_t end;
|
||||
int parent;
|
||||
} json_token_t;
|
||||
|
||||
static uint16_t
|
||||
read_le16(const unsigned char *p) {
|
||||
return (uint16_t)p[0] | (uint16_t)p[1] << 8;
|
||||
}
|
||||
|
||||
static uint32_t
|
||||
read_le32(const unsigned char *p) {
|
||||
return (uint32_t)p[0] | (uint32_t)p[1] << 8 |
|
||||
(uint32_t)p[2] << 16 | (uint32_t)p[3] << 24;
|
||||
}
|
||||
|
||||
static uint64_t
|
||||
read_le64(const unsigned char *p) {
|
||||
return (uint64_t)read_le32(p) | (uint64_t)read_le32(p + 4) << 32;
|
||||
}
|
||||
|
||||
static uint32_t
|
||||
read_be32(const unsigned char *p) {
|
||||
return (uint32_t)p[0] << 24 | (uint32_t)p[1] << 16 |
|
||||
(uint32_t)p[2] << 8 | (uint32_t)p[3];
|
||||
}
|
||||
|
||||
static int
|
||||
range_valid(uint64_t offset, uint64_t size, uint64_t file_size) {
|
||||
return offset <= file_size && size <= file_size - offset;
|
||||
}
|
||||
|
||||
static int
|
||||
read_at(int fd, void *buffer, size_t size, uint64_t offset) {
|
||||
unsigned char *p = buffer;
|
||||
size_t done = 0;
|
||||
|
||||
while(done < size) {
|
||||
ssize_t n = pread(fd, p + done, size - done, (off_t)(offset + done));
|
||||
if(n <= 0) {
|
||||
if(n < 0 && errno == EINTR) continue;
|
||||
return -1;
|
||||
}
|
||||
done += (size_t)n;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int
|
||||
png_signature_valid(const unsigned char *data, size_t size) {
|
||||
static const unsigned char signature[] = "\x89PNG\r\n\x1a\n";
|
||||
|
||||
return size >= sizeof(signature) - 1 &&
|
||||
!memcmp(data, signature, sizeof(signature) - 1);
|
||||
}
|
||||
|
||||
static int
|
||||
pkg_icon_is_png(const pkg_source_t *pkg, uint64_t offset, uint32_t size) {
|
||||
unsigned char signature[8];
|
||||
|
||||
return size >= sizeof(signature) &&
|
||||
!read_at(pkg->fd, signature, sizeof(signature), offset) &&
|
||||
png_signature_valid(signature, sizeof(signature));
|
||||
}
|
||||
|
||||
static void
|
||||
pkg_source_close(pkg_source_t *pkg) {
|
||||
if(pkg->fd >= 0) close(pkg->fd);
|
||||
pkg->fd = -1;
|
||||
}
|
||||
|
||||
static int
|
||||
pkg_source_open(const char *path, pkg_source_t *pkg) {
|
||||
unsigned char header[PKG_HEADER_SIZE];
|
||||
unsigned char *table = NULL;
|
||||
struct stat st;
|
||||
uint64_t container_offset = 0;
|
||||
uint64_t container_size;
|
||||
uint32_t magic;
|
||||
uint32_t entry_count;
|
||||
uint32_t table_offset;
|
||||
size_t table_size;
|
||||
int ret = -1;
|
||||
|
||||
memset(pkg, 0, sizeof(*pkg));
|
||||
pkg->fd = -1;
|
||||
if((pkg->fd = open(path, O_RDONLY)) < 0 || fstat(pkg->fd, &st) ||
|
||||
!S_ISREG(st.st_mode) || st.st_size < PKG_HEADER_SIZE ||
|
||||
read_at(pkg->fd, header, sizeof(header), 0)) goto done;
|
||||
|
||||
pkg->size = (uint64_t)st.st_size;
|
||||
magic = read_be32(header);
|
||||
if(magic == PKG_FIH_MAGIC) {
|
||||
container_offset = read_le64(header + 0x58);
|
||||
} else if(magic == PKG_LIH_MAGIC) {
|
||||
container_offset = read_le64(header + 0x30);
|
||||
} else if(magic != PKG_CNT_MAGIC) {
|
||||
goto done;
|
||||
}
|
||||
if(container_offset) {
|
||||
if(!range_valid(container_offset, sizeof(header), pkg->size) ||
|
||||
read_at(pkg->fd, header, sizeof(header), container_offset) ||
|
||||
read_be32(header) != PKG_CNT_MAGIC) goto done;
|
||||
}
|
||||
container_size = pkg->size - container_offset;
|
||||
|
||||
memcpy(pkg->content_id, header + 0x40, 36);
|
||||
pkg->content_id[36] = 0;
|
||||
pkg->content_type = read_be32(header + 0x74);
|
||||
pkg->content_flags = read_be32(header + 0x78);
|
||||
entry_count = read_be32(header + 0x10);
|
||||
table_offset = read_be32(header + 0x18);
|
||||
if(!entry_count || entry_count > PKG_ENTRY_MAX) goto done;
|
||||
table_size = (size_t)entry_count * PKG_ENTRY_SIZE;
|
||||
if(!range_valid(table_offset, table_size, container_size) ||
|
||||
!(table = malloc(table_size)) ||
|
||||
read_at(pkg->fd, table, table_size, container_offset + table_offset)) {
|
||||
goto done;
|
||||
}
|
||||
|
||||
for(uint32_t i = 0; i < entry_count; i++) {
|
||||
const unsigned char *entry = table + (size_t)i * PKG_ENTRY_SIZE;
|
||||
uint32_t id = read_be32(entry);
|
||||
uint32_t flags1 = read_be32(entry + 0x08);
|
||||
uint32_t offset = read_be32(entry + 0x10);
|
||||
uint32_t size = read_be32(entry + 0x14);
|
||||
int encrypted = (flags1 & PKG_ENTRY_FLAG_ENCRYPTED) != 0;
|
||||
|
||||
if(!range_valid(offset, size, container_size)) goto done;
|
||||
if(encrypted || !size) continue;
|
||||
if(id == PKG_ENTRY_PARAM_SFO && size <= PKG_PARAM_MAX) {
|
||||
pkg->param_offset = container_offset + offset;
|
||||
pkg->param_size = size;
|
||||
pkg->param_type = PKG_PARAM_SFO;
|
||||
} else if(id == PKG_ENTRY_PARAM_JSON && size <= PKG_PARAM_MAX) {
|
||||
pkg->param_offset = container_offset + offset;
|
||||
pkg->param_size = size;
|
||||
pkg->param_type = PKG_PARAM_JSON;
|
||||
} else if(id >= PKG_ENTRY_ICON0_PNG &&
|
||||
id <= PKG_ENTRY_ICON0_LOCALIZED_LAST &&
|
||||
size <= PKG_ICON_MAX &&
|
||||
(id == PKG_ENTRY_ICON0_PNG || !pkg->icon_size) &&
|
||||
pkg_icon_is_png(pkg, container_offset + offset, size)) {
|
||||
/* Prefer icon0.png; otherwise keep the first valid localized icon. */
|
||||
pkg->icon_offset = container_offset + offset;
|
||||
pkg->icon_size = size;
|
||||
}
|
||||
}
|
||||
if(!pkg->param_size) goto done;
|
||||
ret = 0;
|
||||
|
||||
done:
|
||||
free(table);
|
||||
if(ret) {
|
||||
errno = EINVAL;
|
||||
pkg_source_close(pkg);
|
||||
}
|
||||
return ret;
|
||||
}
|
||||
|
||||
static int
|
||||
append_sfo_fields(strbuf_t *json, const unsigned char *sfo, size_t size) {
|
||||
uint32_t key_offset;
|
||||
uint32_t value_offset;
|
||||
uint32_t count;
|
||||
uint64_t index_end;
|
||||
int first = 1;
|
||||
|
||||
if(size < 20 || read_le32(sfo) != SFO_MAGIC) return -1;
|
||||
key_offset = read_le32(sfo + 8);
|
||||
value_offset = read_le32(sfo + 12);
|
||||
count = read_le32(sfo + 16);
|
||||
index_end = 20 + (uint64_t)count * SFO_ENTRY_SIZE;
|
||||
if(count > SFO_ENTRY_MAX || index_end > size || key_offset < index_end ||
|
||||
value_offset < key_offset || value_offset > size) return -1;
|
||||
|
||||
strbuf_append(json, "[");
|
||||
for(uint32_t i = 0; i < count; i++) {
|
||||
const unsigned char *entry = sfo + 20 + (size_t)i * SFO_ENTRY_SIZE;
|
||||
uint16_t name_offset = read_le16(entry);
|
||||
uint16_t format = read_le16(entry + 2);
|
||||
uint32_t value_size = read_le32(entry + 4);
|
||||
uint32_t value_max = read_le32(entry + 8);
|
||||
uint32_t data_offset = read_le32(entry + 12);
|
||||
uint64_t key_pos = (uint64_t)key_offset + name_offset;
|
||||
uint64_t value_pos = (uint64_t)value_offset + data_offset;
|
||||
const unsigned char *key_end;
|
||||
char number[32];
|
||||
char *value = NULL;
|
||||
|
||||
if(key_pos >= value_offset || value_pos > size ||
|
||||
(value_max && value_size > value_max)) continue;
|
||||
key_end = memchr(sfo + key_pos, 0, value_offset - (size_t)key_pos);
|
||||
if(!key_end) continue;
|
||||
if(format == 0x0204) {
|
||||
size_t length;
|
||||
if(!value_size || !range_valid(value_pos, value_size, size)) continue;
|
||||
length = strnlen((const char *)sfo + value_pos, value_size);
|
||||
if(!(value = malloc(length + 1))) continue;
|
||||
memcpy(value, sfo + value_pos, length);
|
||||
value[length] = 0;
|
||||
} else if(format == 0x0404 && range_valid(value_pos, 4, size)) {
|
||||
snprintf(number, sizeof(number), "%u", read_le32(sfo + value_pos));
|
||||
value = strdup(number);
|
||||
}
|
||||
if(!value) continue;
|
||||
if(!first) strbuf_append(json, ",");
|
||||
first = 0;
|
||||
strbuf_append(json, "{\"name\":");
|
||||
json_escape(json, (const char *)sfo + key_pos);
|
||||
strbuf_append(json, ",\"value\":");
|
||||
json_escape(json, value);
|
||||
strbuf_append(json, "}");
|
||||
free(value);
|
||||
}
|
||||
strbuf_append(json, "]");
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int
|
||||
json_add_token(json_token_t *tokens, size_t *count, json_token_type_t type,
|
||||
size_t start, int parent) {
|
||||
if(*count >= JSON_TOKEN_MAX) return -1;
|
||||
tokens[*count] = (json_token_t){
|
||||
.type = type, .start = start, .end = 0, .parent = parent
|
||||
};
|
||||
return (int)(*count)++;
|
||||
}
|
||||
|
||||
static int
|
||||
parse_json_tokens(const unsigned char *data, size_t size, json_token_t *tokens,
|
||||
size_t *token_count) {
|
||||
int parent = -1;
|
||||
size_t count = 0;
|
||||
|
||||
for(size_t i = 0; i < size; i++) {
|
||||
unsigned char c = data[i];
|
||||
if(c == '{' || c == '[') {
|
||||
int index = json_add_token(tokens, &count,
|
||||
c == '{' ? JSON_OBJECT : JSON_ARRAY,
|
||||
i, parent);
|
||||
if(index < 0) return -1;
|
||||
parent = index;
|
||||
} else if(c == '}' || c == ']') {
|
||||
json_token_type_t type = c == '}' ? JSON_OBJECT : JSON_ARRAY;
|
||||
if(parent < 0 || tokens[parent].type != type) return -1;
|
||||
tokens[parent].end = i + 1;
|
||||
parent = tokens[parent].parent;
|
||||
} else if(c == '"') {
|
||||
int index = json_add_token(tokens, &count, JSON_STRING, i + 1, parent);
|
||||
if(index < 0) return -1;
|
||||
for(i++; i < size && data[i] != '"'; i++) {
|
||||
if(data[i] < 0x20) return -1;
|
||||
if(data[i] == '\\') {
|
||||
if(++i >= size || !strchr("\"\\/bfnrtu", data[i])) return -1;
|
||||
if(data[i] == 'u') {
|
||||
for(unsigned int n = 0; n < 4; n++) {
|
||||
if(++i >= size || !((data[i] >= '0' && data[i] <= '9') ||
|
||||
(data[i] >= 'a' && data[i] <= 'f') ||
|
||||
(data[i] >= 'A' && data[i] <= 'F'))) return -1;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
if(i >= size) return -1;
|
||||
tokens[index].end = i;
|
||||
} else if(c == ':' || c == ',' || c == ' ' || c == '\t' ||
|
||||
c == '\r' || c == '\n') {
|
||||
continue;
|
||||
} else {
|
||||
int index = json_add_token(tokens, &count, JSON_PRIMITIVE, i, parent);
|
||||
if(index < 0) return -1;
|
||||
while(i < size && data[i] != ',' && data[i] != ']' && data[i] != '}' &&
|
||||
data[i] != ' ' && data[i] != '\t' && data[i] != '\r' &&
|
||||
data[i] != '\n') i++;
|
||||
if(tokens[index].start == i) return -1;
|
||||
tokens[index].end = i;
|
||||
i--;
|
||||
}
|
||||
}
|
||||
if(parent >= 0 || !count || tokens[0].type != JSON_OBJECT ||
|
||||
!tokens[0].end) return -1;
|
||||
*token_count = count;
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int
|
||||
json_token_equals(const unsigned char *data, const json_token_t *token,
|
||||
const char *text) {
|
||||
size_t length = strlen(text);
|
||||
return token->type == JSON_STRING && token->end - token->start == length &&
|
||||
!memcmp(data + token->start, text, length);
|
||||
}
|
||||
|
||||
static int
|
||||
json_next_child(const json_token_t *tokens, size_t count, int parent,
|
||||
size_t start) {
|
||||
for(size_t i = start; i < count; i++) {
|
||||
if(tokens[i].parent == parent) return (int)i;
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
|
||||
static int
|
||||
json_object_value(const unsigned char *data, const json_token_t *tokens,
|
||||
size_t count, int object, const char *key) {
|
||||
int item = json_next_child(tokens, count, object, (size_t)object + 1);
|
||||
|
||||
while(item >= 0) {
|
||||
int value = json_next_child(tokens, count, object, (size_t)item + 1);
|
||||
if(value < 0) return -1;
|
||||
if(json_token_equals(data, &tokens[item], key)) return value;
|
||||
item = json_next_child(tokens, count, object, (size_t)value + 1);
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
|
||||
static int
|
||||
json_object_value_token(const unsigned char *data, const json_token_t *tokens,
|
||||
size_t count, int object, const json_token_t *key) {
|
||||
int item = json_next_child(tokens, count, object, (size_t)object + 1);
|
||||
size_t key_size = key->end - key->start;
|
||||
|
||||
while(item >= 0) {
|
||||
int value = json_next_child(tokens, count, object, (size_t)item + 1);
|
||||
if(value < 0) return -1;
|
||||
if(tokens[item].type == JSON_STRING &&
|
||||
tokens[item].end - tokens[item].start == key_size &&
|
||||
!memcmp(data + tokens[item].start, data + key->start, key_size)) {
|
||||
return value;
|
||||
}
|
||||
item = json_next_child(tokens, count, object, (size_t)value + 1);
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
|
||||
static void
|
||||
append_json_token_string(strbuf_t *json, const unsigned char *data,
|
||||
const json_token_t *token) {
|
||||
strbuf_append(json, "\"");
|
||||
strbuf_printf(json, "%.*s", (int)(token->end - token->start),
|
||||
data + token->start);
|
||||
strbuf_append(json, "\"");
|
||||
}
|
||||
|
||||
static void
|
||||
append_json_field(strbuf_t *json, const unsigned char *data,
|
||||
const json_token_t *key, const json_token_t *value,
|
||||
int *first) {
|
||||
if(!*first) strbuf_append(json, ",");
|
||||
*first = 0;
|
||||
strbuf_append(json, "{\"name\":");
|
||||
append_json_token_string(json, data, key);
|
||||
strbuf_append(json, ",\"value\":");
|
||||
append_json_token_string(json, data, value);
|
||||
strbuf_append(json, "}");
|
||||
}
|
||||
|
||||
static void
|
||||
append_named_json_field(strbuf_t *json, const char *name,
|
||||
const unsigned char *data, const json_token_t *value,
|
||||
int *first) {
|
||||
if(!*first) strbuf_append(json, ",");
|
||||
*first = 0;
|
||||
strbuf_append(json, "{\"name\":");
|
||||
json_escape(json, name);
|
||||
strbuf_append(json, ",\"value\":");
|
||||
append_json_token_string(json, data, value);
|
||||
strbuf_append(json, "}");
|
||||
}
|
||||
|
||||
static int
|
||||
append_param_json_fields(strbuf_t *json, const unsigned char *data,
|
||||
size_t size) {
|
||||
json_token_t *tokens = calloc(JSON_TOKEN_MAX, sizeof(*tokens));
|
||||
size_t count = 0;
|
||||
int first = 1;
|
||||
int localized;
|
||||
int title = -1;
|
||||
|
||||
if(!tokens || parse_json_tokens(data, size, tokens, &count)) {
|
||||
free(tokens);
|
||||
return -1;
|
||||
}
|
||||
strbuf_append(json, "[");
|
||||
|
||||
localized = json_object_value(data, tokens, count, 0, "localizedParameters");
|
||||
if(localized >= 0 && tokens[localized].type == JSON_OBJECT) {
|
||||
int language = json_object_value(data, tokens, count, localized,
|
||||
"defaultLanguage");
|
||||
int language_data = language >= 0 && tokens[language].type == JSON_STRING ?
|
||||
json_object_value_token(data, tokens, count, localized,
|
||||
&tokens[language]) : -1;
|
||||
if(language_data >= 0 && tokens[language_data].type == JSON_OBJECT) {
|
||||
title = json_object_value(data, tokens, count, language_data,
|
||||
"titleName");
|
||||
}
|
||||
if(title < 0) {
|
||||
int english = json_object_value(data, tokens, count, localized, "en-US");
|
||||
if(english >= 0 && tokens[english].type == JSON_OBJECT) {
|
||||
title = json_object_value(data, tokens, count, english, "titleName");
|
||||
}
|
||||
}
|
||||
if(title < 0) {
|
||||
int item = json_next_child(tokens, count, localized,
|
||||
(size_t)localized + 1);
|
||||
while(item >= 0 && title < 0) {
|
||||
int value = json_next_child(tokens, count, localized,
|
||||
(size_t)item + 1);
|
||||
if(value < 0) break;
|
||||
if(tokens[value].type == JSON_OBJECT) {
|
||||
title = json_object_value(data, tokens, count, value, "titleName");
|
||||
}
|
||||
item = json_next_child(tokens, count, localized, (size_t)value + 1);
|
||||
}
|
||||
}
|
||||
}
|
||||
if(title >= 0 && tokens[title].type == JSON_STRING) {
|
||||
append_named_json_field(json, "titleName", data, &tokens[title], &first);
|
||||
}
|
||||
|
||||
for(int item = json_next_child(tokens, count, 0, 1); item >= 0;) {
|
||||
int value = json_next_child(tokens, count, 0, (size_t)item + 1);
|
||||
if(value < 0) break;
|
||||
if(tokens[item].type == JSON_STRING &&
|
||||
(tokens[value].type == JSON_STRING ||
|
||||
tokens[value].type == JSON_PRIMITIVE)) {
|
||||
append_json_field(json, data, &tokens[item], &tokens[value], &first);
|
||||
}
|
||||
item = json_next_child(tokens, count, 0, (size_t)value + 1);
|
||||
}
|
||||
strbuf_append(json, "]");
|
||||
free(tokens);
|
||||
return 0;
|
||||
}
|
||||
|
||||
static enum MHD_Result
|
||||
pkg_error(struct MHD_Connection *conn, const char *path) {
|
||||
return send_json_error_detail(conn, MHD_HTTP_BAD_REQUEST,
|
||||
"could not read package information",
|
||||
"pkg_info_failed", path);
|
||||
}
|
||||
|
||||
enum MHD_Result
|
||||
api_pkg_info(struct MHD_Connection *conn) {
|
||||
char *path = fs_path_value(query_value(conn, "path"));
|
||||
pkg_source_t pkg = {.fd = -1};
|
||||
unsigned char *param;
|
||||
strbuf_t json = {0};
|
||||
|
||||
if(!path || pkg_source_open(path, &pkg)) {
|
||||
enum MHD_Result ret = pkg_error(conn, path);
|
||||
free(path);
|
||||
return ret;
|
||||
}
|
||||
if(!(param = malloc(pkg.param_size)) ||
|
||||
read_at(pkg.fd, param, pkg.param_size, pkg.param_offset)) {
|
||||
enum MHD_Result ret = pkg_error(conn, path);
|
||||
free(param);
|
||||
pkg_source_close(&pkg);
|
||||
free(path);
|
||||
return ret;
|
||||
}
|
||||
|
||||
strbuf_printf(&json,
|
||||
"{\"ok\":true,\"size\":%llu,\"content_id\":",
|
||||
(unsigned long long)pkg.size);
|
||||
json_escape(&json, pkg.content_id);
|
||||
strbuf_printf(&json,
|
||||
",\"platform\":\"%s\",\"content_type\":%u,"
|
||||
"\"content_flags\":%u,\"has_icon\":%s,\"fields\":",
|
||||
pkg.param_type == PKG_PARAM_JSON ? "PS5" : "PS4",
|
||||
pkg.content_type, pkg.content_flags,
|
||||
pkg.icon_size ? "true" : "false");
|
||||
if((pkg.param_type == PKG_PARAM_JSON &&
|
||||
append_param_json_fields(&json, param, pkg.param_size)) ||
|
||||
(pkg.param_type == PKG_PARAM_SFO &&
|
||||
append_sfo_fields(&json, param, pkg.param_size))) {
|
||||
enum MHD_Result ret;
|
||||
free(json.data);
|
||||
ret = pkg_error(conn, path);
|
||||
free(param);
|
||||
pkg_source_close(&pkg);
|
||||
free(path);
|
||||
return ret;
|
||||
}
|
||||
strbuf_append(&json, "}");
|
||||
|
||||
free(param);
|
||||
pkg_source_close(&pkg);
|
||||
free(path);
|
||||
return send_buffer(conn, MHD_HTTP_OK, json.data, "application/json");
|
||||
}
|
||||
|
||||
enum MHD_Result
|
||||
api_pkg_icon(struct MHD_Connection *conn) {
|
||||
char *path = fs_path_value(query_value(conn, "path"));
|
||||
pkg_source_t pkg = {.fd = -1};
|
||||
unsigned char *icon;
|
||||
struct MHD_Response *response;
|
||||
enum MHD_Result ret;
|
||||
|
||||
if(!path || pkg_source_open(path, &pkg) || !pkg.icon_size ||
|
||||
!(icon = malloc(pkg.icon_size)) ||
|
||||
read_at(pkg.fd, icon, pkg.icon_size, pkg.icon_offset)) {
|
||||
if(path && pkg.fd >= 0) pkg_source_close(&pkg);
|
||||
free(path);
|
||||
return send_json_error(conn, MHD_HTTP_NOT_FOUND, "package icon not found");
|
||||
}
|
||||
if(!png_signature_valid(icon, pkg.icon_size)) {
|
||||
free(icon);
|
||||
pkg_source_close(&pkg);
|
||||
free(path);
|
||||
return send_json_error(conn, MHD_HTTP_NOT_FOUND,
|
||||
"package icon not found");
|
||||
}
|
||||
pkg_source_close(&pkg);
|
||||
free(path);
|
||||
|
||||
response = MHD_create_response_from_buffer(pkg.icon_size, icon,
|
||||
MHD_RESPMEM_MUST_FREE);
|
||||
if(!response) {
|
||||
free(icon);
|
||||
return MHD_NO;
|
||||
}
|
||||
MHD_add_response_header(response, MHD_HTTP_HEADER_CONTENT_TYPE, "image/png");
|
||||
ret = websrv_queue_response(conn, MHD_HTTP_OK, response);
|
||||
MHD_destroy_response(response);
|
||||
return ret;
|
||||
}
|
||||
@@ -0,0 +1,6 @@
|
||||
#pragma once
|
||||
|
||||
#include <microhttpd.h>
|
||||
|
||||
enum MHD_Result api_pkg_info(struct MHD_Connection *conn);
|
||||
enum MHD_Result api_pkg_icon(struct MHD_Connection *conn);
|
||||
@@ -0,0 +1,119 @@
|
||||
#include "pkg_installer.h"
|
||||
|
||||
#ifndef __linux__
|
||||
|
||||
#include <limits.h>
|
||||
#include <pthread.h>
|
||||
#include <stdint.h>
|
||||
#include <stdio.h>
|
||||
#include <string.h>
|
||||
|
||||
typedef struct pkg_metadata {
|
||||
const char *uri;
|
||||
const char *ex_uri;
|
||||
const char *playgo_scenario_id;
|
||||
const char *content_id;
|
||||
const char *content_name;
|
||||
const char *icon_url;
|
||||
uint32_t slot;
|
||||
uint32_t is_playgo_enabled;
|
||||
} pkg_metadata_t;
|
||||
|
||||
_Static_assert(sizeof(pkg_metadata_t) == 0x38,
|
||||
"sceAppInstUtil metadata ABI mismatch");
|
||||
|
||||
typedef struct pkg_info {
|
||||
char content_id[48];
|
||||
int type;
|
||||
int platform;
|
||||
} pkg_info_t;
|
||||
|
||||
typedef struct playgo_info {
|
||||
char languages[30][8];
|
||||
char scenario_ids[64][3];
|
||||
char content_ids[64][48];
|
||||
long unknown[810];
|
||||
} playgo_info_t;
|
||||
|
||||
_Static_assert(sizeof(playgo_info_t) == 0x2700,
|
||||
"sceAppInstUtil PlayGoInfo ABI mismatch");
|
||||
|
||||
int sceAppInstUtilInitialize(void);
|
||||
int sceAppInstUtilInstallByPackage(const pkg_metadata_t *, pkg_info_t *,
|
||||
playgo_info_t *);
|
||||
|
||||
static pthread_mutex_t installer_lock = PTHREAD_MUTEX_INITIALIZER;
|
||||
static int installer_initialized;
|
||||
|
||||
static int
|
||||
initialize_locked(void) {
|
||||
int result;
|
||||
|
||||
if(!installer_initialized) {
|
||||
result = sceAppInstUtilInitialize();
|
||||
if(result) return result;
|
||||
installer_initialized = 1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
int
|
||||
pkg_installer_initialize(void) {
|
||||
int result;
|
||||
|
||||
pthread_mutex_lock(&installer_lock);
|
||||
result = initialize_locked();
|
||||
pthread_mutex_unlock(&installer_lock);
|
||||
return result;
|
||||
}
|
||||
|
||||
int
|
||||
pkg_installer_install(const char *path) {
|
||||
char install_path[PATH_MAX + sizeof("/user")];
|
||||
const char *uri = path;
|
||||
pkg_metadata_t metadata = {
|
||||
.uri = NULL,
|
||||
.ex_uri = "",
|
||||
.playgo_scenario_id = "",
|
||||
.content_id = "",
|
||||
.content_name = "",
|
||||
.icon_url = "",
|
||||
.slot = 0,
|
||||
.is_playgo_enabled = 0
|
||||
};
|
||||
pkg_info_t pkg_info = {0};
|
||||
playgo_info_t playgo_info = {0};
|
||||
int result;
|
||||
|
||||
if(!path) return -1;
|
||||
if(!strncmp(path, "/data/", 6)) {
|
||||
snprintf(install_path, sizeof(install_path), "/user%s", path);
|
||||
uri = install_path;
|
||||
}
|
||||
metadata.uri = uri;
|
||||
|
||||
pthread_mutex_lock(&installer_lock);
|
||||
result = initialize_locked();
|
||||
if(result) {
|
||||
pthread_mutex_unlock(&installer_lock);
|
||||
return result;
|
||||
}
|
||||
result = sceAppInstUtilInstallByPackage(&metadata, &pkg_info, &playgo_info);
|
||||
pthread_mutex_unlock(&installer_lock);
|
||||
return result;
|
||||
}
|
||||
|
||||
#else
|
||||
|
||||
int
|
||||
pkg_installer_initialize(void) {
|
||||
return PKG_INSTALL_UNSUPPORTED;
|
||||
}
|
||||
|
||||
int
|
||||
pkg_installer_install(const char *path) {
|
||||
(void)path;
|
||||
return PKG_INSTALL_UNSUPPORTED;
|
||||
}
|
||||
|
||||
#endif
|
||||
@@ -0,0 +1,6 @@
|
||||
#pragma once
|
||||
|
||||
#define PKG_INSTALL_UNSUPPORTED 0x7fffffff
|
||||
|
||||
int pkg_installer_initialize(void);
|
||||
int pkg_installer_install(const char *path);
|
||||
@@ -0,0 +1,39 @@
|
||||
#pragma once
|
||||
|
||||
/* Safe RAR extraction engine used by the /api/extract task.
|
||||
Wraps the vendored rarlab UnRAR 7.20.1 (third_party/unrar7) through its
|
||||
C-compatible DLL API (unrar_c_api.h facade).
|
||||
|
||||
Reuses the zip_extract types so the dispatch layer can call either engine
|
||||
through the same status / limits / progress protocol.
|
||||
|
||||
See zip_extract.h for the shared limits, conflict, progress and result types.
|
||||
|
||||
Backend notes (v1.9, unrar 7.20.1):
|
||||
* RAR4 and RAR5, any compression version including WinRAR 6/7 "v6".
|
||||
* Multi-volume: unrar merges next .partNN.rar by name automatically.
|
||||
* Encrypted archives work end-to-end (both `-p` data encryption and
|
||||
`-hp` header encryption). Pass the password in `password`; NULL or an
|
||||
empty string means "no password supplied". A missing or wrong password
|
||||
is reported as ZIPX_ERR_PASSWORD so the caller can prompt and retry.
|
||||
|
||||
See third_party/unrar7/VENDORED.md for full integration notes. */
|
||||
|
||||
#include "zip_extract.h"
|
||||
|
||||
#include <stdint.h>
|
||||
#include <stddef.h>
|
||||
|
||||
/* Extract rar_path into dst_dir using the same protocol as zipx_extract().
|
||||
`password` may be NULL when the archive is not encrypted.
|
||||
Returns ZIPX_OK or an error code; *result is always filled in.
|
||||
On any failure the staging directory is removed and dst_dir is left as it
|
||||
was, except for objects already published with the overwrite policy. */
|
||||
zipx_status_t rar_extract(const char *rar_path, const char *dst_dir,
|
||||
zipx_conflict_t conflict,
|
||||
const zipx_limits_t *limits,
|
||||
zipx_cancel_fn cancel,
|
||||
zipx_progress_fn progress,
|
||||
void *userdata,
|
||||
const char *password,
|
||||
zipx_result_t *result);
|
||||
@@ -0,0 +1,186 @@
|
||||
/* sevenz_chain -- 7z folder (coder chain) decoder.
|
||||
part of ps5-web-file-manager
|
||||
|
||||
A 7z archive stores its data as *folders*. One folder is a small directed
|
||||
graph of coders fed by N packed streams and producing a single unpacked
|
||||
stream; entries are slices of the folder output (a folder holding several
|
||||
entries is what makes an archive "solid").
|
||||
|
||||
The bundled LZMA SDK can decode a folder, but only through `CSzFolder`,
|
||||
which is a fixed-size structure capped at 4 coders / 3 bonds. 7-Zip's own
|
||||
BCJ2 chain uses 5 coders (BCJ2 plus four LZMA2 streams), so the SDK rejects
|
||||
it -- while still listing the archive fine, because its *header* scanner is
|
||||
a different, looser parser (64 coders). The C half of the SDK also has no
|
||||
7zAES coder at all.
|
||||
|
||||
This module therefore parses the folder descriptor itself (dynamic arrays,
|
||||
up to 64 coders, mirroring the SDK's header scanner) and drives the coder
|
||||
graph itself. Two properties matter:
|
||||
|
||||
* Streaming. The decoded bytes are pushed into a sink as they are
|
||||
produced; a folder is never materialised as a whole, so a multi-gigabyte
|
||||
solid block is workable. The only buffers sized from the archive are
|
||||
the LZMA/LZMA2 dictionary, the PPMd model and the three side streams of
|
||||
BCJ2 -- each capped by sz_chain_limits_t.
|
||||
|
||||
* Precision. Every rejection names the coder and the method, and the
|
||||
resource limits report the value the archive asked for and the value
|
||||
that was allowed, so the UI can say something useful instead of
|
||||
"corrupt archive".
|
||||
|
||||
Supported here:
|
||||
* Copy, LZMA, LZMA2 and PPMd
|
||||
* the Delta filter and the x86 / PPC / IA64 / ARM / ARMT / SPARC branch
|
||||
converters
|
||||
* BCJ2, whose three side streams are materialised under a limit while
|
||||
MAIN keeps streaming
|
||||
* 7zAES (method 0x06F10701), the coder 7-Zip wraps around the streams when
|
||||
`-p` is used, driven from a caller supplied password
|
||||
|
||||
An encrypted *header* (`-mhe=on`) is not a coder in a folder: it is a second,
|
||||
encrypted copy of the archive header that has to be decoded before any folder
|
||||
exists at all. src/sevenz_header.c does that -- by parsing that one folder's
|
||||
descriptor and running it through this module, password and all.
|
||||
*/
|
||||
|
||||
#ifndef SEVENZ_CHAIN_H
|
||||
#define SEVENZ_CHAIN_H
|
||||
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
|
||||
/* The SDK's header scanner accepts up to 64 coders per folder; folders in the
|
||||
wild have 1-5. Keeping the same ceiling means "the SDK could list it" and
|
||||
"we can decode it" accept the same archives. */
|
||||
#define SZ_CHAIN_MAX_CODERS 64
|
||||
#define SZ_CHAIN_MAX_STREAMS 64
|
||||
|
||||
/* Ceilings for the buffers whose size comes from the (attacker controlled)
|
||||
archive header. */
|
||||
typedef struct {
|
||||
uint64_t max_dict_bytes; /* LZMA / LZMA2 window */
|
||||
uint64_t max_ppmd_bytes; /* PPMd model */
|
||||
uint64_t max_side_bytes; /* BCJ2 CALL + JUMP + RC together */
|
||||
uint32_t max_aes_cycles; /* 7zAES key derivation: 2^n SHA-256 passes */
|
||||
} sz_chain_limits_t;
|
||||
|
||||
#define SZ_CHAIN_LIMITS_DEFAULT 0
|
||||
#define SZ_CHAIN_LIMITS_LARGE 1
|
||||
const sz_chain_limits_t *sz_chain_limits_profile(int profile);
|
||||
const sz_chain_limits_t *sz_chain_default_limits(void);
|
||||
|
||||
typedef enum {
|
||||
SZ_CHAIN_OK = 0,
|
||||
SZ_CHAIN_ERR_PARAM, /* bad arguments from the caller */
|
||||
SZ_CHAIN_ERR_MEM, /* allocation failed */
|
||||
SZ_CHAIN_ERR_HEADER, /* malformed folder descriptor */
|
||||
SZ_CHAIN_ERR_METHOD, /* coder method not supported */
|
||||
SZ_CHAIN_ERR_LAYOUT, /* coder graph shape not supported */
|
||||
SZ_CHAIN_ERR_LIMIT, /* a sz_chain_limits_t ceiling was hit */
|
||||
SZ_CHAIN_ERR_PASSWORD,/* the archive is encrypted and no usable password
|
||||
was supplied (or the one given is wrong) */
|
||||
SZ_CHAIN_ERR_READ, /* the read callback failed */
|
||||
SZ_CHAIN_ERR_WRITE, /* the sink callback failed */
|
||||
SZ_CHAIN_ERR_DATA, /* a decoder rejected the data */
|
||||
SZ_CHAIN_ERR_CANCELED,
|
||||
SZ_CHAIN_ERR_INTERNAL
|
||||
} sz_chain_status_t;
|
||||
|
||||
typedef struct {
|
||||
sz_chain_status_t status;
|
||||
int32_t coder; /* index of the offending coder, -1 when not applicable */
|
||||
uint32_t method; /* its method id, 0 when not applicable */
|
||||
uint64_t offset; /* decoded byte offset at the point of failure */
|
||||
char message[192];
|
||||
} sz_chain_err_t;
|
||||
|
||||
const char *sz_chain_status_string(sz_chain_status_t status);
|
||||
|
||||
/* "LZMA2", "BCJ2", "7zAES", "unknown 0x1234". Never returns NULL. */
|
||||
const char *sz_chain_method_name(uint32_t method);
|
||||
|
||||
/* ---------------------------------------------------------------- folder */
|
||||
|
||||
typedef struct sz_chain sz_chain;
|
||||
|
||||
/* Parses one folder descriptor.
|
||||
|
||||
blob / blob_size
|
||||
the CODERS_INFO bytes of this folder, i.e. the range
|
||||
`CSzAr::CodersData[FoCodersOffsets[i] .. FoCodersOffsets[i + 1])`.
|
||||
pack_positions
|
||||
`CSzAr::PackPositions`, num_pack_streams + 1 entries, offsets of the
|
||||
packed streams relative to the start of the archive's packed-data area.
|
||||
coder_unpack_sizes
|
||||
unpacked size of every coder of this folder, in stored coder order, i.e.
|
||||
`&CSzAr::CoderUnpackSizes[CSzAr::FoToCoderUnpackSizes[i]]`.
|
||||
unpack_size
|
||||
`SzAr_GetFolderUnpackSize(&db, i)`.
|
||||
|
||||
Returns 0 on success. On success *out owns a copy of blob, release it with
|
||||
sz_chain_free(). On failure *out is untouched and err describes the
|
||||
problem. */
|
||||
int sz_chain_parse(sz_chain **out, const uint8_t *blob, size_t blob_size,
|
||||
const uint64_t *pack_positions, uint32_t num_pack_streams,
|
||||
const uint64_t *coder_unpack_sizes, uint64_t unpack_size,
|
||||
const sz_chain_limits_t *limits, sz_chain_err_t *err);
|
||||
|
||||
void sz_chain_free(sz_chain *c);
|
||||
|
||||
uint32_t sz_chain_num_coders(const sz_chain *c);
|
||||
uint32_t sz_chain_num_pack_streams(const sz_chain *c);
|
||||
/* Non-zero when the folder contains a 7zAES coder, i.e. when sz_chain_decode()
|
||||
will need a password. Lets a caller ask for one before touching the disk. */
|
||||
int sz_chain_needs_password(const sz_chain *c);
|
||||
/* Method id of coder `index`, or -1 when out of range. */
|
||||
int64_t sz_chain_coder_method(const sz_chain *c, uint32_t index);
|
||||
|
||||
/* True when the folder is a single plain LZMA2 coder -- the shape the SDK's
|
||||
multithreaded decoder covers. Fills the coder's props byte and the packed
|
||||
input size; both are only valid when this returns non-zero. */
|
||||
int sz_chain_lzma2_root(const sz_chain *c, uint8_t *prop, uint64_t *in_size);
|
||||
|
||||
/* Writes e.g. "LZMA2 + BCJ2 (5 coders, 4 pack streams)" into buf. */
|
||||
void sz_chain_describe(const sz_chain *c, char *buf, size_t size);
|
||||
|
||||
/* Walks every coder and the graph shape without touching any data, so callers
|
||||
can refuse an archive before creating anything on disk. Fills err with the
|
||||
same precision sz_chain_decode() would. */
|
||||
int sz_chain_check(const sz_chain *c, sz_chain_err_t *err);
|
||||
|
||||
/* -------------------------------------------------------------- decoding */
|
||||
|
||||
/* Fills exactly size bytes at offset inside the packed-data area.
|
||||
Returns 0 on success, non-zero on failure. */
|
||||
typedef int (*sz_chain_read_fn)(void *ctx, uint64_t offset, void *dst,
|
||||
size_t size);
|
||||
|
||||
/* Receives the decoded bytes in order. Returns 0 to continue. */
|
||||
typedef int (*sz_chain_sink_fn)(void *ctx, const void *data, size_t size);
|
||||
|
||||
/* Returns non-zero to abort. May be NULL. */
|
||||
typedef int (*sz_chain_cancel_fn)(void *ctx);
|
||||
|
||||
/* Decodes the whole folder, pushing the result into sink.
|
||||
|
||||
The bytes are delivered strictly in order and the total is the folder's
|
||||
declared unpack size. When crc_out is not NULL it receives the CRC-32 of
|
||||
the delivered bytes, for the caller to compare with the folder CRC.
|
||||
|
||||
`password` is the archive password as UTF-8, or NULL / "" when the caller
|
||||
has none. It is only consulted by folders that contain a 7zAES coder; a
|
||||
folder that needs one without a password fails as SZ_CHAIN_ERR_PASSWORD
|
||||
before any data is read, so the caller can prompt and retry. A password
|
||||
containing NUL is not supported: 7-Zip stores it as UTF-16LE and the
|
||||
conversion stops at the terminator.
|
||||
|
||||
Returns 0 on success, -1 on failure with err filled. A failing sink is
|
||||
reported as SZ_CHAIN_ERR_WRITE; the caller is expected to make its own
|
||||
message more specific. */
|
||||
int sz_chain_decode(sz_chain *c, sz_chain_read_fn read_at, void *read_ctx,
|
||||
sz_chain_sink_fn sink, void *sink_ctx,
|
||||
sz_chain_cancel_fn cancel, void *cancel_ctx,
|
||||
const char *password, uint32_t *crc_out,
|
||||
sz_chain_err_t *err);
|
||||
|
||||
#endif /* SEVENZ_CHAIN_H */
|
||||
@@ -0,0 +1,38 @@
|
||||
#pragma once
|
||||
|
||||
/* Standalone 7z extraction engine, the third sibling of zip_extract.c and
|
||||
rar_extract.c. Like them it has no HTTP or task dependencies, and it fills
|
||||
in the same zipx_result_t so a caller can treat every format alike.
|
||||
|
||||
Input may be a single `name.7z` or a byte-split set (`name.7z.001`, ...):
|
||||
both reach the decoder through src/sevenz_volstream.c.
|
||||
|
||||
The publish / staging / rollback / name-validation machinery is mirrored
|
||||
from rar_extract.c on purpose -- three self-contained engines is the shape
|
||||
this project has settled on, so that a format's bugs stay inside its file.
|
||||
|
||||
Backend notes (LZMA SDK 26.03 + src/sevenz_chain.c):
|
||||
* Copy / LZMA / LZMA2 / PPMd, the Delta filter and the x86 / PPC / IA64 /
|
||||
ARM / ARMT / SPARC branch converters, BCJ2, and 7zAES.
|
||||
* An encrypted *header* (`-mhe=on`) is decrypted by src/sevenz_header.c
|
||||
first: the SDK refuses such an archive before any folder is known, so
|
||||
the header has to be readable before the SDK is asked to read it.
|
||||
*/
|
||||
|
||||
#include "zip_extract.h"
|
||||
|
||||
/* Extract sevenz_path into dst_dir.
|
||||
`password` is the archive password as UTF-8, or NULL / "" when the caller
|
||||
has none. It is only consulted by archives that encrypt their streams.
|
||||
|
||||
Returns ZIPX_OK or an error code; *result is always filled in. A missing or
|
||||
wrong password comes back as ZIPX_ERR_PASSWORD so the caller can ask for one
|
||||
and retry. On any failure the staging directory is removed and dst_dir is
|
||||
left as it was, except for objects already published under the overwrite
|
||||
policy. */
|
||||
zipx_status_t sevenz_extract(const char *sevenz_path, const char *dst_dir,
|
||||
zipx_conflict_t conflict,
|
||||
const zipx_limits_t *limits,
|
||||
zipx_cancel_fn cancel,
|
||||
zipx_progress_fn progress, void *userdata,
|
||||
const char *password, zipx_result_t *result);
|
||||
@@ -0,0 +1,917 @@
|
||||
/* sevenz_header -- see sevenz_header.h for what this does and why. */
|
||||
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
|
||||
#include "7z.h"
|
||||
#include "7zCrc.h"
|
||||
#include "7zTypes.h"
|
||||
|
||||
#include "sevenz_chain.h"
|
||||
#include "sevenz_header.h"
|
||||
|
||||
/* The header property ids we have to recognise. They are the same enum the
|
||||
SDK's header scanner uses (7zArcIn.c), spelled out here so this file does
|
||||
not depend on that translation unit's internals. */
|
||||
#define SZH_ID_END 0x00
|
||||
#define SZH_ID_HEADER 0x01
|
||||
#define SZH_ID_PACK_INFO 0x06
|
||||
#define SZH_ID_UNPACK_INFO 0x07
|
||||
#define SZH_ID_SIZE 0x09
|
||||
#define SZH_ID_CRC 0x0A
|
||||
#define SZH_ID_FOLDER 0x0B
|
||||
#define SZH_ID_CODERS_UNPACK_SIZE 0x0C
|
||||
#define SZH_ID_ENCODED_HEADER 0x17
|
||||
|
||||
#define SZH_MAX_CODERS SZ_CHAIN_MAX_CODERS
|
||||
#define SZH_MAX_STREAMS SZ_CHAIN_MAX_STREAMS
|
||||
|
||||
/* --------------------------------------------------------------- numbers */
|
||||
|
||||
static uint32_t
|
||||
szh_le32(const uint8_t *p) {
|
||||
return (uint32_t)p[0] | ((uint32_t)p[1] << 8) | ((uint32_t)p[2] << 16) |
|
||||
((uint32_t)p[3] << 24);
|
||||
}
|
||||
|
||||
static uint64_t
|
||||
szh_le64(const uint8_t *p) {
|
||||
return (uint64_t)szh_le32(p) | ((uint64_t)szh_le32(p + 4) << 32);
|
||||
}
|
||||
|
||||
static void
|
||||
szh_put_le32(uint8_t *p, uint32_t v) {
|
||||
p[0] = (uint8_t)v;
|
||||
p[1] = (uint8_t)(v >> 8);
|
||||
p[2] = (uint8_t)(v >> 16);
|
||||
p[3] = (uint8_t)(v >> 24);
|
||||
}
|
||||
|
||||
static void
|
||||
szh_put_le64(uint8_t *p, uint64_t v) {
|
||||
szh_put_le32(p, (uint32_t)v);
|
||||
szh_put_le32(p + 4, (uint32_t)(v >> 32));
|
||||
}
|
||||
|
||||
/* The 7z variable length number: the high bits of the first byte say how many
|
||||
more bytes follow, and the remaining bits of the first byte are the high
|
||||
part of the value. This mirrors ReadNumber() in 7zArcIn.c byte for byte,
|
||||
including its habit of returning a partial value when all eight flag bits
|
||||
are set -- the caller checks the CRC of the whole record anyway. */
|
||||
static int
|
||||
szh_num(const uint8_t *d, size_t size, size_t *pos, uint64_t *value) {
|
||||
size_t p = *pos;
|
||||
unsigned first, mask, v, i;
|
||||
|
||||
if(p >= size) {
|
||||
return -1;
|
||||
}
|
||||
first = d[p++];
|
||||
if((first & 0x80) == 0) {
|
||||
*value = first;
|
||||
*pos = p;
|
||||
return 0;
|
||||
}
|
||||
if(p >= size) {
|
||||
return -1;
|
||||
}
|
||||
v = d[p++];
|
||||
if((first & 0x40) == 0) {
|
||||
*value = ((uint64_t)(first & 0x3F) << 8) | v;
|
||||
*pos = p;
|
||||
return 0;
|
||||
}
|
||||
if(p >= size) {
|
||||
return -1;
|
||||
}
|
||||
mask = d[p++];
|
||||
*value = (uint64_t)v | ((uint64_t)mask << 8);
|
||||
mask = 0x20;
|
||||
for(i = 16; i < 64; i += 8) {
|
||||
if((first & mask) == 0) {
|
||||
*value |= (uint64_t)(first & (mask - 1)) << i;
|
||||
*pos = p;
|
||||
return 0;
|
||||
}
|
||||
mask >>= 1;
|
||||
if(p >= size) {
|
||||
return -1;
|
||||
}
|
||||
*value |= (uint64_t)d[p++] << i;
|
||||
}
|
||||
*pos = p;
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* The 32 bit form used for counts and property sizes. */
|
||||
static int
|
||||
szh_num32(const uint8_t *d, size_t size, size_t *pos, uint32_t *value) {
|
||||
uint64_t v;
|
||||
|
||||
if(*pos < size && (d[*pos] & 0x80) == 0) {
|
||||
*value = d[(*pos)++];
|
||||
return 0;
|
||||
}
|
||||
if(szh_num(d, size, pos, &v)) {
|
||||
return -1;
|
||||
}
|
||||
if(v >= (uint64_t)0x80000000u - 1) {
|
||||
return -1;
|
||||
}
|
||||
*value = (uint32_t)v;
|
||||
return 0;
|
||||
}
|
||||
|
||||
static uint32_t
|
||||
szh_count_bits(const uint8_t *d, uint32_t num_items) {
|
||||
uint32_t n = 0, i;
|
||||
|
||||
for(i = 0; i < num_items; i++) {
|
||||
if(d[i >> 3] & (0x80u >> (i & 7))) {
|
||||
n++;
|
||||
}
|
||||
}
|
||||
return n;
|
||||
}
|
||||
|
||||
/* A digest block: one "all are defined" byte, an optional bit vector, then one
|
||||
little endian CRC per defined item. With `first` the value of the first
|
||||
defined item comes back, which is the folder CRC when the block covers a
|
||||
single folder. */
|
||||
static int
|
||||
szh_digests(const uint8_t *d, size_t size, size_t *pos, uint32_t num_items,
|
||||
uint32_t *first, int *has_first) {
|
||||
size_t p = *pos;
|
||||
uint32_t defined = num_items;
|
||||
unsigned all;
|
||||
|
||||
if(p >= size) {
|
||||
return -1;
|
||||
}
|
||||
all = d[p++];
|
||||
if(!all) {
|
||||
size_t bytes = ((size_t)num_items + 7) >> 3;
|
||||
|
||||
if(bytes > size - p) {
|
||||
return -1;
|
||||
}
|
||||
defined = szh_count_bits(d + p, num_items);
|
||||
p += bytes;
|
||||
}
|
||||
if((size_t)defined > (size - p) >> 2) {
|
||||
return -1;
|
||||
}
|
||||
if(first && has_first) {
|
||||
if(defined) {
|
||||
*first = szh_le32(d + p);
|
||||
*has_first = 1;
|
||||
} else {
|
||||
*has_first = 0;
|
||||
}
|
||||
}
|
||||
p += (size_t)defined * 4;
|
||||
*pos = p;
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* The SDK's SkipData(): a length, then that many bytes. */
|
||||
static int
|
||||
szh_skip_data(const uint8_t *d, size_t size, size_t *pos) {
|
||||
uint64_t n;
|
||||
|
||||
if(szh_num(d, size, pos, &n)) {
|
||||
return -1;
|
||||
}
|
||||
if(n > (uint64_t)(size - *pos)) {
|
||||
return -1;
|
||||
}
|
||||
*pos += (size_t)n;
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* --------------------------------------------------------- encoded header */
|
||||
|
||||
typedef struct {
|
||||
uint64_t pack_pos; /* where this folder's packs live, from offset 32 */
|
||||
uint32_t num_pack;
|
||||
uint64_t pack_sizes[SZH_MAX_STREAMS];
|
||||
uint64_t pack_positions[SZH_MAX_STREAMS + 1]; /* cumulative, as the SDK builds */
|
||||
const uint8_t *blob; /* coder descriptor, in place */
|
||||
size_t blob_size;
|
||||
uint32_t num_coders;
|
||||
uint32_t main_coder;
|
||||
uint64_t coder_unpack_sizes[SZH_MAX_CODERS];
|
||||
uint64_t unpack_size;
|
||||
uint32_t crc;
|
||||
int has_crc;
|
||||
} szh_streams_t;
|
||||
|
||||
/* One folder descriptor: the coders, then the bond pairs and the pack-stream
|
||||
indices that follow them. `*pos` ends up just past the last of those, so
|
||||
[start, *pos) is exactly the byte range the SDK keeps as
|
||||
CSzAr::CodersData[FoCodersOffsets[0] .. FoCodersOffsets[1]). */
|
||||
static int
|
||||
szh_parse_folder(const uint8_t *d, size_t size, size_t *pos, szh_streams_t *ss) {
|
||||
const size_t start = *pos;
|
||||
size_t p = start;
|
||||
uint32_t num_coders = 0, num_in = 0, num_bonds, num_pack, i;
|
||||
uint8_t coder_used[SZH_MAX_CODERS];
|
||||
uint8_t stream_used[SZH_MAX_STREAMS];
|
||||
uint32_t main_index = 0;
|
||||
|
||||
if(szh_num32(d, size, &p, &num_coders)) {
|
||||
return -1;
|
||||
}
|
||||
if(num_coders == 0 || num_coders > SZH_MAX_CODERS) {
|
||||
return -1;
|
||||
}
|
||||
|
||||
for(i = 0; i < num_coders; i++) {
|
||||
uint8_t main_byte;
|
||||
uint32_t id_size, coder_in = 1;
|
||||
|
||||
if(p >= size) {
|
||||
return -1;
|
||||
}
|
||||
main_byte = d[p++];
|
||||
if(main_byte & 0xC0) {
|
||||
return -1;
|
||||
}
|
||||
id_size = main_byte & 0x0F;
|
||||
if(id_size > 8 || (size_t)id_size > size - p) {
|
||||
return -1;
|
||||
}
|
||||
p += id_size;
|
||||
if(main_byte & 0x10) {
|
||||
uint32_t coder_out;
|
||||
|
||||
if(szh_num32(d, size, &p, &coder_in) ||
|
||||
szh_num32(d, size, &p, &coder_out)) {
|
||||
return -1;
|
||||
}
|
||||
/* The header scanner accepts exactly one output stream per coder, so a
|
||||
coder index and its output-stream index coincide. */
|
||||
if(coder_out != 1) {
|
||||
return -1;
|
||||
}
|
||||
}
|
||||
if(num_in >= SZH_MAX_STREAMS || coder_in > SZH_MAX_STREAMS - num_in) {
|
||||
return -1;
|
||||
}
|
||||
num_in += coder_in;
|
||||
if(main_byte & 0x20) {
|
||||
uint32_t props_size;
|
||||
|
||||
if(szh_num32(d, size, &p, &props_size) ||
|
||||
(size_t)props_size > size - p) {
|
||||
return -1;
|
||||
}
|
||||
p += props_size;
|
||||
}
|
||||
}
|
||||
|
||||
num_bonds = num_coders - 1;
|
||||
if(num_in < num_bonds) {
|
||||
return -1;
|
||||
}
|
||||
num_pack = num_in - num_bonds;
|
||||
if(num_pack != ss->num_pack) {
|
||||
return -1;
|
||||
}
|
||||
|
||||
memset(coder_used, 0, sizeof(coder_used));
|
||||
memset(stream_used, 0, sizeof(stream_used));
|
||||
|
||||
for(i = 0; i < num_bonds; i++) {
|
||||
uint32_t in_index, out_index;
|
||||
|
||||
if(szh_num32(d, size, &p, &in_index)) {
|
||||
return -1;
|
||||
}
|
||||
if(in_index >= num_in || stream_used[in_index]) {
|
||||
return -1;
|
||||
}
|
||||
stream_used[in_index] = 1;
|
||||
if(szh_num32(d, size, &p, &out_index)) {
|
||||
return -1;
|
||||
}
|
||||
if(out_index >= num_coders || coder_used[out_index]) {
|
||||
return -1;
|
||||
}
|
||||
coder_used[out_index] = 1;
|
||||
}
|
||||
if(num_pack != 1) {
|
||||
for(i = 0; i < num_pack; i++) {
|
||||
uint32_t index;
|
||||
|
||||
if(szh_num32(d, size, &p, &index)) {
|
||||
return -1;
|
||||
}
|
||||
if(index >= num_in || stream_used[index]) {
|
||||
return -1;
|
||||
}
|
||||
stream_used[index] = 1;
|
||||
}
|
||||
}
|
||||
|
||||
/* The coder no bond consumes produces the folder's output. */
|
||||
while(main_index < num_coders && coder_used[main_index]) {
|
||||
main_index++;
|
||||
}
|
||||
if(main_index >= num_coders) {
|
||||
return -1;
|
||||
}
|
||||
|
||||
ss->blob = d + start;
|
||||
ss->blob_size = p - start;
|
||||
ss->num_coders = num_coders;
|
||||
ss->main_coder = main_index;
|
||||
*pos = p;
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* Reads the StreamsInfo of a k7zIdEncodedHeader record. Only the parts this
|
||||
module acts on are interpreted; anything else is skipped the way the SDK
|
||||
skips an id it does not know. */
|
||||
static int
|
||||
szh_parse_streams(const uint8_t *d, size_t size, szh_streams_t *ss) {
|
||||
size_t pos = 1; /* past k7zIdEncodedHeader */
|
||||
uint64_t id;
|
||||
uint32_t i;
|
||||
int seen_folder = 0, seen_sizes = 0;
|
||||
|
||||
memset(ss, 0, sizeof(*ss));
|
||||
|
||||
if(szh_num(d, size, &pos, &id) || id != SZH_ID_PACK_INFO) {
|
||||
return -1;
|
||||
}
|
||||
if(szh_num(d, size, &pos, &ss->pack_pos) ||
|
||||
szh_num32(d, size, &pos, &ss->num_pack)) {
|
||||
return -1;
|
||||
}
|
||||
if(ss->num_pack == 0 || ss->num_pack > SZH_MAX_STREAMS) {
|
||||
return -1;
|
||||
}
|
||||
{
|
||||
int got_sizes = 0;
|
||||
|
||||
for(;;) {
|
||||
if(szh_num(d, size, &pos, &id)) {
|
||||
return -1;
|
||||
}
|
||||
if(id == SZH_ID_END) {
|
||||
break;
|
||||
}
|
||||
if(id == SZH_ID_SIZE) {
|
||||
if(got_sizes) {
|
||||
return -1;
|
||||
}
|
||||
for(i = 0; i < ss->num_pack; i++) {
|
||||
if(szh_num(d, size, &pos, &ss->pack_sizes[i])) {
|
||||
return -1;
|
||||
}
|
||||
}
|
||||
got_sizes = 1;
|
||||
continue;
|
||||
}
|
||||
if(szh_skip_data(d, size, &pos)) {
|
||||
return -1;
|
||||
}
|
||||
}
|
||||
if(!got_sizes) {
|
||||
return -1;
|
||||
}
|
||||
}
|
||||
ss->pack_positions[0] = 0;
|
||||
for(i = 0; i < ss->num_pack; i++) {
|
||||
ss->pack_positions[i + 1] = ss->pack_positions[i] + ss->pack_sizes[i];
|
||||
if(ss->pack_positions[i + 1] < ss->pack_positions[i]) {
|
||||
return -1;
|
||||
}
|
||||
}
|
||||
|
||||
if(szh_num(d, size, &pos, &id) || id != SZH_ID_UNPACK_INFO) {
|
||||
return -1;
|
||||
}
|
||||
for(;;) {
|
||||
if(szh_num(d, size, &pos, &id)) {
|
||||
return -1;
|
||||
}
|
||||
if(id == SZH_ID_END) {
|
||||
break;
|
||||
}
|
||||
if(id == SZH_ID_FOLDER) {
|
||||
uint32_t num_folders;
|
||||
|
||||
if(seen_folder || szh_num32(d, size, &pos, &num_folders)) {
|
||||
return -1;
|
||||
}
|
||||
/* SzArEx_Open2 decodes this record with numFoldersMax = 1, and the
|
||||
`external` flag has to be clear for the table to be inline. */
|
||||
if(num_folders != 1 || pos >= size || d[pos] != 0) {
|
||||
return -1;
|
||||
}
|
||||
pos++;
|
||||
if(szh_parse_folder(d, size, &pos, ss)) {
|
||||
return -1;
|
||||
}
|
||||
seen_folder = 1;
|
||||
continue;
|
||||
}
|
||||
if(id == SZH_ID_CODERS_UNPACK_SIZE) {
|
||||
if(!seen_folder || seen_sizes) {
|
||||
return -1;
|
||||
}
|
||||
for(i = 0; i < ss->num_coders; i++) {
|
||||
if(szh_num(d, size, &pos, &ss->coder_unpack_sizes[i])) {
|
||||
return -1;
|
||||
}
|
||||
}
|
||||
seen_sizes = 1;
|
||||
continue;
|
||||
}
|
||||
if(id == SZH_ID_CRC) {
|
||||
uint32_t crc = 0;
|
||||
int has = 0;
|
||||
|
||||
if(ss->has_crc || szh_digests(d, size, &pos, 1, &crc, &has)) {
|
||||
return -1;
|
||||
}
|
||||
if(has) {
|
||||
ss->crc = crc;
|
||||
ss->has_crc = 1;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
if(szh_skip_data(d, size, &pos)) {
|
||||
return -1;
|
||||
}
|
||||
}
|
||||
if(!seen_folder || !seen_sizes) {
|
||||
return -1;
|
||||
}
|
||||
ss->unpack_size = ss->coder_unpack_sizes[ss->main_coder];
|
||||
|
||||
/* A SubStreamsInfo may hold the folder CRC when UnpackInfo carried none, but
|
||||
nothing past this point changes what we decode. */
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* ------------------------------------------------------------- stream I/O */
|
||||
|
||||
static uint64_t
|
||||
szh_size_of(ISeekInStream *s) {
|
||||
Int64 pos = 0;
|
||||
|
||||
if(s->Seek(s, &pos, SZ_SEEK_END) != SZ_OK || pos < 0) {
|
||||
return 0;
|
||||
}
|
||||
return (uint64_t)pos;
|
||||
}
|
||||
|
||||
static int
|
||||
szh_read_at(ISeekInStream *s, uint64_t offset, void *dst, size_t size) {
|
||||
Int64 pos = (Int64)offset;
|
||||
size_t got = 0;
|
||||
|
||||
if(s->Seek(s, &pos, SZ_SEEK_SET) != SZ_OK) {
|
||||
return -1;
|
||||
}
|
||||
while(got < size) {
|
||||
size_t want = size - got;
|
||||
|
||||
if(s->Read(s, (uint8_t *)dst + got, &want) != SZ_OK) {
|
||||
return -1;
|
||||
}
|
||||
if(want == 0) {
|
||||
return -1;
|
||||
}
|
||||
got += want;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* The virtual view handed to the SDK. `vt` has to stay first: the SDK casts
|
||||
the interface pointer straight back to this struct, exactly as
|
||||
sevenz_volstream.c does. */
|
||||
typedef struct {
|
||||
ISeekInStream vt;
|
||||
|
||||
ISeekInStream *raw;
|
||||
uint64_t raw_size;
|
||||
uint64_t pos;
|
||||
uint64_t total; /* the virtual length; the header may stick out past the file */
|
||||
|
||||
uint8_t sig[k7zStartHeaderSize]; /* start header, rewritten for the plaintext */
|
||||
uint64_t hdr_off;
|
||||
uint8_t *hdr;
|
||||
uint64_t hdr_len;
|
||||
} szh_view;
|
||||
|
||||
struct szh_prep {
|
||||
szh_view view;
|
||||
};
|
||||
|
||||
static SRes
|
||||
szh_view_read(const ISeekInStream *p, void *buf, size_t *size) {
|
||||
szh_view *v = (szh_view *)p;
|
||||
uint8_t *dst = (uint8_t *)buf;
|
||||
const size_t want = *size;
|
||||
size_t got = 0;
|
||||
|
||||
*size = 0;
|
||||
while(got < want) {
|
||||
const uint64_t pos = v->pos;
|
||||
const uint64_t hdr_end = v->hdr_off + v->hdr_len;
|
||||
|
||||
if(pos < k7zStartHeaderSize) {
|
||||
size_t n = (size_t)(k7zStartHeaderSize - pos);
|
||||
|
||||
if(n > want - got) {
|
||||
n = want - got;
|
||||
}
|
||||
memcpy(dst + got, v->sig + pos, n);
|
||||
got += n;
|
||||
v->pos += n;
|
||||
continue;
|
||||
}
|
||||
if(pos >= v->hdr_off && pos < hdr_end) {
|
||||
size_t n = (size_t)(hdr_end - pos);
|
||||
|
||||
if(n > want - got) {
|
||||
n = want - got;
|
||||
}
|
||||
memcpy(dst + got, v->hdr + (pos - v->hdr_off), n);
|
||||
got += n;
|
||||
v->pos += n;
|
||||
continue;
|
||||
}
|
||||
/* Everything else is the real archive. The header region can reach past
|
||||
its end, so a read is allowed to stop short here. */
|
||||
if(pos >= v->raw_size) {
|
||||
break;
|
||||
}
|
||||
{
|
||||
Int64 raw_pos = (Int64)pos;
|
||||
size_t n = want - got;
|
||||
const uint64_t avail = v->raw_size - pos;
|
||||
|
||||
if((uint64_t)n > avail) {
|
||||
n = (size_t)avail;
|
||||
}
|
||||
if(v->raw->Seek(v->raw, &raw_pos, SZ_SEEK_SET) != SZ_OK) {
|
||||
return SZ_ERROR_READ;
|
||||
}
|
||||
if(v->raw->Read(v->raw, dst + got, &n) != SZ_OK) {
|
||||
return SZ_ERROR_READ;
|
||||
}
|
||||
if(n == 0) {
|
||||
break;
|
||||
}
|
||||
got += n;
|
||||
v->pos += n;
|
||||
}
|
||||
}
|
||||
*size = got;
|
||||
return SZ_OK;
|
||||
}
|
||||
|
||||
static SRes
|
||||
szh_view_seek(const ISeekInStream *p, Int64 *pos, ESzSeek origin) {
|
||||
szh_view *v = (szh_view *)p;
|
||||
Int64 base;
|
||||
Int64 next;
|
||||
|
||||
switch(origin) {
|
||||
case SZ_SEEK_SET:
|
||||
base = 0;
|
||||
break;
|
||||
case SZ_SEEK_CUR:
|
||||
base = (Int64)v->pos;
|
||||
break;
|
||||
case SZ_SEEK_END:
|
||||
base = (Int64)v->total;
|
||||
break;
|
||||
default:
|
||||
return SZ_ERROR_PARAM;
|
||||
}
|
||||
next = base + *pos;
|
||||
if(next < 0) {
|
||||
next = 0;
|
||||
}
|
||||
v->pos = (uint64_t)next;
|
||||
*pos = next;
|
||||
return SZ_OK;
|
||||
}
|
||||
|
||||
/* ------------------------------------------------------ decoding the header */
|
||||
|
||||
typedef struct {
|
||||
ISeekInStream *raw;
|
||||
uint64_t base;
|
||||
} szh_reader_t;
|
||||
|
||||
static int
|
||||
szh_chain_read(void *ctx, uint64_t offset, void *dst, size_t size) {
|
||||
szh_reader_t *r = (szh_reader_t *)ctx;
|
||||
|
||||
return szh_read_at(r->raw, r->base + offset, dst, size);
|
||||
}
|
||||
|
||||
/* Collects the decrypted header. It grows on demand rather than trusting the
|
||||
declared unpack size, which is attacker controlled. */
|
||||
typedef struct {
|
||||
uint8_t *data;
|
||||
size_t len;
|
||||
size_t cap;
|
||||
int failed;
|
||||
} szh_sink_t;
|
||||
|
||||
static int
|
||||
szh_sink_write(void *ctx, const void *data, size_t size) {
|
||||
szh_sink_t *s = (szh_sink_t *)ctx;
|
||||
|
||||
if(s->failed) {
|
||||
return -1;
|
||||
}
|
||||
if((uint64_t)size > SZH_MAX_HEADER - (uint64_t)s->len) {
|
||||
s->failed = 1;
|
||||
return -1;
|
||||
}
|
||||
if(s->len + size > s->cap) {
|
||||
size_t cap = s->cap ? s->cap : 4096;
|
||||
uint8_t *grown;
|
||||
|
||||
while(cap < s->len + size) {
|
||||
cap *= 2;
|
||||
}
|
||||
grown = (uint8_t *)realloc(s->data, cap);
|
||||
if(!grown) {
|
||||
s->failed = 1;
|
||||
return -1;
|
||||
}
|
||||
s->data = grown;
|
||||
s->cap = cap;
|
||||
}
|
||||
memcpy(s->data + s->len, data, size);
|
||||
s->len += size;
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* ------------------------------------------------------------------ public */
|
||||
|
||||
const char *
|
||||
szh_status_string(szh_status_t status) {
|
||||
switch(status) {
|
||||
case SZH_PLAIN:
|
||||
return "the archive header is readable";
|
||||
case SZH_PATCHED:
|
||||
return "the archive header was decrypted";
|
||||
case SZH_ERR_PASSWORD:
|
||||
return "the archive header is encrypted";
|
||||
case SZH_ERR_UNSUPPORTED:
|
||||
return "the archive header uses an unsupported arrangement";
|
||||
case SZH_ERR_FORMAT:
|
||||
return "the archive header is malformed";
|
||||
case SZH_ERR_IO:
|
||||
return "the archive header could not be read";
|
||||
}
|
||||
return "unknown";
|
||||
}
|
||||
|
||||
static void
|
||||
szh_set_msg(char *msg, size_t msg_size, const char *text, const char *detail) {
|
||||
if(!msg || !msg_size) {
|
||||
return;
|
||||
}
|
||||
if(detail) {
|
||||
snprintf(msg, msg_size, "%s: %s", text, detail);
|
||||
} else {
|
||||
snprintf(msg, msg_size, "%s", text);
|
||||
}
|
||||
}
|
||||
|
||||
szh_status_t
|
||||
szh_prepare(szh_prep **out, ISeekInStream *raw, const char *password,
|
||||
char *msg, size_t msg_size) {
|
||||
uint8_t sig[k7zStartHeaderSize];
|
||||
uint64_t raw_size, next_off, next_size;
|
||||
uint32_t next_crc;
|
||||
uint8_t *raw_hdr = NULL;
|
||||
szh_streams_t ss;
|
||||
sz_chain *chain = NULL;
|
||||
sz_chain_err_t cerr;
|
||||
szh_sink_t sink;
|
||||
szh_reader_t reader;
|
||||
szh_prep *prep = NULL;
|
||||
szh_status_t status = SZH_PLAIN;
|
||||
uint32_t decoded_crc = 0;
|
||||
|
||||
if(out) {
|
||||
*out = NULL;
|
||||
}
|
||||
if(!out || !raw) {
|
||||
return SZH_ERR_IO;
|
||||
}
|
||||
if(msg && msg_size) {
|
||||
msg[0] = 0;
|
||||
}
|
||||
|
||||
/* Sevenz extraction runs this before SzArEx_Open, but the table is what
|
||||
every CRC below needs and generating it twice costs nothing. */
|
||||
CrcGenerateTable();
|
||||
memset(&cerr, 0, sizeof(cerr));
|
||||
memset(&sink, 0, sizeof(sink));
|
||||
|
||||
/* --- the start header ------------------------------------------------ */
|
||||
raw_size = szh_size_of(raw);
|
||||
if(raw_size < k7zStartHeaderSize || szh_read_at(raw, 0, sig, sizeof(sig))) {
|
||||
return SZH_PLAIN;
|
||||
}
|
||||
if(memcmp(sig, k7zSignature, k7zSignatureSize) != 0 || sig[6] != 0) {
|
||||
return SZH_PLAIN;
|
||||
}
|
||||
if(CrcCalc(sig + 12, 20) != szh_le32(sig + 8)) {
|
||||
return SZH_PLAIN;
|
||||
}
|
||||
|
||||
next_off = szh_le64(sig + 12);
|
||||
next_size = szh_le64(sig + 20);
|
||||
next_crc = szh_le32(sig + 28);
|
||||
|
||||
if(next_size == 0 || next_size > SZH_MAX_HEADER) {
|
||||
return SZH_PLAIN;
|
||||
}
|
||||
if(next_off > raw_size || next_size > raw_size - next_off) {
|
||||
return SZH_PLAIN;
|
||||
}
|
||||
|
||||
/* --- the next header ------------------------------------------------- */
|
||||
/* One byte decides it: an ordinary header (k7zIdHeader) is none of our
|
||||
business, and a big uncompressed one is not worth reading twice. */
|
||||
{
|
||||
uint8_t first_byte = 0;
|
||||
|
||||
if(szh_read_at(raw, k7zStartHeaderSize + next_off, &first_byte, 1) ||
|
||||
first_byte != SZH_ID_ENCODED_HEADER) {
|
||||
return SZH_PLAIN;
|
||||
}
|
||||
}
|
||||
|
||||
raw_hdr = (uint8_t *)malloc((size_t)next_size);
|
||||
if(!raw_hdr) {
|
||||
szh_set_msg(msg, msg_size, "out of memory reading the archive header",
|
||||
NULL);
|
||||
return SZH_ERR_IO;
|
||||
}
|
||||
if(szh_read_at(raw, k7zStartHeaderSize + next_off, raw_hdr,
|
||||
(size_t)next_size) ||
|
||||
CrcCalc(raw_hdr, (size_t)next_size) != next_crc) {
|
||||
/* Broken or truncated: let the SDK diagnose it exactly as it always has. */
|
||||
status = SZH_PLAIN;
|
||||
goto done;
|
||||
}
|
||||
if(szh_parse_streams(raw_hdr, (size_t)next_size, &ss)) {
|
||||
status = SZH_PLAIN;
|
||||
goto done;
|
||||
}
|
||||
|
||||
/* --- the folder behind it -------------------------------------------- */
|
||||
if(sz_chain_parse(&chain, ss.blob, ss.blob_size, ss.pack_positions,
|
||||
ss.num_pack, ss.coder_unpack_sizes, ss.unpack_size,
|
||||
sz_chain_default_limits(), &cerr) != 0) {
|
||||
/* Not ours to report: the SDK rejects such an archive too, and it is the
|
||||
one that knows how to describe it. */
|
||||
status = SZH_PLAIN;
|
||||
goto done;
|
||||
}
|
||||
if(!sz_chain_needs_password(chain)) {
|
||||
/* An ordinary compressed header (-mhc=on), which the SDK decodes itself. */
|
||||
status = SZH_PLAIN;
|
||||
goto done;
|
||||
}
|
||||
if(!password || !password[0]) {
|
||||
szh_set_msg(msg, msg_size,
|
||||
"the archive header is encrypted (-mhe=on), so the file names, "
|
||||
"the folder table and the entry sizes are all inside it and the "
|
||||
"archive cannot be listed or unpacked without the password",
|
||||
NULL);
|
||||
status = SZH_ERR_PASSWORD;
|
||||
goto done;
|
||||
}
|
||||
|
||||
reader.raw = raw;
|
||||
reader.base = (uint64_t)k7zStartHeaderSize + ss.pack_pos;
|
||||
if(sz_chain_decode(chain, szh_chain_read, &reader, szh_sink_write, &sink, NULL,
|
||||
NULL, password, &decoded_crc, &cerr) != 0 ||
|
||||
sink.failed) {
|
||||
switch(cerr.status) {
|
||||
case SZ_CHAIN_ERR_PASSWORD:
|
||||
case SZ_CHAIN_ERR_DATA:
|
||||
szh_set_msg(msg, msg_size,
|
||||
"the encrypted archive header did not decrypt: the password "
|
||||
"is wrong, or the archive is damaged",
|
||||
cerr.message);
|
||||
status = SZH_ERR_PASSWORD;
|
||||
break;
|
||||
case SZ_CHAIN_ERR_LIMIT:
|
||||
case SZ_CHAIN_ERR_METHOD:
|
||||
case SZ_CHAIN_ERR_LAYOUT:
|
||||
szh_set_msg(msg, msg_size, "cannot decode the encrypted archive header",
|
||||
cerr.message);
|
||||
status = SZH_ERR_UNSUPPORTED;
|
||||
break;
|
||||
case SZ_CHAIN_ERR_MEM:
|
||||
szh_set_msg(msg, msg_size, "out of memory decoding the archive header",
|
||||
NULL);
|
||||
status = SZH_ERR_IO;
|
||||
break;
|
||||
default:
|
||||
szh_set_msg(msg, msg_size, "cannot decode the encrypted archive header",
|
||||
cerr.message);
|
||||
status = SZH_ERR_FORMAT;
|
||||
break;
|
||||
}
|
||||
goto done;
|
||||
}
|
||||
if(sink.len == 0) {
|
||||
szh_set_msg(msg, msg_size,
|
||||
"the encrypted archive header decrypted to nothing", NULL);
|
||||
status = SZH_ERR_FORMAT;
|
||||
goto done;
|
||||
}
|
||||
if(ss.has_crc && decoded_crc != ss.crc) {
|
||||
szh_set_msg(msg, msg_size,
|
||||
"the encrypted archive header decrypted to data that fails its "
|
||||
"CRC -- the password is wrong",
|
||||
NULL);
|
||||
status = SZH_ERR_PASSWORD;
|
||||
goto done;
|
||||
}
|
||||
if(sink.data[0] != SZH_ID_HEADER) {
|
||||
szh_set_msg(msg, msg_size,
|
||||
"the encrypted archive header did not decrypt to a 7z header "
|
||||
"-- the password is wrong",
|
||||
NULL);
|
||||
status = SZH_ERR_PASSWORD;
|
||||
goto done;
|
||||
}
|
||||
|
||||
/* --- hand the plaintext to the SDK ----------------------------------- */
|
||||
prep = (szh_prep *)calloc(1, sizeof(*prep));
|
||||
if(!prep) {
|
||||
szh_set_msg(msg, msg_size, "out of memory reading the archive header",
|
||||
NULL);
|
||||
status = SZH_ERR_IO;
|
||||
goto done;
|
||||
}
|
||||
prep->view.raw = raw;
|
||||
prep->view.raw_size = raw_size;
|
||||
prep->view.pos = 0;
|
||||
prep->view.hdr_off = (uint64_t)k7zStartHeaderSize + next_off;
|
||||
prep->view.hdr = sink.data;
|
||||
prep->view.hdr_len = sink.len;
|
||||
prep->view.total = prep->view.hdr_off + prep->view.hdr_len;
|
||||
if(prep->view.total < raw_size) {
|
||||
prep->view.total = raw_size;
|
||||
}
|
||||
sink.data = NULL; /* owned by the view from here on */
|
||||
|
||||
memcpy(prep->view.sig, sig, sizeof(prep->view.sig));
|
||||
szh_put_le64(prep->view.sig + 12, prep->view.hdr_off - k7zStartHeaderSize);
|
||||
szh_put_le64(prep->view.sig + 20, prep->view.hdr_len);
|
||||
szh_put_le32(prep->view.sig + 28,
|
||||
CrcCalc(prep->view.hdr, (size_t)prep->view.hdr_len));
|
||||
szh_put_le32(prep->view.sig + 8, CrcCalc(prep->view.sig + 12, 20));
|
||||
|
||||
prep->view.vt.Read = szh_view_read;
|
||||
prep->view.vt.Seek = szh_view_seek;
|
||||
|
||||
*out = prep;
|
||||
prep = NULL;
|
||||
status = SZH_PATCHED;
|
||||
|
||||
done:
|
||||
free(raw_hdr);
|
||||
free(sink.data);
|
||||
sz_chain_free(chain);
|
||||
if(prep) {
|
||||
free(prep->view.hdr);
|
||||
free(prep);
|
||||
}
|
||||
return status;
|
||||
}
|
||||
|
||||
ISeekInStream *
|
||||
szh_stream(const szh_prep *p) {
|
||||
return p ? (ISeekInStream *)&p->view.vt : NULL;
|
||||
}
|
||||
|
||||
void
|
||||
szh_prep_free(szh_prep *p) {
|
||||
if(p) {
|
||||
free(p->view.hdr);
|
||||
free(p);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,106 @@
|
||||
/* sevenz_header -- reads the 7z header, decrypting it when it is encrypted.
|
||||
part of ps5-web-file-manager
|
||||
|
||||
A 7z archive keeps its header at the END of the file, and when that header
|
||||
grows past a threshold 7-Zip stores it *compressed*: the next-header region
|
||||
then starts with a `k7zIdEncodedHeader` (0x17) record describing a single
|
||||
folder whose output is the real header. That folder is one of two things:
|
||||
|
||||
* LZMA / LZMA2 -- `-mhc=on`, the default. The vendored SDK decodes it
|
||||
itself, so this module reads a few bytes, sees no AES coder and steps
|
||||
aside without changing anything.
|
||||
* LZMA + 7zAES -- `-mhe=on`. The C half of the SDK has no 7zAES coder at
|
||||
all, so SzArEx_Open() gives up with SZ_ERROR_UNSUPPORTED before a single
|
||||
folder is known: the archive cannot even be listed.
|
||||
|
||||
The second case is what this module exists for. It decodes that one folder
|
||||
with src/sevenz_chain.c -- the same decoder the archive's content goes
|
||||
through -- and then hands the SDK a stream in which the encoded header has
|
||||
been replaced by its plaintext. The plaintext is longer than the record it
|
||||
replaces, so the stream is a small virtual view over the real one:
|
||||
|
||||
[0, 32) the start header, rewritten to describe the
|
||||
plaintext (offset, size and CRC)
|
||||
[32, hdr_off) the real archive: packed streams
|
||||
[hdr_off, hdr_off + L) the decrypted header
|
||||
beyond that the real archive again
|
||||
|
||||
`hdr_off` is where the encoded header already lived, so no offset that the
|
||||
archive itself stores has to move: the SDK reads the plaintext at exactly
|
||||
the position it expected the encoded record, and the main data position it
|
||||
derives from the plaintext still points at the real packed streams.
|
||||
|
||||
Nothing on disk is touched, the archive is opened read-only, and an archive
|
||||
whose header is not encrypted is never touched at all.
|
||||
*/
|
||||
|
||||
#ifndef SEVENZ_HEADER_H
|
||||
#define SEVENZ_HEADER_H
|
||||
|
||||
#include <stddef.h>
|
||||
|
||||
#include "7zTypes.h"
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
/* The plaintext header is tiny in every real archive (kilobytes); the ceiling
|
||||
only exists so a hostile encoded header cannot ask for a gigabyte. */
|
||||
#define SZH_MAX_HEADER ((uint64_t)64 * 1024 * 1024)
|
||||
|
||||
typedef enum {
|
||||
/* The header is readable as it stands. Use the stream you passed in and
|
||||
let the SDK parse it, exactly as before this module existed. */
|
||||
SZH_PLAIN = 0,
|
||||
|
||||
/* The header was encrypted and is now decrypted: hand szh_stream() to
|
||||
SzArEx_Open() instead of the raw stream. */
|
||||
SZH_PATCHED,
|
||||
|
||||
/* The header is encrypted and the password given was missing or wrong.
|
||||
Actionable: the caller should ask for one and retry. */
|
||||
SZH_ERR_PASSWORD,
|
||||
|
||||
/* The encoded header uses an arrangement this module does not read. */
|
||||
SZH_ERR_UNSUPPORTED,
|
||||
|
||||
/* The encoded header is malformed. */
|
||||
SZH_ERR_FORMAT,
|
||||
|
||||
/* Reading the archive failed. */
|
||||
SZH_ERR_IO
|
||||
} szh_status_t;
|
||||
|
||||
typedef struct szh_prep szh_prep;
|
||||
|
||||
/* Inspects the archive header behind `raw`.
|
||||
|
||||
On SZH_PLAIN *out is NULL and the caller proceeds with `raw` untouched.
|
||||
On SZH_PATCHED *out owns everything and szh_stream(*out) must be used.
|
||||
Otherwise *out is NULL and `msg` says why, in a form meant for the user.
|
||||
|
||||
`password` is the archive password as UTF-8, or NULL / "" when the caller
|
||||
has none. It is only consulted when the header turns out to be encrypted.
|
||||
|
||||
The function never reports an error for an archive the SDK would diagnose
|
||||
better: anything unexpected *before* an AES coder is found -- a short file,
|
||||
a bad signature, a header CRC mismatch, an unparsable StreamsInfo -- comes
|
||||
back as SZH_PLAIN so the SDK keeps producing the message it always did. */
|
||||
szh_status_t szh_prepare(szh_prep **out, ISeekInStream *raw, const char *password,
|
||||
char *msg, size_t msg_size);
|
||||
|
||||
/* The stream to give SzArEx_Open(); NULL when p is NULL. Valid until
|
||||
szh_prep_free(). */
|
||||
ISeekInStream *szh_stream(const szh_prep *p);
|
||||
|
||||
/* Static description of a status, for messages that have no better text. */
|
||||
const char *szh_status_string(szh_status_t status);
|
||||
|
||||
void szh_prep_free(szh_prep *p);
|
||||
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
#endif
|
||||
|
||||
#endif /* SEVENZ_HEADER_H */
|
||||
@@ -0,0 +1,197 @@
|
||||
#include "sevenz_mt.h"
|
||||
|
||||
#include <stdarg.h>
|
||||
#include <stdio.h>
|
||||
#include <string.h>
|
||||
|
||||
#include "7zTypes.h"
|
||||
#include "7zCrc.h"
|
||||
#include "Alloc.h"
|
||||
#include "Lzma2DecMt.h"
|
||||
|
||||
/* ------------------------------------------------------------ adapters --
|
||||
The SDK's decoders speak ISeqInStream / ISeqOutStream / ICompressProgress;
|
||||
the engine speaks plain callbacks. These three structs translate. All of
|
||||
them run on the calling thread -- MtDec only ever hands output to the
|
||||
thread that called Lzma2DecMt_Decode, which is what makes the plain sink
|
||||
safe to reuse here. */
|
||||
|
||||
typedef struct {
|
||||
ISeqInStream vt;
|
||||
sz_chain_read_fn read_at;
|
||||
void *read_ctx;
|
||||
uint64_t pos; /* absolute offset of the next byte to hand out */
|
||||
uint64_t end; /* one past the last byte of the packed stream */
|
||||
} mt_seq_in;
|
||||
|
||||
static SRes mt_seq_read(const ISeqInStream *pp, void *buf, size_t *size) {
|
||||
mt_seq_in *s = (mt_seq_in *)pp;
|
||||
size_t want = *size;
|
||||
|
||||
*size = 0;
|
||||
if(want == 0) {
|
||||
return SZ_OK;
|
||||
}
|
||||
if(s->end - s->pos < (uint64_t)want) {
|
||||
want = (size_t)(s->end - s->pos);
|
||||
}
|
||||
if(want != 0 && s->read_at(s->read_ctx, s->pos, buf, want) != 0) {
|
||||
return SZ_ERROR_READ;
|
||||
}
|
||||
s->pos += want;
|
||||
*size = want;
|
||||
/* A short read means "end of stream" to the SDK; since we only ever hand
|
||||
it exactly in_size bytes, hitting the end early is the caller's bug and
|
||||
the decoder's outSize check will flag it. */
|
||||
return SZ_OK;
|
||||
}
|
||||
|
||||
typedef struct {
|
||||
ISeqOutStream vt;
|
||||
sz_chain_sink_fn sink;
|
||||
void *sink_ctx;
|
||||
uint32_t crc;
|
||||
int failed;
|
||||
} mt_seq_out;
|
||||
|
||||
static size_t mt_seq_write(const ISeqOutStream *pp, const void *buf,
|
||||
size_t size) {
|
||||
mt_seq_out *s = (mt_seq_out *)pp;
|
||||
|
||||
if(s->failed) {
|
||||
return 0;
|
||||
}
|
||||
s->crc = CrcUpdate(s->crc, buf, size);
|
||||
if(s->sink(s->sink_ctx, buf, size) != 0) {
|
||||
s->failed = 1;
|
||||
/* Returning less than `size` tells the SDK the output side is done; it
|
||||
reports SZ_ERROR_WRITE. */
|
||||
return 0;
|
||||
}
|
||||
return size;
|
||||
}
|
||||
|
||||
typedef struct {
|
||||
ICompressProgress vt;
|
||||
sz_chain_cancel_fn cancel;
|
||||
void *cancel_ctx;
|
||||
} mt_progress;
|
||||
|
||||
static SRes mt_progress_report(const ICompressProgress *pp, UInt64 in_size,
|
||||
UInt64 out_size) {
|
||||
mt_progress *s = (mt_progress *)pp;
|
||||
|
||||
(void)in_size;
|
||||
(void)out_size;
|
||||
if(s->cancel && s->cancel(s->cancel_ctx)) {
|
||||
return SZ_ERROR_PROGRESS;
|
||||
}
|
||||
return SZ_OK;
|
||||
}
|
||||
|
||||
/* --------------------------------------------------------------- decode --
|
||||
Errors are mapped onto sz_chain_err_t so the facade's reporting stays in
|
||||
one vocabulary. SZX_MT_ERR_THREADS is reserved for "the platform cannot
|
||||
give me a thread pool": Lzma2DecMt returns SZ_ERROR_THREAD only from its
|
||||
threading primitives, everything else is data or memory. */
|
||||
|
||||
static void mt_fail(sz_chain_err_t *err, sz_chain_status_t status,
|
||||
const char *fmt, ...) {
|
||||
va_list ap;
|
||||
|
||||
if(!err) {
|
||||
return;
|
||||
}
|
||||
err->status = status;
|
||||
err->coder = 0;
|
||||
err->method = 0x21; /* SZ_M_LZMA2; kept literal to avoid dragging chain.c in */
|
||||
err->offset = 0;
|
||||
va_start(ap, fmt);
|
||||
vsnprintf(err->message, sizeof(err->message), fmt, ap);
|
||||
va_end(ap);
|
||||
}
|
||||
|
||||
int szx_mt_decode(sz_chain_read_fn read_at, void *read_ctx, uint64_t in_offset,
|
||||
uint64_t in_size, uint8_t prop, uint64_t out_size,
|
||||
sz_chain_sink_fn sink, void *sink_ctx,
|
||||
sz_chain_cancel_fn cancel, void *cancel_ctx,
|
||||
uint32_t *crc_out, sz_chain_err_t *err) {
|
||||
mt_seq_in in;
|
||||
mt_seq_out out;
|
||||
mt_progress progress;
|
||||
CLzma2DecMtProps props;
|
||||
CLzma2DecMtHandle mt;
|
||||
UInt64 in_processed = 0;
|
||||
int is_mt = 0;
|
||||
SRes res;
|
||||
|
||||
memset(&in, 0, sizeof(in));
|
||||
in.vt.Read = mt_seq_read;
|
||||
in.read_at = read_at;
|
||||
in.read_ctx = read_ctx;
|
||||
in.pos = in_offset;
|
||||
in.end = in_offset + in_size;
|
||||
|
||||
memset(&out, 0, sizeof(out));
|
||||
out.vt.Write = mt_seq_write;
|
||||
out.sink = sink;
|
||||
out.sink_ctx = sink_ctx;
|
||||
out.crc = CRC_INIT_VAL;
|
||||
|
||||
memset(&progress, 0, sizeof(progress));
|
||||
progress.vt.Progress = mt_progress_report;
|
||||
progress.cancel = cancel;
|
||||
progress.cancel_ctx = cancel_ctx;
|
||||
|
||||
Lzma2DecMtProps_Init(&props);
|
||||
props.numThreads = SZX_MT_THREADS;
|
||||
props.inBufSize_MT = 1 << 20;
|
||||
|
||||
mt = Lzma2DecMt_Create(&g_Alloc, &g_MidAlloc);
|
||||
if(!mt) {
|
||||
mt_fail(err, SZ_CHAIN_ERR_INTERNAL, "LZMA2 MT: out of memory");
|
||||
return -1;
|
||||
}
|
||||
res = Lzma2DecMt_Decode(mt, prop, &props, &out.vt, &out_size, 1, &in.vt,
|
||||
&in_processed, &is_mt,
|
||||
cancel ? &progress.vt : NULL);
|
||||
Lzma2DecMt_Destroy(mt);
|
||||
|
||||
if(res == SZ_ERROR_THREAD) {
|
||||
/* No usable thread pool (pthread init failure, thread creation denied).
|
||||
The caller retries on the single-threaded chain path. */
|
||||
return SZX_MT_ERR_THREADS;
|
||||
}
|
||||
if(out.failed) {
|
||||
mt_fail(err, SZ_CHAIN_ERR_WRITE, "LZMA2 MT: sink rejected decoded data");
|
||||
return -1;
|
||||
}
|
||||
if(res != SZ_OK) {
|
||||
switch(res) {
|
||||
case SZ_ERROR_PROGRESS:
|
||||
mt_fail(err, SZ_CHAIN_ERR_CANCELED, "canceled");
|
||||
break;
|
||||
case SZ_ERROR_MEM:
|
||||
mt_fail(err, SZ_CHAIN_ERR_INTERNAL, "LZMA2 MT: out of memory");
|
||||
break;
|
||||
case SZ_ERROR_WRITE:
|
||||
mt_fail(err, SZ_CHAIN_ERR_WRITE, "LZMA2 MT: output stream failed");
|
||||
break;
|
||||
default:
|
||||
mt_fail(err, SZ_CHAIN_ERR_DATA, "LZMA2 MT: decode failed (res=%d)",
|
||||
(int)res);
|
||||
break;
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
if(in_processed != in_size) {
|
||||
mt_fail(err, SZ_CHAIN_ERR_DATA,
|
||||
"LZMA2 MT: consumed %llu of %llu packed bytes",
|
||||
(unsigned long long)in_processed, (unsigned long long)in_size);
|
||||
return -1;
|
||||
}
|
||||
if(crc_out) {
|
||||
*crc_out = out.crc;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,40 @@
|
||||
/* Multithreaded LZMA2 decode for the 7z engine.
|
||||
*
|
||||
* Most 7z archives are a single plain LZMA2 coder (7-Zip's -m0=lzma2
|
||||
* default). For that shape the SDK's own parallel decoder -- the same code
|
||||
* 7-Zip runs for -mmt -- replaces the single-threaded chain walk and decodes
|
||||
* consecutive LZMA2 blocks on worker threads while the main thread streams
|
||||
* the output into the staging sink. Measured on a 329 MiB fixture this is
|
||||
* worth ~1.7x on an 8-core host, on top of the assembly kernel.
|
||||
*
|
||||
* Threads are rented, not owned: any thread error falls back to the caller's
|
||||
* single-threaded path, so a platform without working pthreads only ever
|
||||
* loses speed, never correctness. */
|
||||
|
||||
#ifndef SEVENZ_MT_H
|
||||
#define SEVENZ_MT_H
|
||||
|
||||
#include "sevenz_chain.h"
|
||||
|
||||
/* 8-core Zen 2 on the PS5: 4 decoders leave the HTTP server, the task
|
||||
system and the kernel half of the machine. */
|
||||
#define SZX_MT_THREADS 8
|
||||
|
||||
/* Decodes one folder that sz_chain_lzma2_root() has recognised. The
|
||||
callbacks mirror sz_chain_decode()'s: read_at/ctx for the packed data,
|
||||
sink/ctx for the decoded bytes (both run on the calling thread; the sink
|
||||
sees the same ordered byte stream the chain would have produced).
|
||||
cancel/ctx is polled from the progress callback and may be NULL.
|
||||
crc_out, when not NULL, receives the CRC-32 of the delivered bytes.
|
||||
err, when not NULL, receives a chain-style error description.
|
||||
Returns 0 on success; SZX_MT_ERR_THREADS means "no working thread pool"
|
||||
and the caller should retry single-threaded; other failures are terminal. */
|
||||
#define SZX_MT_ERR_THREADS 2
|
||||
|
||||
int szx_mt_decode(sz_chain_read_fn read_at, void *read_ctx,
|
||||
uint64_t in_offset, uint64_t in_size, uint8_t prop,
|
||||
uint64_t out_size, sz_chain_sink_fn sink, void *sink_ctx,
|
||||
sz_chain_cancel_fn cancel, void *cancel_ctx,
|
||||
uint32_t *crc_out, sz_chain_err_t *err);
|
||||
|
||||
#endif /* SEVENZ_MT_H */
|
||||
@@ -0,0 +1,356 @@
|
||||
/* sevenz_volstream -- see sevenz_volstream.h for what this does and why. */
|
||||
|
||||
#include <stdarg.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
|
||||
#if defined(_WIN32)
|
||||
#include <wchar.h>
|
||||
#endif
|
||||
|
||||
#include "7zFile.h"
|
||||
|
||||
#include "sevenz_volstream.h"
|
||||
#include "zipx_volume.h"
|
||||
|
||||
struct sevenz_volstream {
|
||||
ISeekInStream vt;
|
||||
|
||||
zipx_volume_t vol; /* owns the ordered paths */
|
||||
CSzFile *file; /* one per part */
|
||||
uint64_t *start; /* count + 1 prefix offsets into the logical archive */
|
||||
uint32_t count;
|
||||
uint32_t open_count; /* how many entries of `file` were opened */
|
||||
uint64_t pos; /* current offset in the logical archive */
|
||||
uint32_t cur; /* part `pos` currently sits in, to skip redundant seeks */
|
||||
uint64_t cur_off; /* file offset within that part */
|
||||
char name[256]; /* stem of the set, for messages */
|
||||
};
|
||||
|
||||
/* ---------------------------------------------------------------- helpers */
|
||||
|
||||
static char *err_printf(const char *fmt, ...) {
|
||||
va_list ap;
|
||||
char buf[512];
|
||||
char *out;
|
||||
|
||||
va_start(ap, fmt);
|
||||
vsnprintf(buf, sizeof(buf), fmt, ap);
|
||||
va_end(ap);
|
||||
|
||||
out = (char *)malloc(strlen(buf) + 1);
|
||||
if(out) memcpy(out, buf, strlen(buf) + 1);
|
||||
return out;
|
||||
}
|
||||
|
||||
static const char *file_base(const char *path) {
|
||||
const char *slash = strrchr(path, '/');
|
||||
const char *back = strrchr(path, '\\');
|
||||
|
||||
if(back && (!slash || back > slash)) slash = back;
|
||||
return slash ? slash + 1 : path;
|
||||
}
|
||||
|
||||
#if defined(_WIN32)
|
||||
|
||||
/* The SDK opens through CreateFileA otherwise, which cannot see non-ASCII
|
||||
entry names. */
|
||||
static void utf8_to_utf16(const char *src, WCHAR *dst, size_t cap) {
|
||||
size_t out = 0;
|
||||
|
||||
while(*src && out + 2 < cap) {
|
||||
unsigned char c = (unsigned char)*src++;
|
||||
UInt32 cp;
|
||||
|
||||
if(c < 0x80) {
|
||||
cp = c;
|
||||
} else if((c & 0xE0) == 0xC0 && (src[0] & 0xC0) == 0x80) {
|
||||
cp = ((UInt32)(c & 0x1F) << 6) | (UInt32)(*src++ & 0x3F);
|
||||
} else if((c & 0xF0) == 0xE0 && (src[0] & 0xC0) == 0x80 &&
|
||||
(src[1] & 0xC0) == 0x80) {
|
||||
cp = ((UInt32)(c & 0x0F) << 12) | ((UInt32)(src[0] & 0x3F) << 6) |
|
||||
(UInt32)(src[1] & 0x3F);
|
||||
src += 2;
|
||||
} else if((c & 0xF8) == 0xF0 && (src[0] & 0xC0) == 0x80 &&
|
||||
(src[1] & 0xC0) == 0x80 && (src[2] & 0xC0) == 0x80) {
|
||||
cp = ((UInt32)(c & 0x07) << 18) | ((UInt32)(src[0] & 0x3F) << 12) |
|
||||
((UInt32)(src[1] & 0x3F) << 6) | (UInt32)(src[2] & 0x3F);
|
||||
src += 3;
|
||||
} else {
|
||||
cp = '?';
|
||||
}
|
||||
|
||||
if(cp >= 0x10000) {
|
||||
cp -= 0x10000;
|
||||
dst[out++] = (WCHAR)(0xD800 | (cp >> 10));
|
||||
dst[out++] = (WCHAR)(0xDC00 | (cp & 0x3FF));
|
||||
} else {
|
||||
dst[out++] = (WCHAR)cp;
|
||||
}
|
||||
}
|
||||
dst[out] = 0;
|
||||
}
|
||||
|
||||
static int open_part(CSzFile *file, const char *path) {
|
||||
WCHAR wide[4096];
|
||||
|
||||
utf8_to_utf16(path, wide, sizeof(wide) / sizeof(wide[0]));
|
||||
return InFile_OpenW(file, wide) == 0 ? 0 : -1;
|
||||
}
|
||||
|
||||
#else
|
||||
|
||||
static int open_part(CSzFile *file, const char *path) {
|
||||
return InFile_Open(file, path) == 0 ? 0 : -1;
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
/* ----------------------------------------------------------------- stream */
|
||||
|
||||
static uint32_t part_at(const sevenz_volstream *v, uint64_t pos) {
|
||||
uint32_t i;
|
||||
|
||||
for(i = 0; i < v->count; i++) {
|
||||
if(pos < v->start[i + 1]) return i;
|
||||
}
|
||||
return v->count;
|
||||
}
|
||||
|
||||
static SRes vol_read(const ISeekInStream *p, void *buf, size_t *size) {
|
||||
sevenz_volstream *v = (sevenz_volstream *)p;
|
||||
uint8_t *dst = (uint8_t *)buf;
|
||||
size_t want = *size;
|
||||
size_t got = 0;
|
||||
|
||||
*size = 0;
|
||||
while(got < want) {
|
||||
uint32_t i = part_at(v, v->pos);
|
||||
uint64_t avail, off;
|
||||
size_t take;
|
||||
|
||||
if(i >= v->count) break; /* end of the set: a short read means EOF */
|
||||
|
||||
off = v->pos - v->start[i];
|
||||
avail = (v->start[i + 1] - v->start[i]) - off;
|
||||
take = (size_t)(avail < (uint64_t)(want - got) ? avail
|
||||
: (uint64_t)(want - got));
|
||||
if(take == 0) break;
|
||||
|
||||
if(v->cur != i || v->cur_off != off) {
|
||||
Int64 seek = (Int64)off;
|
||||
if(File_Seek(&v->file[i], &seek, SZ_SEEK_SET) != 0) return SZ_ERROR_READ;
|
||||
v->cur = i;
|
||||
v->cur_off = off;
|
||||
}
|
||||
|
||||
{
|
||||
size_t part_got = take;
|
||||
if(File_Read(&v->file[i], dst + got, &part_got) != 0) return SZ_ERROR_READ;
|
||||
if(part_got == 0) break;
|
||||
got += part_got;
|
||||
v->pos += part_got;
|
||||
v->cur_off += part_got;
|
||||
if(part_got < take) break;
|
||||
}
|
||||
}
|
||||
|
||||
*size = got;
|
||||
return SZ_OK;
|
||||
}
|
||||
|
||||
static SRes vol_seek(const ISeekInStream *p, Int64 *pos, ESzSeek origin) {
|
||||
sevenz_volstream *v = (sevenz_volstream *)p;
|
||||
uint64_t total = v->start[v->count];
|
||||
uint64_t target;
|
||||
|
||||
switch(origin) {
|
||||
case SZ_SEEK_SET:
|
||||
if(*pos < 0) return SZ_ERROR_PARAM;
|
||||
target = (uint64_t)*pos;
|
||||
break;
|
||||
case SZ_SEEK_CUR:
|
||||
if(*pos < 0) {
|
||||
UInt64 back = (UInt64)(-*pos);
|
||||
if(back > v->pos) return SZ_ERROR_PARAM;
|
||||
target = v->pos - back;
|
||||
} else {
|
||||
target = v->pos + (UInt64)*pos;
|
||||
}
|
||||
break;
|
||||
case SZ_SEEK_END:
|
||||
if(*pos < 0) {
|
||||
UInt64 back = (UInt64)(-*pos);
|
||||
if(back > total) return SZ_ERROR_PARAM;
|
||||
target = total - back;
|
||||
} else {
|
||||
target = total + (UInt64)*pos;
|
||||
}
|
||||
break;
|
||||
default:
|
||||
return SZ_ERROR_PARAM;
|
||||
}
|
||||
|
||||
if(target > total) target = total;
|
||||
v->pos = target;
|
||||
*pos = (Int64)target;
|
||||
return SZ_OK;
|
||||
}
|
||||
|
||||
/* -------------------------------------------------------------------- open */
|
||||
|
||||
int sevenz_volstream_open(sevenz_volstream **out, const char *path, int *is_set,
|
||||
char **err) {
|
||||
sevenz_volstream *v;
|
||||
char *vol_err = NULL;
|
||||
int detected;
|
||||
uint32_t i;
|
||||
|
||||
if(out) *out = NULL;
|
||||
if(is_set) *is_set = 0;
|
||||
if(err) *err = NULL;
|
||||
if(!out || !path) {
|
||||
if(err) *err = err_printf("no archive path given");
|
||||
return -1;
|
||||
}
|
||||
|
||||
v = (sevenz_volstream *)calloc(1, sizeof(*v));
|
||||
if(!v) {
|
||||
if(err) *err = err_printf("out of memory");
|
||||
return -1;
|
||||
}
|
||||
|
||||
/* One call does both jobs: it either reports "ordinary file" and clears the
|
||||
struct, or fills in the ordered part list. A -1 here already carries the
|
||||
message the user needs (a hole in the numbering names the missing part). */
|
||||
detected = zipx_volume_detect(path, &v->vol, &vol_err);
|
||||
if(detected < 0) {
|
||||
if(err) {
|
||||
*err = vol_err ? vol_err
|
||||
: err_printf("'%s' cannot be read", file_base(path));
|
||||
} else {
|
||||
free(vol_err);
|
||||
}
|
||||
zipx_volume_free(&v->vol);
|
||||
free(v);
|
||||
return -1;
|
||||
}
|
||||
|
||||
if(detected == 0) {
|
||||
v->vol.paths = (char **)malloc(sizeof(char *));
|
||||
if(v->vol.paths) v->vol.paths[0] = (char *)malloc(strlen(path) + 1);
|
||||
if(!v->vol.paths || !v->vol.paths[0]) {
|
||||
free(v->vol.paths);
|
||||
free(v);
|
||||
if(err) *err = err_printf("out of memory");
|
||||
return -1;
|
||||
}
|
||||
memcpy(v->vol.paths[0], path, strlen(path) + 1);
|
||||
v->vol.count = 1;
|
||||
v->vol.mode = ZIPX_VOL_MODE_CONCAT;
|
||||
v->vol.is_set = 0;
|
||||
} else if(is_set) {
|
||||
*is_set = 1;
|
||||
}
|
||||
|
||||
snprintf(v->name, sizeof(v->name), "%s", file_base(path));
|
||||
|
||||
v->count = (uint32_t)v->vol.count;
|
||||
v->file = (CSzFile *)calloc(v->count, sizeof(CSzFile));
|
||||
v->start = (uint64_t *)calloc((size_t)v->count + 1, sizeof(uint64_t));
|
||||
if(!v->file || !v->start) {
|
||||
if(err) *err = err_printf("out of memory");
|
||||
goto fail;
|
||||
}
|
||||
|
||||
for(i = 0; i < v->count; i++) {
|
||||
UInt64 length = 0;
|
||||
|
||||
File_Construct(&v->file[i]);
|
||||
v->open_count = i + 1; /* File_Close() ignores a never-opened handle */
|
||||
if(open_part(&v->file[i], v->vol.paths[i]) != 0) {
|
||||
if(err) {
|
||||
*err = err_printf("cannot open volume '%s' of '%s'",
|
||||
file_base(v->vol.paths[i]), v->name);
|
||||
}
|
||||
goto fail;
|
||||
}
|
||||
if(File_GetLength(&v->file[i], &length) != 0) {
|
||||
if(err) {
|
||||
*err = err_printf("cannot measure volume '%s' of '%s'",
|
||||
file_base(v->vol.paths[i]), v->name);
|
||||
}
|
||||
goto fail;
|
||||
}
|
||||
v->start[i + 1] = v->start[i] + length;
|
||||
}
|
||||
|
||||
if(v->start[v->count] == 0) {
|
||||
if(err) *err = err_printf("'%s' is empty", v->name);
|
||||
goto fail;
|
||||
}
|
||||
|
||||
/* 7-Zip cuts equal sized parts and lets only the last one be short. A part
|
||||
of a different size in the middle means the set is damaged or was mixed
|
||||
with another one, and decoding would fail much later with a message that
|
||||
points nowhere useful. */
|
||||
for(i = 0; i + 1 < v->count; i++) {
|
||||
uint64_t size = v->start[i + 1] - v->start[i];
|
||||
if(size != v->start[1]) {
|
||||
if(err) {
|
||||
*err = err_printf("volume '%s' of '%s' is %llu bytes, but the earlier "
|
||||
"volumes are %llu bytes: the set is not a clean split",
|
||||
file_base(v->vol.paths[i]), v->name,
|
||||
(unsigned long long)size,
|
||||
(unsigned long long)v->start[1]);
|
||||
}
|
||||
goto fail;
|
||||
}
|
||||
}
|
||||
|
||||
v->vt.Read = vol_read;
|
||||
v->vt.Seek = vol_seek;
|
||||
*out = v;
|
||||
return 0;
|
||||
|
||||
fail:
|
||||
sevenz_volstream_free(v);
|
||||
return -1;
|
||||
}
|
||||
|
||||
ISeekInStream *sevenz_volstream_stream(sevenz_volstream *v) {
|
||||
return v ? &v->vt : NULL;
|
||||
}
|
||||
|
||||
uint32_t sevenz_volstream_count(const sevenz_volstream *v) {
|
||||
return v ? v->count : 0;
|
||||
}
|
||||
|
||||
uint64_t sevenz_volstream_size(const sevenz_volstream *v) {
|
||||
return v ? v->start[v->count] : 0;
|
||||
}
|
||||
|
||||
const char *sevenz_volstream_describe(const sevenz_volstream *v, char *buf,
|
||||
unsigned size) {
|
||||
if(!buf || size == 0) return buf;
|
||||
if(!v) {
|
||||
snprintf(buf, size, "no archive");
|
||||
} else if(v->count <= 1) {
|
||||
snprintf(buf, size, "%s", v->name);
|
||||
} else {
|
||||
snprintf(buf, size, "%s (%u volumes)", v->name, (unsigned)v->count);
|
||||
}
|
||||
return buf;
|
||||
}
|
||||
|
||||
void sevenz_volstream_free(sevenz_volstream *v) {
|
||||
uint32_t i;
|
||||
|
||||
if(!v) return;
|
||||
for(i = 0; i < v->open_count; i++) File_Close(&v->file[i]);
|
||||
free(v->file);
|
||||
free(v->start);
|
||||
zipx_volume_free(&v->vol);
|
||||
free(v);
|
||||
}
|
||||
@@ -0,0 +1,59 @@
|
||||
/* sevenz_volstream -- present a multi-file 7z volume set as one stream.
|
||||
part of ps5-web-file-manager
|
||||
|
||||
A split 7z is a plain byte split: `name.7z.001`, `name.7z.002`, ... are
|
||||
consecutive slices of one archive, so byte N of the logical archive is byte
|
||||
N of the concatenation and every offset stored inside the stream header is
|
||||
already absolute. Nothing has to be merged on disk -- a 160 GiB set would
|
||||
otherwise need a second 160 GiB scratch copy.
|
||||
|
||||
The LZMA SDK reads through ISeekInStream, so this module implements that
|
||||
interface over the ordered part list produced by zipx_volume. The ordered
|
||||
list is what makes a set with a hole in it fail loudly instead of decoding
|
||||
garbage: zipx_volume names the missing part. */
|
||||
|
||||
#ifndef SEVENZ_VOLSTREAM_H
|
||||
#define SEVENZ_VOLSTREAM_H
|
||||
|
||||
#include <stdint.h>
|
||||
|
||||
#include "7zTypes.h"
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
typedef struct sevenz_volstream sevenz_volstream;
|
||||
|
||||
/* Opens `path` together with every volume of the set it belongs to and exposes
|
||||
them as one seekable byte stream. `path` may be any member of the set; an
|
||||
ordinary single-file archive is the degenerate one-file case.
|
||||
|
||||
Returns 0 on success, with *is_set set to 1 when the path was part of a
|
||||
multi-file set (non-NULL only). Returns -1 on failure and, when `err` is
|
||||
non-NULL, stores a malloc'd message the caller must free -- an incomplete
|
||||
set reports the missing volume by name. */
|
||||
int sevenz_volstream_open(sevenz_volstream **out, const char *path, int *is_set,
|
||||
char **err);
|
||||
|
||||
/* The stream to hand to SzArEx_Open(); valid until sevenz_volstream_free(). */
|
||||
ISeekInStream *sevenz_volstream_stream(sevenz_volstream *v);
|
||||
|
||||
/* Number of files backing the stream (1 for an ordinary archive). */
|
||||
uint32_t sevenz_volstream_count(const sevenz_volstream *v);
|
||||
|
||||
/* Size of the whole logical archive. */
|
||||
uint64_t sevenz_volstream_size(const sevenz_volstream *v);
|
||||
|
||||
/* Human readable description, e.g. "name.7z (3 volumes)"; writes into buf and
|
||||
returns buf. */
|
||||
const char *sevenz_volstream_describe(const sevenz_volstream *v, char *buf,
|
||||
unsigned size);
|
||||
|
||||
void sevenz_volstream_free(sevenz_volstream *v);
|
||||
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
#endif
|
||||
|
||||
#endif
|
||||
@@ -0,0 +1,128 @@
|
||||
#include "filemgr_internal.h"
|
||||
|
||||
#include <limits.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <sys/stat.h>
|
||||
#include <sys/statvfs.h>
|
||||
#ifdef __SCE__
|
||||
#include <sys/mount.h>
|
||||
#endif
|
||||
|
||||
#include "json_util.h"
|
||||
#include "path_util.h"
|
||||
|
||||
static unsigned long long
|
||||
vfs_bytes(fsblkcnt_t blocks, unsigned long block_size) {
|
||||
return (unsigned long long)blocks * (unsigned long long)block_size;
|
||||
}
|
||||
|
||||
#ifdef __SCE__
|
||||
static int
|
||||
path_is_mounted(const char *path, const struct stat *st) {
|
||||
char parent[PATH_MAX];
|
||||
struct stat parent_st;
|
||||
|
||||
if(!strcmp(path, "/")) {
|
||||
return 1;
|
||||
}
|
||||
if(path_dirname(path, parent, sizeof(parent)) || stat(parent, &parent_st)) {
|
||||
return 0;
|
||||
}
|
||||
return st->st_dev != parent_st.st_dev;
|
||||
}
|
||||
#endif
|
||||
|
||||
enum MHD_Result
|
||||
api_space(struct MHD_Connection *conn) {
|
||||
#ifdef __SCE__
|
||||
static const struct {
|
||||
const char *label_key;
|
||||
const char *path;
|
||||
} mounts[] = {
|
||||
{"storageInternal", "/data"},
|
||||
{"storageUsb", "/mnt/usb0"},
|
||||
{"storageUsb", "/mnt/usb1"},
|
||||
{"storageUsb", "/mnt/usb2"},
|
||||
{"storageUsb", "/mnt/usb3"},
|
||||
{"storageUsb", "/mnt/usb4"},
|
||||
{"storageUsb", "/mnt/usb5"},
|
||||
{"storageUsb", "/mnt/usb6"},
|
||||
{"storageUsb", "/mnt/usb7"},
|
||||
{"storageM2", "/mnt/ext1"},
|
||||
{"storageExtended", "/mnt/ext0"},
|
||||
};
|
||||
#endif
|
||||
char *current = fs_path_value(query_value(conn, "path"));
|
||||
strbuf_t b = {0};
|
||||
int first = 1;
|
||||
|
||||
strbuf_append(&b, "{\"ok\":true,\"spaces\":[");
|
||||
#ifdef __SCE__
|
||||
for(size_t i = 0; i < sizeof(mounts) / sizeof(mounts[0]); i++) {
|
||||
struct statvfs vfs;
|
||||
struct stat st;
|
||||
unsigned long block_size;
|
||||
unsigned long long free_bytes;
|
||||
unsigned long long total_bytes;
|
||||
int is_current = 0;
|
||||
|
||||
if(stat(mounts[i].path, &st) || !path_is_mounted(mounts[i].path, &st) ||
|
||||
statvfs(mounts[i].path, &vfs)) {
|
||||
continue;
|
||||
}
|
||||
block_size = vfs.f_frsize ? vfs.f_frsize : vfs.f_bsize;
|
||||
free_bytes = vfs_bytes(vfs.f_bavail, block_size);
|
||||
total_bytes = vfs_bytes(vfs.f_blocks, block_size);
|
||||
if(current) {
|
||||
if(!strcmp(mounts[i].path, "/")) {
|
||||
is_current = !strcmp(current, "/");
|
||||
} else if(!strncmp(current, mounts[i].path, strlen(mounts[i].path)) &&
|
||||
(current[strlen(mounts[i].path)] == 0 ||
|
||||
current[strlen(mounts[i].path)] == '/')) {
|
||||
is_current = 1;
|
||||
}
|
||||
}
|
||||
|
||||
if(!first) {
|
||||
strbuf_append(&b, ",");
|
||||
}
|
||||
first = 0;
|
||||
strbuf_append(&b, "{\"label_key\":");
|
||||
json_escape(&b, mounts[i].label_key);
|
||||
strbuf_append(&b, ",\"path\":");
|
||||
json_escape(&b, mounts[i].path);
|
||||
strbuf_printf(&b, ",\"free\":%llu,\"total\":%llu,\"current\":%s}",
|
||||
free_bytes, total_bytes, is_current ? "true" : "false");
|
||||
}
|
||||
#else
|
||||
const char *paths[] = {"/", current && strcmp(current, "/") ? current : NULL};
|
||||
const char *labels[] = {"storageRoot", "storageCurrent"};
|
||||
for(size_t i = 0; i < sizeof(paths) / sizeof(paths[0]); i++) {
|
||||
struct statvfs vfs;
|
||||
unsigned long block_size;
|
||||
unsigned long long free_bytes;
|
||||
unsigned long long total_bytes;
|
||||
|
||||
if(!paths[i] || statvfs(paths[i], &vfs)) {
|
||||
continue;
|
||||
}
|
||||
block_size = vfs.f_frsize ? vfs.f_frsize : vfs.f_bsize;
|
||||
free_bytes = vfs_bytes(vfs.f_bavail, block_size);
|
||||
total_bytes = vfs_bytes(vfs.f_blocks, block_size);
|
||||
if(!first) {
|
||||
strbuf_append(&b, ",");
|
||||
}
|
||||
first = 0;
|
||||
strbuf_append(&b, "{\"label_key\":");
|
||||
json_escape(&b, labels[i]);
|
||||
strbuf_append(&b, ",\"path\":");
|
||||
json_escape(&b, paths[i]);
|
||||
strbuf_printf(&b, ",\"free\":%llu,\"total\":%llu,\"current\":%s}",
|
||||
free_bytes, total_bytes, i ? "true" : "false");
|
||||
}
|
||||
#endif
|
||||
free(current);
|
||||
strbuf_append(&b, "]}");
|
||||
return send_buffer(conn, MHD_HTTP_OK, b.data, "application/json");
|
||||
}
|
||||
@@ -0,0 +1,221 @@
|
||||
#include <errno.h>
|
||||
#include <pthread.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <time.h>
|
||||
|
||||
#include "filemgr_internal.h"
|
||||
#include "path_util.h"
|
||||
|
||||
#define ETA_AVERAGE_WINDOW_SECONDS 30
|
||||
|
||||
pthread_mutex_t g_tasks_lock = PTHREAD_MUTEX_INITIALIZER;
|
||||
file_task_t *g_tasks = NULL;
|
||||
unsigned long g_next_task_id = 1;
|
||||
|
||||
const char *
|
||||
task_op_name(task_op_t op) {
|
||||
switch(op) {
|
||||
case TASK_COPY: return "copy";
|
||||
case TASK_MOVE: return "move";
|
||||
case TASK_DELETE: return "delete";
|
||||
case TASK_CHMOD: return "chmod";
|
||||
case TASK_DOWNLOAD: return "download";
|
||||
case TASK_UPLOAD: return "upload";
|
||||
case TASK_PKG_INSTALL: return "pkg_install";
|
||||
case TASK_EXTRACT: return "extract";
|
||||
default: return "unknown";
|
||||
}
|
||||
}
|
||||
|
||||
const char *
|
||||
task_state_name(task_state_t state) {
|
||||
switch(state) {
|
||||
case TASK_QUEUED: return "queued";
|
||||
case TASK_RUNNING: return "running";
|
||||
case TASK_DONE: return "done";
|
||||
case TASK_FAILED: return "failed";
|
||||
case TASK_CANCELED: return "canceled";
|
||||
default: return "unknown";
|
||||
}
|
||||
}
|
||||
|
||||
int
|
||||
task_is_active(const file_task_t *task) {
|
||||
return task->state == TASK_QUEUED || task->state == TASK_RUNNING;
|
||||
}
|
||||
|
||||
int
|
||||
has_active_task_locked(void) {
|
||||
file_task_t *task;
|
||||
|
||||
for(task = g_tasks; task; task = task->next) {
|
||||
if(task->op != TASK_PKG_INSTALL && task_is_active(task)) {
|
||||
return 1;
|
||||
}
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
int
|
||||
has_active_task(void) {
|
||||
int active;
|
||||
|
||||
pthread_mutex_lock(&g_tasks_lock);
|
||||
active = has_active_task_locked();
|
||||
pthread_mutex_unlock(&g_tasks_lock);
|
||||
return active;
|
||||
}
|
||||
|
||||
void
|
||||
free_task(file_task_t *task) {
|
||||
if(!task) {
|
||||
return;
|
||||
}
|
||||
free_paths(task->srcs, task->src_count);
|
||||
free(task);
|
||||
}
|
||||
|
||||
void
|
||||
remove_finished_tasks_locked(void) {
|
||||
file_task_t **link = &g_tasks;
|
||||
|
||||
while(*link) {
|
||||
file_task_t *task = *link;
|
||||
|
||||
if(task_is_active(task) || task->active_streams ||
|
||||
(task->op == TASK_PKG_INSTALL && !task->reported)) {
|
||||
link = &task->next;
|
||||
continue;
|
||||
}
|
||||
|
||||
*link = task->next;
|
||||
free_task(task);
|
||||
}
|
||||
}
|
||||
|
||||
int
|
||||
task_cancel_requested(file_task_t *task) {
|
||||
int cancel;
|
||||
|
||||
pthread_mutex_lock(&g_tasks_lock);
|
||||
cancel = task->cancel_requested;
|
||||
pthread_mutex_unlock(&g_tasks_lock);
|
||||
if(cancel) {
|
||||
errno = ECANCELED;
|
||||
}
|
||||
return cancel;
|
||||
}
|
||||
|
||||
file_task_t *
|
||||
find_task_locked(unsigned long id) {
|
||||
file_task_t *task;
|
||||
|
||||
for(task = g_tasks; task; task = task->next) {
|
||||
if(task->id == id) {
|
||||
return task;
|
||||
}
|
||||
}
|
||||
return NULL;
|
||||
}
|
||||
|
||||
static long long
|
||||
timespec_delta_ns(const struct timespec *end, const struct timespec *start) {
|
||||
return (long long)(end->tv_sec - start->tv_sec) * 1000000000LL +
|
||||
(long long)(end->tv_nsec - start->tv_nsec);
|
||||
}
|
||||
|
||||
static void
|
||||
task_update_eta_locked(file_task_t *task, const struct timespec *now_mono) {
|
||||
task_eta_sample_t *sample;
|
||||
task_eta_sample_t *base = NULL;
|
||||
unsigned int i;
|
||||
|
||||
if(!task->total || !task->done || task->done >= task->total) {
|
||||
task->eta = 0;
|
||||
return;
|
||||
}
|
||||
|
||||
sample = &task->eta_samples[task->eta_sample_next];
|
||||
sample->done = task->done;
|
||||
sample->time = *now_mono;
|
||||
task->eta_sample_next = (task->eta_sample_next + 1) % ETA_SAMPLE_SLOTS;
|
||||
if(task->eta_sample_count < ETA_SAMPLE_SLOTS) {
|
||||
task->eta_sample_count++;
|
||||
}
|
||||
|
||||
for(i = 0; i < task->eta_sample_count; i++) {
|
||||
task_eta_sample_t *candidate = &task->eta_samples[i];
|
||||
long long age_ns;
|
||||
|
||||
if(!candidate->time.tv_sec || candidate->done >= task->done) {
|
||||
continue;
|
||||
}
|
||||
age_ns = timespec_delta_ns(now_mono, &candidate->time);
|
||||
if(age_ns <= 0 || age_ns > (long long)ETA_AVERAGE_WINDOW_SECONDS * 1000000000LL) {
|
||||
continue;
|
||||
}
|
||||
if(!base || age_ns > timespec_delta_ns(now_mono, &base->time)) {
|
||||
base = candidate;
|
||||
}
|
||||
}
|
||||
|
||||
if(base) {
|
||||
long long elapsed_ns = timespec_delta_ns(now_mono, &base->time);
|
||||
unsigned long long delta = task->done - base->done;
|
||||
unsigned long long remaining = task->total - task->done;
|
||||
if(delta && elapsed_ns > 0) {
|
||||
long double seconds = (long double)elapsed_ns / 1000000000.0L;
|
||||
long double eta = ((long double)remaining / (long double)delta) * seconds;
|
||||
task->eta = eta > 0 ? (unsigned long long)(eta + 0.999999L) : 0;
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
task->eta = task->speed ? (task->total - task->done + task->speed - 1) / task->speed : 0;
|
||||
}
|
||||
|
||||
void
|
||||
task_update(file_task_t *task, task_state_t state, const char *current,
|
||||
unsigned long long add_done, const char *error) {
|
||||
struct timespec now_mono;
|
||||
time_t now;
|
||||
|
||||
clock_gettime(CLOCK_MONOTONIC, &now_mono);
|
||||
now = time(NULL);
|
||||
|
||||
pthread_mutex_lock(&g_tasks_lock);
|
||||
task->state = state;
|
||||
if(current) {
|
||||
snprintf(task->current, sizeof(task->current), "%s", current);
|
||||
}
|
||||
if(add_done) {
|
||||
if(!task->transfer_started_at) {
|
||||
task->transfer_started_at = now;
|
||||
}
|
||||
task->done += add_done;
|
||||
if(task->total && task->done > task->total) {
|
||||
task->done = task->total;
|
||||
}
|
||||
if(task->speed_sample_time.tv_sec) {
|
||||
long long elapsed_ns = timespec_delta_ns(&now_mono, &task->speed_sample_time);
|
||||
if(elapsed_ns >= 250000000LL) {
|
||||
unsigned long long delta = task->done - task->speed_sample_done;
|
||||
task->speed = (unsigned long long)((delta * 1000000000ULL) /
|
||||
(unsigned long long)elapsed_ns);
|
||||
task->speed_sample_done = task->done;
|
||||
task->speed_sample_time = now_mono;
|
||||
}
|
||||
} else {
|
||||
task->speed_sample_done = task->done;
|
||||
task->speed_sample_time = now_mono;
|
||||
}
|
||||
task_update_eta_locked(task, &now_mono);
|
||||
}
|
||||
if(error) {
|
||||
snprintf(task->error, sizeof(task->error), "%s", error);
|
||||
}
|
||||
task->updated_at = now;
|
||||
pthread_mutex_unlock(&g_tasks_lock);
|
||||
}
|
||||
@@ -0,0 +1,445 @@
|
||||
#include "filemgr_internal.h"
|
||||
|
||||
#include <errno.h>
|
||||
#include <fcntl.h>
|
||||
#include <limits.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <strings.h>
|
||||
#include <sys/stat.h>
|
||||
#include <time.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#include "json_util.h"
|
||||
#include "path_util.h"
|
||||
#include "websrv.h"
|
||||
|
||||
#define TEXT_FILE_MAX_SIZE (1024 * 1024)
|
||||
|
||||
typedef enum text_newline {
|
||||
TEXT_NEWLINE_LF,
|
||||
TEXT_NEWLINE_CRLF,
|
||||
TEXT_NEWLINE_CR,
|
||||
} text_newline_t;
|
||||
|
||||
static enum MHD_Result
|
||||
send_json_version(struct MHD_Connection *conn, unsigned long long version) {
|
||||
char *data;
|
||||
|
||||
if(asprintf(&data, "{\"ok\":true,\"version\":\"%016llx\"}", version) < 0) {
|
||||
return MHD_NO;
|
||||
}
|
||||
return send_buffer(conn, MHD_HTTP_OK, data, "application/json");
|
||||
}
|
||||
|
||||
static int
|
||||
text_extension_allowed(const char *path) {
|
||||
static const char *extensions[] = {
|
||||
".txt", ".json", ".xml", ".ini", ".cfg", ".conf", ".md",
|
||||
".log", ".lua", ".js", ".css", ".html", ".htm", ".c", ".h",
|
||||
".cpp", ".hpp", ".sh", ".csv", ".yaml", ".yml", ".shn"
|
||||
};
|
||||
const char *extension = strrchr(path, '.');
|
||||
size_t i;
|
||||
|
||||
if(!extension) {
|
||||
return 0;
|
||||
}
|
||||
for(i = 0; i < sizeof(extensions) / sizeof(extensions[0]); i++) {
|
||||
if(!strcasecmp(extension, extensions[i])) {
|
||||
return 1;
|
||||
}
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int
|
||||
valid_utf8(const unsigned char *data, size_t size) {
|
||||
size_t i = 0;
|
||||
|
||||
while(i < size) {
|
||||
unsigned char c = data[i++];
|
||||
size_t trailing;
|
||||
unsigned int codepoint;
|
||||
|
||||
if(!c) return 0;
|
||||
if(c < 0x80) continue;
|
||||
if(c >= 0xc2 && c <= 0xdf) {
|
||||
trailing = 1;
|
||||
codepoint = c & 0x1f;
|
||||
} else if(c >= 0xe0 && c <= 0xef) {
|
||||
trailing = 2;
|
||||
codepoint = c & 0x0f;
|
||||
} else if(c >= 0xf0 && c <= 0xf4) {
|
||||
trailing = 3;
|
||||
codepoint = c & 0x07;
|
||||
} else {
|
||||
return 0;
|
||||
}
|
||||
if(trailing > size - i) return 0;
|
||||
while(trailing--) {
|
||||
unsigned char next = data[i++];
|
||||
if((next & 0xc0) != 0x80) return 0;
|
||||
codepoint = (codepoint << 6) | (next & 0x3f);
|
||||
}
|
||||
if((codepoint >= 0xd800 && codepoint <= 0xdfff) || codepoint > 0x10ffff ||
|
||||
(codepoint < 0x800 && c >= 0xe0) ||
|
||||
(codepoint < 0x10000 && c >= 0xf0)) {
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
return 1;
|
||||
}
|
||||
|
||||
static unsigned long long
|
||||
text_version(const unsigned char *data, size_t size) {
|
||||
unsigned long long hash = 1469598103934665603ULL;
|
||||
size_t i;
|
||||
|
||||
for(i = 0; i < size; i++) {
|
||||
hash ^= data[i];
|
||||
hash *= 1099511628211ULL;
|
||||
}
|
||||
return hash;
|
||||
}
|
||||
|
||||
static text_newline_t
|
||||
detect_text_newline(const unsigned char *data, size_t size) {
|
||||
size_t crlf = 0;
|
||||
size_t lf = 0;
|
||||
size_t cr = 0;
|
||||
size_t i;
|
||||
|
||||
for(i = 0; i < size; i++) {
|
||||
if(data[i] == '\r') {
|
||||
if(i + 1 < size && data[i + 1] == '\n') {
|
||||
crlf++;
|
||||
i++;
|
||||
} else {
|
||||
cr++;
|
||||
}
|
||||
} else if(data[i] == '\n') {
|
||||
lf++;
|
||||
}
|
||||
}
|
||||
if(crlf > lf && crlf >= cr) return TEXT_NEWLINE_CRLF;
|
||||
if(cr > lf && cr > crlf) return TEXT_NEWLINE_CR;
|
||||
return TEXT_NEWLINE_LF;
|
||||
}
|
||||
|
||||
static int
|
||||
format_text_for_save(const char *body, size_t body_size,
|
||||
const unsigned char *current, size_t current_size,
|
||||
char **output, size_t *output_size) {
|
||||
static const unsigned char bom[] = {0xef, 0xbb, 0xbf};
|
||||
int keep_bom = current_size >= sizeof(bom) &&
|
||||
!memcmp(current, bom, sizeof(bom));
|
||||
text_newline_t newline = detect_text_newline(
|
||||
current + (keep_bom ? sizeof(bom) : 0),
|
||||
current_size - (keep_bom ? sizeof(bom) : 0));
|
||||
const unsigned char *input = (const unsigned char *)(body ? body : "");
|
||||
size_t input_size = body_size;
|
||||
size_t capacity = body_size * (newline == TEXT_NEWLINE_CRLF ? 2 : 1) +
|
||||
sizeof(bom) + 1;
|
||||
char *formatted;
|
||||
size_t i;
|
||||
size_t len = 0;
|
||||
|
||||
if(input_size >= sizeof(bom) && !memcmp(input, bom, sizeof(bom))) {
|
||||
input += sizeof(bom);
|
||||
input_size -= sizeof(bom);
|
||||
}
|
||||
if(!(formatted = malloc(capacity))) {
|
||||
errno = ENOMEM;
|
||||
return -1;
|
||||
}
|
||||
if(keep_bom) {
|
||||
memcpy(formatted + len, bom, sizeof(bom));
|
||||
len += sizeof(bom);
|
||||
}
|
||||
for(i = 0; i < input_size; i++) {
|
||||
unsigned char c = input[i];
|
||||
|
||||
if(c != '\r' && c != '\n') {
|
||||
formatted[len++] = (char)c;
|
||||
continue;
|
||||
}
|
||||
if(c == '\r' && i + 1 < input_size && input[i + 1] == '\n') {
|
||||
i++;
|
||||
}
|
||||
if(newline == TEXT_NEWLINE_CRLF) {
|
||||
formatted[len++] = '\r';
|
||||
formatted[len++] = '\n';
|
||||
} else {
|
||||
formatted[len++] = newline == TEXT_NEWLINE_CR ? '\r' : '\n';
|
||||
}
|
||||
}
|
||||
if(len > TEXT_FILE_MAX_SIZE) {
|
||||
free(formatted);
|
||||
errno = EFBIG;
|
||||
return -1;
|
||||
}
|
||||
formatted[len] = 0;
|
||||
*output = formatted;
|
||||
*output_size = len;
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int
|
||||
read_text_file(const char *path, char **data, size_t *size,
|
||||
struct stat *st) {
|
||||
FILE *file;
|
||||
size_t read_size;
|
||||
|
||||
*data = NULL;
|
||||
*size = 0;
|
||||
if(lstat(path, st) || !S_ISREG(st->st_mode)) {
|
||||
return -1;
|
||||
}
|
||||
if(st->st_size < 0 || (unsigned long long)st->st_size > TEXT_FILE_MAX_SIZE) {
|
||||
errno = EFBIG;
|
||||
return -1;
|
||||
}
|
||||
if(!(file = fopen(path, "rb"))) {
|
||||
return -1;
|
||||
}
|
||||
if(!(*data = malloc((size_t)st->st_size + 1))) {
|
||||
fclose(file);
|
||||
errno = ENOMEM;
|
||||
return -1;
|
||||
}
|
||||
read_size = fread(*data, 1, (size_t)st->st_size, file);
|
||||
if(read_size != (size_t)st->st_size || ferror(file)) {
|
||||
free(*data);
|
||||
*data = NULL;
|
||||
fclose(file);
|
||||
return -1;
|
||||
}
|
||||
fclose(file);
|
||||
(*data)[read_size] = 0;
|
||||
*size = read_size;
|
||||
return 0;
|
||||
}
|
||||
|
||||
static enum MHD_Result
|
||||
send_text_file(struct MHD_Connection *conn, char *data, size_t size,
|
||||
unsigned long long version) {
|
||||
struct MHD_Response *resp;
|
||||
enum MHD_Result ret;
|
||||
char version_text[24];
|
||||
|
||||
if(!(resp = MHD_create_response_from_buffer(size, data,
|
||||
MHD_RESPMEM_MUST_FREE))) {
|
||||
free(data);
|
||||
return MHD_NO;
|
||||
}
|
||||
snprintf(version_text, sizeof(version_text), "%016llx", version);
|
||||
MHD_add_response_header(resp, MHD_HTTP_HEADER_CONTENT_TYPE,
|
||||
"text/plain; charset=utf-8");
|
||||
MHD_add_response_header(resp, "X-Text-Version", version_text);
|
||||
ret = websrv_queue_response(conn, MHD_HTTP_OK, resp);
|
||||
MHD_destroy_response(resp);
|
||||
return ret;
|
||||
}
|
||||
|
||||
enum MHD_Result
|
||||
api_text(struct MHD_Connection *conn) {
|
||||
char *path = fs_path_value(query_value(conn, "path"));
|
||||
char *data;
|
||||
size_t size;
|
||||
struct stat st;
|
||||
unsigned long long version;
|
||||
|
||||
if(has_active_task()) {
|
||||
free(path);
|
||||
return send_json_error(conn, MHD_HTTP_CONFLICT, "another task is running");
|
||||
}
|
||||
if(!path || !text_extension_allowed(path)) {
|
||||
free(path);
|
||||
return send_json_error(conn, MHD_HTTP_BAD_REQUEST,
|
||||
"file type is not editable");
|
||||
}
|
||||
if(read_text_file(path, &data, &size, &st)) {
|
||||
int error = errno;
|
||||
free(path);
|
||||
if(error == EFBIG) {
|
||||
return send_json_error(conn, MHD_HTTP_CONTENT_TOO_LARGE,
|
||||
"text file is too large");
|
||||
}
|
||||
errno = error;
|
||||
return send_json_error(conn, MHD_HTTP_NOT_FOUND, "file not found");
|
||||
}
|
||||
free(path);
|
||||
if(!valid_utf8((const unsigned char *)data, size)) {
|
||||
free(data);
|
||||
return send_json_error(conn, MHD_HTTP_UNSUPPORTED_MEDIA_TYPE,
|
||||
"file is not valid UTF-8");
|
||||
}
|
||||
version = text_version((const unsigned char *)data, size);
|
||||
return send_text_file(conn, data, size, version);
|
||||
}
|
||||
|
||||
enum MHD_Result
|
||||
api_text_create(struct MHD_Connection *conn) {
|
||||
char *path = fs_path_value(query_value(conn, "path"));
|
||||
char *name = fs_path_value(query_value(conn, "name"));
|
||||
char target[PATH_MAX];
|
||||
int fd = -1;
|
||||
int ret = -1;
|
||||
int error = 0;
|
||||
int created = 0;
|
||||
|
||||
if(has_active_task()) {
|
||||
free(path); free(name);
|
||||
return send_json_error(conn, MHD_HTTP_CONFLICT, "another task is running");
|
||||
}
|
||||
if(!path || !name || path_join(target, sizeof(target), path, name)) {
|
||||
free(path); free(name);
|
||||
return send_json_error(conn, MHD_HTTP_BAD_REQUEST, "invalid path");
|
||||
}
|
||||
if(mode_access(path, W_OK | X_OK)) {
|
||||
free(path); free(name);
|
||||
return send_json_error(conn, MHD_HTTP_FORBIDDEN,
|
||||
"text file is not writable");
|
||||
}
|
||||
if((fd = open(target, O_WRONLY | O_CREAT | O_EXCL, 0777)) >= 0) {
|
||||
created = 1;
|
||||
ret = fchmod_0777(fd);
|
||||
if(close(fd) && !ret) ret = -1;
|
||||
fd = -1;
|
||||
}
|
||||
if(ret) {
|
||||
error = errno;
|
||||
if(fd >= 0) close(fd);
|
||||
if(created) unlink(target);
|
||||
}
|
||||
free(path); free(name);
|
||||
if(!ret) {
|
||||
return send_json_version(conn,
|
||||
text_version((const unsigned char *)"", 0));
|
||||
}
|
||||
errno = error;
|
||||
return error == EEXIST ?
|
||||
send_json_error(conn, MHD_HTTP_CONFLICT, "file already exists") :
|
||||
send_json_error(conn, MHD_HTTP_INTERNAL_SERVER_ERROR, NULL);
|
||||
}
|
||||
|
||||
static int
|
||||
write_text_atomic(const char *path, const char *body, size_t body_size,
|
||||
mode_t mode) {
|
||||
struct timespec now;
|
||||
char temp[PATH_MAX];
|
||||
size_t written = 0;
|
||||
int fd = -1;
|
||||
int ret = -1;
|
||||
int n;
|
||||
|
||||
clock_gettime(CLOCK_MONOTONIC, &now);
|
||||
n = snprintf(temp, sizeof(temp), "%s.wfm-%ld-%ld.tmp", path,
|
||||
(long)getpid(), now.tv_nsec);
|
||||
if(n < 0 || (size_t)n >= sizeof(temp)) {
|
||||
errno = ENAMETOOLONG;
|
||||
return -1;
|
||||
}
|
||||
if((fd = open(temp, O_WRONLY | O_CREAT | O_EXCL, 0600)) < 0) {
|
||||
return -1;
|
||||
}
|
||||
while(written < body_size) {
|
||||
ssize_t count = write(fd, body + written, body_size - written);
|
||||
if(count <= 0) {
|
||||
goto done;
|
||||
}
|
||||
written += (size_t)count;
|
||||
}
|
||||
if(fchmod(fd, mode & 07777) && !ignore_chmod_error(errno)) {
|
||||
goto done;
|
||||
}
|
||||
if(fsync(fd)) {
|
||||
goto done;
|
||||
}
|
||||
if(close(fd)) {
|
||||
fd = -1;
|
||||
goto done;
|
||||
}
|
||||
fd = -1;
|
||||
if(rename(temp, path)) {
|
||||
goto done;
|
||||
}
|
||||
ret = 0;
|
||||
|
||||
done:
|
||||
if(fd >= 0) close(fd);
|
||||
if(ret) unlink(temp);
|
||||
return ret;
|
||||
}
|
||||
|
||||
enum MHD_Result
|
||||
api_text_save(struct MHD_Connection *conn, const char *body,
|
||||
size_t body_size) {
|
||||
char *path = fs_path_value(query_value(conn, "path"));
|
||||
char *expected = query_value(conn, "version");
|
||||
char *current = NULL;
|
||||
size_t current_size = 0;
|
||||
struct stat st;
|
||||
char version_text[24];
|
||||
char parent[PATH_MAX];
|
||||
char *formatted = NULL;
|
||||
size_t formatted_size = 0;
|
||||
int ret;
|
||||
|
||||
if(has_active_task()) {
|
||||
free(path); free(expected);
|
||||
return send_json_error(conn, MHD_HTTP_CONFLICT, "another task is running");
|
||||
}
|
||||
if(!path || !expected) {
|
||||
free(path); free(expected);
|
||||
return send_json_error(conn, MHD_HTTP_BAD_REQUEST, "invalid path");
|
||||
}
|
||||
if(body_size > TEXT_FILE_MAX_SIZE) {
|
||||
free(path); free(expected);
|
||||
return send_json_error(conn, MHD_HTTP_CONTENT_TOO_LARGE,
|
||||
"text file is too large");
|
||||
}
|
||||
if(!valid_utf8((const unsigned char *)(body ? body : ""), body_size)) {
|
||||
free(path); free(expected);
|
||||
return send_json_error(conn, MHD_HTTP_UNSUPPORTED_MEDIA_TYPE,
|
||||
"file is not valid UTF-8");
|
||||
}
|
||||
if(read_text_file(path, ¤t, ¤t_size, &st)) {
|
||||
free(path); free(expected);
|
||||
return send_json_error(conn, MHD_HTTP_NOT_FOUND, "file not found");
|
||||
}
|
||||
snprintf(version_text, sizeof(version_text), "%016llx",
|
||||
text_version((const unsigned char *)current, current_size));
|
||||
if(strcmp(expected, version_text)) {
|
||||
free(current);
|
||||
free(path); free(expected);
|
||||
return send_json_error(conn, MHD_HTTP_CONFLICT,
|
||||
"file changed since it was opened");
|
||||
}
|
||||
if(path_dirname(path, parent, sizeof(parent)) ||
|
||||
mode_access(path, W_OK) || mode_access(parent, W_OK | X_OK)) {
|
||||
free(current);
|
||||
free(path); free(expected);
|
||||
return send_json_error(conn, MHD_HTTP_FORBIDDEN,
|
||||
"text file is not writable");
|
||||
}
|
||||
if(format_text_for_save(body, body_size, (const unsigned char *)current,
|
||||
current_size, &formatted, &formatted_size)) {
|
||||
int error = errno;
|
||||
free(current);
|
||||
free(path); free(expected);
|
||||
errno = error;
|
||||
return error == EFBIG ?
|
||||
send_json_error(conn, MHD_HTTP_CONTENT_TOO_LARGE,
|
||||
"text file is too large") :
|
||||
send_json_error(conn, MHD_HTTP_INTERNAL_SERVER_ERROR, NULL);
|
||||
}
|
||||
free(current);
|
||||
ret = write_text_atomic(path, formatted, formatted_size, st.st_mode);
|
||||
free(formatted);
|
||||
free(path); free(expected);
|
||||
return ret ? send_json_error(conn, MHD_HTTP_INTERNAL_SERVER_ERROR, NULL)
|
||||
: send_json_ok(conn);
|
||||
}
|
||||
@@ -0,0 +1,582 @@
|
||||
#include "filemgr.h"
|
||||
|
||||
#include <errno.h>
|
||||
#include <fcntl.h>
|
||||
#include <limits.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <sys/stat.h>
|
||||
#include <time.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#include "filemgr_internal.h"
|
||||
#include "json_util.h"
|
||||
#include "path_util.h"
|
||||
|
||||
#define UPLOAD_BUFFER_SIZE (1024 * 1024)
|
||||
|
||||
typedef struct upload_context {
|
||||
file_task_t *task;
|
||||
int fd;
|
||||
char *buffer;
|
||||
size_t buffered;
|
||||
char temp[PATH_MAX];
|
||||
char target[PATH_MAX];
|
||||
unsigned long long expected;
|
||||
unsigned long long written;
|
||||
int failed;
|
||||
int error;
|
||||
int task_done;
|
||||
const char *stage;
|
||||
char error_message[256];
|
||||
} upload_context_t;
|
||||
|
||||
static const char *
|
||||
upload_error_message(upload_context_t *ctx) {
|
||||
if(!ctx->error_message[0]) {
|
||||
snprintf(ctx->error_message, sizeof(ctx->error_message), "%s failed%s%s: %s",
|
||||
ctx->stage ? ctx->stage : "upload",
|
||||
ctx->target[0] ? " for " : "",
|
||||
ctx->target[0] ? ctx->target : "",
|
||||
strerror(ctx->error ? ctx->error : EIO));
|
||||
}
|
||||
return ctx->error_message;
|
||||
}
|
||||
|
||||
static int
|
||||
upload_flush(upload_context_t *ctx) {
|
||||
size_t written = 0;
|
||||
|
||||
while(written < ctx->buffered) {
|
||||
ssize_t n;
|
||||
|
||||
if(ctx->task && task_cancel_requested(ctx->task)) {
|
||||
ctx->failed = 1;
|
||||
ctx->error = ECANCELED;
|
||||
return -1;
|
||||
}
|
||||
n = write(ctx->fd, ctx->buffer + written, ctx->buffered - written);
|
||||
if(n <= 0) {
|
||||
ctx->failed = 1;
|
||||
ctx->error = errno ? errno : EIO;
|
||||
return -1;
|
||||
}
|
||||
written += (size_t)n;
|
||||
ctx->written += (unsigned long long)n;
|
||||
}
|
||||
if(ctx->task && written) {
|
||||
task_update(ctx->task, TASK_RUNNING, ctx->target,
|
||||
(unsigned long long)written, NULL);
|
||||
}
|
||||
ctx->buffered = 0;
|
||||
return 0;
|
||||
}
|
||||
|
||||
static void
|
||||
finish_upload_task(file_task_t *task, task_state_t state, const char *current,
|
||||
const char *error) {
|
||||
if(!task) {
|
||||
return;
|
||||
}
|
||||
pthread_mutex_lock(&g_tasks_lock);
|
||||
if(task_is_active(task)) {
|
||||
time_t completed_at = time(NULL);
|
||||
task->state = state;
|
||||
if(current) {
|
||||
snprintf(task->current, sizeof(task->current), "%s", current);
|
||||
}
|
||||
if(state == TASK_DONE && task->total) {
|
||||
task->done = task->total;
|
||||
}
|
||||
if(state == TASK_DONE) {
|
||||
record_task_completion_locked(task, completed_at);
|
||||
}
|
||||
if(error) {
|
||||
snprintf(task->error, sizeof(task->error), "%s", error);
|
||||
}
|
||||
if(state == TASK_FAILED) {
|
||||
snprintf(task->error_code, sizeof(task->error_code), "upload_failed");
|
||||
snprintf(task->error_arg, sizeof(task->error_arg), "%s",
|
||||
current ? current : task->src);
|
||||
}
|
||||
task->updated_at = completed_at;
|
||||
}
|
||||
pthread_mutex_unlock(&g_tasks_lock);
|
||||
}
|
||||
|
||||
static void
|
||||
finish_upload_context_error(upload_context_t *ctx) {
|
||||
if(!ctx->task || ctx->task_done) {
|
||||
return;
|
||||
}
|
||||
upload_error_message(ctx);
|
||||
finish_upload_task(ctx->task,
|
||||
ctx->error == ECANCELED ? TASK_CANCELED : TASK_FAILED,
|
||||
ctx->target[0] ? ctx->target : ctx->task->current,
|
||||
ctx->error == ECANCELED ? "canceled" : ctx->error_message);
|
||||
ctx->task_done = 1;
|
||||
}
|
||||
|
||||
static file_task_t *
|
||||
upload_task_from_conn(struct MHD_Connection *conn) {
|
||||
char *idstr = request_value(conn, "X-WFM-Task-ID", "task_id");
|
||||
unsigned long id = idstr ? strtoul(idstr, NULL, 10) : 0;
|
||||
file_task_t *task = NULL;
|
||||
|
||||
free(idstr);
|
||||
if(!id) {
|
||||
return NULL;
|
||||
}
|
||||
pthread_mutex_lock(&g_tasks_lock);
|
||||
task = find_task_locked(id);
|
||||
if(!task || task->op != TASK_UPLOAD || !task_is_active(task)) {
|
||||
task = NULL;
|
||||
}
|
||||
pthread_mutex_unlock(&g_tasks_lock);
|
||||
return task;
|
||||
}
|
||||
|
||||
static int
|
||||
check_upload_manifest_space(const char *base, const char *rels,
|
||||
const char *sizes, unsigned long long fallback_total,
|
||||
char *error, size_t error_size,
|
||||
char *code, size_t code_size,
|
||||
char *arg, size_t arg_size) {
|
||||
char *rels_copy = NULL;
|
||||
char *sizes_copy = NULL;
|
||||
char *rel;
|
||||
char *size_text;
|
||||
char *rel_save;
|
||||
char *size_save;
|
||||
unsigned long long available;
|
||||
int ret = -1;
|
||||
|
||||
if(!rels || !sizes) {
|
||||
return check_target_space(base, fallback_total, error, error_size,
|
||||
code, code_size, arg, arg_size);
|
||||
}
|
||||
if(target_available_space(base, &available)) {
|
||||
snprintf(error, error_size, "cannot read target free space");
|
||||
snprintf(code, code_size, "space_check_failed");
|
||||
snprintf(arg, arg_size, "%s", base);
|
||||
return -1;
|
||||
}
|
||||
if(!(rels_copy = strdup(rels)) || !(sizes_copy = strdup(sizes))) {
|
||||
errno = ENOMEM;
|
||||
goto done;
|
||||
}
|
||||
|
||||
rel = strtok_r(rels_copy, "\n", &rel_save);
|
||||
size_text = strtok_r(sizes_copy, "\n", &size_save);
|
||||
while(rel || size_text) {
|
||||
char target[PATH_MAX];
|
||||
unsigned long long size;
|
||||
|
||||
if(!rel || !size_text || path_join_relative(target, sizeof(target), base, rel)) {
|
||||
snprintf(error, error_size, "invalid path");
|
||||
snprintf(code, code_size, "invalid_path");
|
||||
snprintf(arg, arg_size, "%s", rel ? rel : "");
|
||||
errno = EINVAL;
|
||||
goto done;
|
||||
}
|
||||
|
||||
size = strtoull(size_text, NULL, 10);
|
||||
if(available < size) {
|
||||
snprintf(error, error_size,
|
||||
"not enough target space, required %llu bytes, available %llu bytes",
|
||||
size, available);
|
||||
snprintf(code, code_size, "no_space");
|
||||
snprintf(arg, arg_size, "%llu,%llu", size, available);
|
||||
errno = ENOSPC;
|
||||
goto done;
|
||||
}
|
||||
|
||||
available -= size;
|
||||
|
||||
rel = strtok_r(NULL, "\n", &rel_save);
|
||||
size_text = strtok_r(NULL, "\n", &size_save);
|
||||
}
|
||||
|
||||
ret = 0;
|
||||
|
||||
done:
|
||||
free(rels_copy);
|
||||
free(sizes_copy);
|
||||
return ret;
|
||||
}
|
||||
|
||||
enum MHD_Result
|
||||
api_upload_prepare(struct MHD_Connection *conn, const char *body,
|
||||
size_t body_size) {
|
||||
char *path = fs_path_value(body_form_value(body, body_size, "path"));
|
||||
char *src = body_form_value(body, body_size, "src");
|
||||
char *total_text = body_form_value(body, body_size, "total");
|
||||
char *count_text = body_form_value(body, body_size, "count");
|
||||
char *rels = body_form_value(body, body_size, "rels");
|
||||
char *sizes = body_form_value(body, body_size, "sizes");
|
||||
char *overwrite = body_form_value(body, body_size, "overwrite");
|
||||
file_task_t *task = calloc(1, sizeof(*task));
|
||||
strbuf_t b = {0};
|
||||
unsigned long long total = total_text ? strtoull(total_text, NULL, 10) : 0;
|
||||
size_t count = count_text ? (size_t)strtoull(count_text, NULL, 10) : 0;
|
||||
char error[128] = {0};
|
||||
char code[64] = {0};
|
||||
char arg[PATH_MAX + 96] = {0};
|
||||
|
||||
if(!task) {
|
||||
free(path); free(src); free(total_text); free(count_text);
|
||||
free(rels); free(sizes); free(overwrite);
|
||||
return send_json_error(conn, MHD_HTTP_INTERNAL_SERVER_ERROR, "out of memory");
|
||||
}
|
||||
if(!path || !src || !count) {
|
||||
free_task(task);
|
||||
free(path); free(src); free(total_text); free(count_text);
|
||||
free(rels); free(sizes); free(overwrite);
|
||||
return send_json_error(conn, MHD_HTTP_BAD_REQUEST, "invalid path");
|
||||
}
|
||||
|
||||
pthread_mutex_lock(&g_tasks_lock);
|
||||
remove_finished_tasks_locked();
|
||||
if(has_active_task_locked()) {
|
||||
pthread_mutex_unlock(&g_tasks_lock);
|
||||
free_task(task);
|
||||
free(path); free(src); free(total_text); free(count_text);
|
||||
free(rels); free(sizes); free(overwrite);
|
||||
return send_json_error(conn, MHD_HTTP_CONFLICT, "another task is running");
|
||||
}
|
||||
pthread_mutex_unlock(&g_tasks_lock);
|
||||
|
||||
if(check_upload_manifest_space(path, rels, sizes, total, error, sizeof(error),
|
||||
code, sizeof(code), arg, sizeof(arg))) {
|
||||
free_task(task);
|
||||
free(path); free(src); free(total_text); free(count_text);
|
||||
free(rels); free(sizes); free(overwrite);
|
||||
return send_json_error_detail(conn,
|
||||
errno == ENOSPC ? MHD_HTTP_INSUFFICIENT_STORAGE :
|
||||
MHD_HTTP_INTERNAL_SERVER_ERROR,
|
||||
error[0] ? error : NULL,
|
||||
code[0] ? code : NULL,
|
||||
arg[0] ? arg : NULL);
|
||||
}
|
||||
|
||||
task->op = TASK_UPLOAD;
|
||||
task->state = TASK_RUNNING;
|
||||
task->src_count = count;
|
||||
task->file_count = count;
|
||||
task->total = total;
|
||||
snprintf(task->src, sizeof(task->src), "%s", src);
|
||||
snprintf(task->dst, sizeof(task->dst), "%s", path);
|
||||
snprintf(task->current, sizeof(task->current), "%s", src);
|
||||
task->created_at = time(NULL);
|
||||
task->updated_at = task->created_at;
|
||||
|
||||
pthread_mutex_lock(&g_tasks_lock);
|
||||
remove_finished_tasks_locked();
|
||||
if(has_active_task_locked()) {
|
||||
pthread_mutex_unlock(&g_tasks_lock);
|
||||
free_task(task);
|
||||
free(path); free(src); free(total_text); free(count_text);
|
||||
free(rels); free(sizes); free(overwrite);
|
||||
return send_json_error(conn, MHD_HTTP_CONFLICT, "another task is running");
|
||||
}
|
||||
task->id = g_next_task_id++;
|
||||
task->next = g_tasks;
|
||||
g_tasks = task;
|
||||
pthread_mutex_unlock(&g_tasks_lock);
|
||||
|
||||
free(path); free(src); free(total_text); free(count_text);
|
||||
free(rels); free(sizes); free(overwrite);
|
||||
strbuf_printf(&b, "{\"ok\":true,\"task_id\":%lu}", task->id);
|
||||
return send_buffer(conn, MHD_HTTP_OK, b.data, "application/json");
|
||||
}
|
||||
|
||||
enum MHD_Result
|
||||
api_upload_finish(struct MHD_Connection *conn) {
|
||||
file_task_t *task = upload_task_from_conn(conn);
|
||||
|
||||
if(!task) {
|
||||
return send_json_error(conn, MHD_HTTP_NOT_FOUND, "active task not found");
|
||||
}
|
||||
if(task_cancel_requested(task)) {
|
||||
finish_upload_task(task, TASK_CANCELED,
|
||||
task->current[0] ? task->current : task->src,
|
||||
"canceled");
|
||||
return send_json_error(conn, MHD_HTTP_CONFLICT, "canceled");
|
||||
}
|
||||
finish_upload_task(task, TASK_DONE,
|
||||
task->current[0] ? task->current : task->src, NULL);
|
||||
return send_json_ok(conn);
|
||||
}
|
||||
|
||||
int
|
||||
filemgr_upload_begin(struct MHD_Connection *conn, void **upload_ctx) {
|
||||
upload_context_t *ctx;
|
||||
char *base = fs_path_value(request_value(conn, "X-WFM-Path", "path"));
|
||||
char *rel = request_value(conn, "X-WFM-Rel", "rel");
|
||||
char *overwrite = request_value(conn, "X-WFM-Overwrite", "overwrite");
|
||||
char *size_text = request_value(conn, "X-WFM-Size", "size");
|
||||
file_task_t *task = upload_task_from_conn(conn);
|
||||
char **checked_dirs = NULL;
|
||||
size_t checked_dir_count = 0;
|
||||
struct stat st;
|
||||
char error[128] = {0};
|
||||
char code[64] = {0};
|
||||
char arg[PATH_MAX + 96] = {0};
|
||||
int n;
|
||||
|
||||
*upload_ctx = NULL;
|
||||
if(!(ctx = calloc(1, sizeof(*ctx)))) {
|
||||
free(base); free(rel); free(overwrite); free(size_text);
|
||||
errno = ENOMEM;
|
||||
return -1;
|
||||
}
|
||||
ctx->fd = -1;
|
||||
ctx->stage = "preparing upload";
|
||||
*upload_ctx = ctx;
|
||||
|
||||
if(size_text) {
|
||||
ctx->expected = strtoull(size_text, NULL, 10);
|
||||
}
|
||||
ctx->task = task;
|
||||
if(task) {
|
||||
pthread_mutex_lock(&g_tasks_lock);
|
||||
task->active_streams++;
|
||||
pthread_mutex_unlock(&g_tasks_lock);
|
||||
}
|
||||
if(task && task_cancel_requested(task)) {
|
||||
ctx->failed = 1;
|
||||
ctx->error = ECANCELED;
|
||||
goto fail;
|
||||
}
|
||||
if(!task && has_active_task()) {
|
||||
ctx->failed = 1;
|
||||
ctx->error = EBUSY;
|
||||
goto fail;
|
||||
}
|
||||
ctx->stage = "validating target path";
|
||||
if(!base || !rel || path_join_relative(ctx->target, sizeof(ctx->target),
|
||||
base, rel)) {
|
||||
ctx->failed = 1;
|
||||
ctx->error = errno ? errno : EINVAL;
|
||||
goto fail;
|
||||
}
|
||||
ctx->stage = "creating target folders";
|
||||
if(ensure_parent_dirs(base, rel)) {
|
||||
ctx->failed = 1;
|
||||
ctx->error = errno ? errno : EACCES;
|
||||
goto fail;
|
||||
}
|
||||
ctx->stage = "checking target path";
|
||||
if(!lstat(ctx->target, &st)) {
|
||||
if(S_ISDIR(st.st_mode) || !overwrite || strcmp(overwrite, "1")) {
|
||||
ctx->failed = 1;
|
||||
ctx->error = S_ISDIR(st.st_mode) ? EISDIR : EEXIST;
|
||||
goto fail;
|
||||
}
|
||||
} else if(errno != ENOENT) {
|
||||
ctx->failed = 1;
|
||||
ctx->error = errno;
|
||||
goto fail;
|
||||
}
|
||||
ctx->stage = "checking target permissions";
|
||||
if(check_target_writable(ctx->target, &checked_dirs, &checked_dir_count,
|
||||
error, sizeof(error), code, sizeof(code),
|
||||
arg, sizeof(arg))) {
|
||||
snprintf(ctx->error_message, sizeof(ctx->error_message), "%s",
|
||||
error[0] ? error : "target is not writable");
|
||||
ctx->failed = 1;
|
||||
ctx->error = errno ? errno : EACCES;
|
||||
goto fail;
|
||||
}
|
||||
ctx->stage = "checking target space";
|
||||
if(check_target_space(ctx->target, ctx->expected, error, sizeof(error),
|
||||
code, sizeof(code), arg, sizeof(arg))) {
|
||||
snprintf(ctx->error_message, sizeof(ctx->error_message), "%s",
|
||||
error[0] ? error : "not enough target space");
|
||||
ctx->failed = 1;
|
||||
ctx->error = errno ? errno : ENOSPC;
|
||||
goto fail;
|
||||
}
|
||||
ctx->stage = "creating temporary path";
|
||||
n = snprintf(ctx->temp, sizeof(ctx->temp), "%s.wfm-upload-%ld-%lld.tmp",
|
||||
ctx->target, (long)getpid(), (long long)time(NULL));
|
||||
if(n < 0 || (size_t)n >= sizeof(ctx->temp)) {
|
||||
ctx->failed = 1;
|
||||
ctx->error = ENAMETOOLONG;
|
||||
goto fail;
|
||||
}
|
||||
ctx->stage = "opening temporary file";
|
||||
ctx->fd = open(ctx->temp, O_WRONLY | O_CREAT | O_TRUNC, 0600);
|
||||
if(ctx->fd < 0) {
|
||||
ctx->failed = 1;
|
||||
ctx->error = errno;
|
||||
goto fail;
|
||||
}
|
||||
ctx->stage = "allocating upload buffer";
|
||||
if(!(ctx->buffer = malloc(UPLOAD_BUFFER_SIZE))) {
|
||||
ctx->failed = 1;
|
||||
ctx->error = ENOMEM;
|
||||
goto fail;
|
||||
}
|
||||
|
||||
fail:
|
||||
free_paths(checked_dirs, checked_dir_count);
|
||||
free(base); free(rel); free(overwrite); free(size_text);
|
||||
if(ctx->failed) {
|
||||
finish_upload_context_error(ctx);
|
||||
if(ctx->fd >= 0) {
|
||||
close(ctx->fd);
|
||||
ctx->fd = -1;
|
||||
}
|
||||
if(ctx->temp[0]) {
|
||||
unlink(ctx->temp);
|
||||
}
|
||||
errno = ctx->error ? ctx->error : EIO;
|
||||
return -1;
|
||||
}
|
||||
ctx->stage = "writing file";
|
||||
return 0;
|
||||
}
|
||||
|
||||
int
|
||||
filemgr_upload_data(void *upload_ctx, const char *data, size_t size) {
|
||||
upload_context_t *ctx = upload_ctx;
|
||||
|
||||
if(!ctx || ctx->failed) {
|
||||
return -1;
|
||||
}
|
||||
if(ctx->task && task_cancel_requested(ctx->task)) {
|
||||
ctx->failed = 1;
|
||||
ctx->error = ECANCELED;
|
||||
finish_upload_context_error(ctx);
|
||||
return -1;
|
||||
}
|
||||
while(size) {
|
||||
size_t space = UPLOAD_BUFFER_SIZE - ctx->buffered;
|
||||
size_t take = size < space ? size : space;
|
||||
|
||||
memcpy(ctx->buffer + ctx->buffered, data, take);
|
||||
ctx->buffered += take;
|
||||
data += take;
|
||||
size -= take;
|
||||
if(ctx->buffered == UPLOAD_BUFFER_SIZE && upload_flush(ctx)) {
|
||||
finish_upload_context_error(ctx);
|
||||
return -1;
|
||||
}
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
enum MHD_Result
|
||||
filemgr_upload_finish(struct MHD_Connection *conn, void *upload_ctx) {
|
||||
upload_context_t *ctx = upload_ctx;
|
||||
int ret = -1;
|
||||
|
||||
if(!ctx) {
|
||||
return send_json_error(conn, MHD_HTTP_BAD_REQUEST, "invalid path");
|
||||
}
|
||||
if(ctx->failed) {
|
||||
finish_upload_context_error(ctx);
|
||||
errno = ctx->error ? ctx->error : EIO;
|
||||
return send_json_error_detail(conn,
|
||||
errno == EEXIST ? MHD_HTTP_CONFLICT :
|
||||
errno == EBUSY ? MHD_HTTP_CONFLICT :
|
||||
errno == ECANCELED ? MHD_HTTP_CONFLICT :
|
||||
errno == ENOSPC ? MHD_HTTP_INSUFFICIENT_STORAGE :
|
||||
MHD_HTTP_INTERNAL_SERVER_ERROR,
|
||||
ctx->error == ECANCELED ? "canceled" : ctx->error_message,
|
||||
ctx->error == ECANCELED ? NULL : "upload_failed",
|
||||
ctx->target[0] ? ctx->target : NULL);
|
||||
}
|
||||
ctx->stage = "writing file";
|
||||
if(ctx->buffered && upload_flush(ctx)) {
|
||||
ctx->error = ctx->error ? ctx->error : EIO;
|
||||
goto done;
|
||||
}
|
||||
ctx->stage = "verifying uploaded size";
|
||||
if(ctx->expected && ctx->written != ctx->expected) {
|
||||
ctx->error = EIO;
|
||||
goto done;
|
||||
}
|
||||
ctx->stage = "setting file permissions";
|
||||
if(fchmod_0777(ctx->fd)) {
|
||||
ctx->error = errno;
|
||||
goto done;
|
||||
}
|
||||
ctx->stage = "syncing file data";
|
||||
if(fsync(ctx->fd)) {
|
||||
ctx->error = errno;
|
||||
goto done;
|
||||
}
|
||||
ctx->stage = "closing temporary file";
|
||||
if(close(ctx->fd)) {
|
||||
ctx->fd = -1;
|
||||
ctx->error = errno;
|
||||
goto done;
|
||||
}
|
||||
ctx->fd = -1;
|
||||
ctx->stage = "moving temporary file into place";
|
||||
if(rename(ctx->temp, ctx->target)) {
|
||||
ctx->error = errno;
|
||||
goto done;
|
||||
}
|
||||
if(ctx->task) {
|
||||
pthread_mutex_lock(&g_tasks_lock);
|
||||
ctx->task->upload_completed++;
|
||||
pthread_mutex_unlock(&g_tasks_lock);
|
||||
}
|
||||
ret = 0;
|
||||
|
||||
done:
|
||||
if(ctx->fd >= 0) {
|
||||
close(ctx->fd);
|
||||
ctx->fd = -1;
|
||||
}
|
||||
if(ret) {
|
||||
upload_error_message(ctx);
|
||||
if(ctx->temp[0]) {
|
||||
unlink(ctx->temp);
|
||||
}
|
||||
finish_upload_context_error(ctx);
|
||||
errno = ctx->error ? ctx->error : EIO;
|
||||
return send_json_error_detail(conn,
|
||||
errno == ECANCELED ? MHD_HTTP_CONFLICT :
|
||||
MHD_HTTP_INTERNAL_SERVER_ERROR,
|
||||
ctx->error == ECANCELED ? "canceled" : ctx->error_message,
|
||||
ctx->error == ECANCELED ? NULL : "upload_failed",
|
||||
ctx->target[0] ? ctx->target : NULL);
|
||||
}
|
||||
return send_json_ok(conn);
|
||||
}
|
||||
|
||||
void
|
||||
filemgr_upload_free(void *upload_ctx) {
|
||||
upload_context_t *ctx = upload_ctx;
|
||||
|
||||
if(!ctx) {
|
||||
return;
|
||||
}
|
||||
if(ctx->fd >= 0) {
|
||||
close(ctx->fd);
|
||||
if(ctx->temp[0]) {
|
||||
unlink(ctx->temp);
|
||||
}
|
||||
if(ctx->task && !ctx->task_done) {
|
||||
int canceled = task_cancel_requested(ctx->task);
|
||||
finish_upload_task(ctx->task, canceled ? TASK_CANCELED : TASK_FAILED,
|
||||
ctx->target[0] ? ctx->target : ctx->task->current,
|
||||
canceled ? "canceled" : "client disconnected");
|
||||
ctx->task_done = 1;
|
||||
}
|
||||
}
|
||||
free(ctx->buffer);
|
||||
if(ctx->task) {
|
||||
pthread_mutex_lock(&g_tasks_lock);
|
||||
if(ctx->task->active_streams) {
|
||||
ctx->task->active_streams--;
|
||||
}
|
||||
pthread_mutex_unlock(&g_tasks_lock);
|
||||
}
|
||||
free(ctx);
|
||||
}
|
||||
@@ -0,0 +1,34 @@
|
||||
/*
|
||||
* /api/version -- hands the build's VERSION_TAG to the web UI.
|
||||
*
|
||||
* The footer in the browser shows a version string, and for a long time that
|
||||
* string was a literal in assets/main.js, so it drifted out of sync the moment
|
||||
* the Makefile moved on (v1.9 stayed on screen through the whole v1.9.1
|
||||
* release). Exposing it over the API keeps a single source of truth: bump
|
||||
* VERSION_TAG in the Makefile and every surface -- startup notification
|
||||
* (src/main.c), stdout banner, ELF file name and the UI footer -- follows.
|
||||
*
|
||||
* The response is tiny and immutable, so the client caches it for the session.
|
||||
*/
|
||||
|
||||
#include "filemgr_internal.h"
|
||||
|
||||
#include <string.h>
|
||||
|
||||
#include "json_util.h"
|
||||
|
||||
#ifndef VERSION_TAG
|
||||
#define VERSION_TAG "unknown"
|
||||
#endif
|
||||
|
||||
enum MHD_Result
|
||||
api_version(struct MHD_Connection *conn) {
|
||||
strbuf_t b = {0};
|
||||
|
||||
strbuf_append(&b, "{\"ok\":true,\"version\":");
|
||||
json_escape(&b, VERSION_TAG);
|
||||
strbuf_append(&b, ",\"titleId\":");
|
||||
json_escape(&b, TITLE_ID);
|
||||
strbuf_append(&b, "}");
|
||||
return send_buffer(conn, MHD_HTTP_OK, b.data, "application/json");
|
||||
}
|
||||
@@ -6,6 +6,7 @@
|
||||
#include <arpa/inet.h>
|
||||
#include <microhttpd.h>
|
||||
#include <netinet/in.h>
|
||||
#include <netinet/tcp.h>
|
||||
#include <sys/socket.h>
|
||||
#include <unistd.h>
|
||||
|
||||
@@ -13,6 +14,53 @@
|
||||
#include "filemgr.h"
|
||||
#include "websrv.h"
|
||||
|
||||
#define REQUEST_BODY_MAX (4 * 1024 * 1024)
|
||||
#define HTTP_CONNECTION_MEMORY_LIMIT (8 * 1024 * 1024)
|
||||
#define HTTP_CONNECTION_MEMORY_INCREMENT (2 * 1024 * 1024)
|
||||
#define HTTP_SOCKET_RCVBUF_SIZE (4 * 1024 * 1024)
|
||||
#define HTTP_SOCKET_SNDBUF_SIZE (4 * 1024 * 1024)
|
||||
|
||||
static volatile sig_atomic_t g_stop_requested;
|
||||
static int g_listen_fd = -1;
|
||||
|
||||
static void
|
||||
websrv_tune_connection_socket(int fd) {
|
||||
const int sndbuf = HTTP_SOCKET_SNDBUF_SIZE;
|
||||
const int nodelay = 1;
|
||||
|
||||
/* OrbisOS HTTP sockets need an explicit send buffer to fill a GbE link. */
|
||||
(void)setsockopt(fd, SOL_SOCKET, SO_SNDBUF, &sndbuf, sizeof(sndbuf));
|
||||
(void)setsockopt(fd, IPPROTO_TCP, TCP_NODELAY, &nodelay, sizeof(nodelay));
|
||||
}
|
||||
|
||||
typedef struct request_context {
|
||||
char *body;
|
||||
size_t size;
|
||||
int too_large;
|
||||
int upload_stream;
|
||||
void *upload_ctx;
|
||||
} request_context_t;
|
||||
|
||||
static enum MHD_Result
|
||||
websrv_body_too_large(struct MHD_Connection *conn) {
|
||||
static const char json[] =
|
||||
"{\"ok\":false,\"error\":\"request body is too large\","
|
||||
"\"error_code\":\"request_body_too_large\",\"error_arg\":\"\"}";
|
||||
struct MHD_Response *resp =
|
||||
MHD_create_response_from_buffer(sizeof(json) - 1, (void *)json,
|
||||
MHD_RESPMEM_PERSISTENT);
|
||||
enum MHD_Result ret;
|
||||
|
||||
if(!resp) {
|
||||
return MHD_NO;
|
||||
}
|
||||
MHD_add_response_header(resp, MHD_HTTP_HEADER_CONTENT_TYPE,
|
||||
"application/json");
|
||||
ret = websrv_queue_response(conn, MHD_HTTP_CONTENT_TOO_LARGE, resp);
|
||||
MHD_destroy_response(resp);
|
||||
return ret;
|
||||
}
|
||||
|
||||
enum MHD_Result
|
||||
websrv_queue_response(struct MHD_Connection *conn, unsigned int status,
|
||||
struct MHD_Response *resp) {
|
||||
@@ -21,15 +69,27 @@ websrv_queue_response(struct MHD_Connection *conn, unsigned int status,
|
||||
return MHD_queue_response(conn, status, resp);
|
||||
}
|
||||
|
||||
void
|
||||
websrv_stop(void) {
|
||||
g_stop_requested = 1;
|
||||
if(g_listen_fd >= 0) {
|
||||
shutdown(g_listen_fd, SHUT_RDWR);
|
||||
}
|
||||
}
|
||||
|
||||
int
|
||||
websrv_stop_requested(void) {
|
||||
return g_stop_requested;
|
||||
}
|
||||
|
||||
static enum MHD_Result
|
||||
websrv_on_request(void *cls, struct MHD_Connection *conn, const char *url,
|
||||
const char *method, const char *version,
|
||||
const char *upload_data, size_t *upload_data_size,
|
||||
void **con_cls) {
|
||||
request_context_t *ctx = *con_cls;
|
||||
(void)cls;
|
||||
(void)version;
|
||||
(void)upload_data;
|
||||
(void)upload_data_size;
|
||||
|
||||
if(strcmp(method, MHD_HTTP_METHOD_GET) &&
|
||||
strcmp(method, MHD_HTTP_METHOD_POST) &&
|
||||
@@ -37,13 +97,62 @@ websrv_on_request(void *cls, struct MHD_Connection *conn, const char *url,
|
||||
return MHD_NO;
|
||||
}
|
||||
|
||||
if(!*con_cls) {
|
||||
*con_cls = (void *)1;
|
||||
if(!ctx) {
|
||||
if(!(ctx = calloc(1, sizeof(*ctx)))) {
|
||||
return MHD_NO;
|
||||
}
|
||||
ctx->upload_stream = !strcmp(url, "/api/upload-file") &&
|
||||
!strcmp(method, MHD_HTTP_METHOD_POST);
|
||||
*con_cls = ctx;
|
||||
return MHD_YES;
|
||||
}
|
||||
|
||||
if(*upload_data_size) {
|
||||
size_t chunk_size = *upload_data_size;
|
||||
|
||||
if(ctx->upload_stream) {
|
||||
if(!ctx->upload_ctx && filemgr_upload_begin(conn, &ctx->upload_ctx)) {
|
||||
ctx->too_large = 1;
|
||||
}
|
||||
if(ctx->upload_ctx && filemgr_upload_data(ctx->upload_ctx, upload_data,
|
||||
chunk_size)) {
|
||||
ctx->too_large = 1;
|
||||
}
|
||||
*upload_data_size = 0;
|
||||
return MHD_YES;
|
||||
}
|
||||
|
||||
if(chunk_size > REQUEST_BODY_MAX - ctx->size) {
|
||||
ctx->too_large = 1;
|
||||
} else if(!ctx->too_large) {
|
||||
char *body = realloc(ctx->body, ctx->size + chunk_size + 1);
|
||||
if(!body) {
|
||||
return MHD_NO;
|
||||
}
|
||||
ctx->body = body;
|
||||
memcpy(ctx->body + ctx->size, upload_data, chunk_size);
|
||||
ctx->size += chunk_size;
|
||||
ctx->body[ctx->size] = 0;
|
||||
}
|
||||
*upload_data_size = 0;
|
||||
return MHD_YES;
|
||||
}
|
||||
|
||||
if(ctx->too_large) {
|
||||
return ctx->upload_stream ?
|
||||
filemgr_upload_finish(conn, ctx->upload_ctx) :
|
||||
websrv_body_too_large(conn);
|
||||
}
|
||||
|
||||
if(ctx->upload_stream) {
|
||||
if(!ctx->upload_ctx && filemgr_upload_begin(conn, &ctx->upload_ctx)) {
|
||||
return filemgr_upload_finish(conn, ctx->upload_ctx);
|
||||
}
|
||||
return filemgr_upload_finish(conn, ctx->upload_ctx);
|
||||
}
|
||||
|
||||
if(!strncmp(url, "/api/", 5)) {
|
||||
return filemgr_api_request(conn, url);
|
||||
return filemgr_api_request(conn, url, method, ctx->body, ctx->size);
|
||||
}
|
||||
if(!strcmp(url, "/fs")) {
|
||||
return filemgr_fs_request(conn);
|
||||
@@ -51,6 +160,9 @@ websrv_on_request(void *cls, struct MHD_Connection *conn, const char *url,
|
||||
if(!strcmp(url, "/") || !url[0]) {
|
||||
return asset_request(conn, "/index.html");
|
||||
}
|
||||
if(!strcmp(url, "/favicon.ico")) {
|
||||
return asset_request(conn, "/icon0.png");
|
||||
}
|
||||
return asset_request(conn, url);
|
||||
}
|
||||
|
||||
@@ -60,6 +172,15 @@ websrv_on_completed(void *cls, struct MHD_Connection *connection,
|
||||
(void)cls;
|
||||
(void)connection;
|
||||
(void)toe;
|
||||
request_context_t *ctx = *con_cls;
|
||||
|
||||
if(ctx) {
|
||||
if(ctx->upload_ctx) {
|
||||
filemgr_upload_free(ctx->upload_ctx);
|
||||
}
|
||||
free(ctx->body);
|
||||
free(ctx);
|
||||
}
|
||||
*con_cls = NULL;
|
||||
}
|
||||
|
||||
@@ -84,6 +205,12 @@ websrv_listen(unsigned short port) {
|
||||
close(srvfd);
|
||||
return -1;
|
||||
}
|
||||
{
|
||||
const int rcvbuf = HTTP_SOCKET_RCVBUF_SIZE;
|
||||
|
||||
/* Set before listen so accepted PS5 sockets inherit the larger window. */
|
||||
(void)setsockopt(srvfd, SOL_SOCKET, SO_RCVBUF, &rcvbuf, sizeof(rcvbuf));
|
||||
}
|
||||
|
||||
memset(&server_addr, 0, sizeof(server_addr));
|
||||
server_addr.sin_family = AF_INET;
|
||||
@@ -100,11 +227,17 @@ websrv_listen(unsigned short port) {
|
||||
close(srvfd);
|
||||
return -1;
|
||||
}
|
||||
g_stop_requested = 0;
|
||||
g_listen_fd = srvfd;
|
||||
|
||||
if(!(httpd = MHD_start_daemon(MHD_USE_THREAD_PER_CONNECTION | MHD_USE_ITC |
|
||||
MHD_USE_NO_LISTEN_SOCKET | MHD_USE_DEBUG |
|
||||
MHD_USE_INTERNAL_POLLING_THREAD,
|
||||
MHD_USE_INTERNAL_POLLING_THREAD | MHD_USE_TURBO,
|
||||
0, NULL, NULL, &websrv_on_request, NULL,
|
||||
MHD_OPTION_CONNECTION_MEMORY_LIMIT,
|
||||
(size_t)HTTP_CONNECTION_MEMORY_LIMIT,
|
||||
MHD_OPTION_CONNECTION_MEMORY_INCREMENT,
|
||||
(size_t)HTTP_CONNECTION_MEMORY_INCREMENT,
|
||||
MHD_OPTION_NOTIFY_COMPLETED,
|
||||
&websrv_on_completed, NULL, MHD_OPTION_END))) {
|
||||
perror("MHD_start_daemon");
|
||||
@@ -112,12 +245,13 @@ websrv_listen(unsigned short port) {
|
||||
return -1;
|
||||
}
|
||||
|
||||
while(1) {
|
||||
while(!g_stop_requested) {
|
||||
addr_len = sizeof(client_addr);
|
||||
if((connfd = accept(srvfd, (struct sockaddr *)&client_addr, &addr_len)) < 0) {
|
||||
perror("accept");
|
||||
if(!g_stop_requested) perror("accept");
|
||||
break;
|
||||
}
|
||||
websrv_tune_connection_socket(connfd);
|
||||
if(MHD_add_connection(httpd, connfd, (struct sockaddr *)&client_addr,
|
||||
addr_len) != MHD_YES) {
|
||||
perror("MHD_add_connection");
|
||||
@@ -127,5 +261,6 @@ websrv_listen(unsigned short port) {
|
||||
}
|
||||
|
||||
MHD_stop_daemon(httpd);
|
||||
g_listen_fd = -1;
|
||||
return close(srvfd);
|
||||
}
|
||||
@@ -23,3 +23,5 @@ enum MHD_Result websrv_queue_response(struct MHD_Connection *conn,
|
||||
struct MHD_Response *resp);
|
||||
|
||||
int websrv_listen(unsigned short port);
|
||||
void websrv_stop(void);
|
||||
int websrv_stop_requested(void);
|
||||
@@ -0,0 +1,120 @@
|
||||
#pragma once
|
||||
|
||||
/* Standalone ZIP extraction engine.
|
||||
No HTTP / task dependencies: it can be compiled and tested on its own. */
|
||||
|
||||
#include <stdint.h>
|
||||
#include <stddef.h>
|
||||
|
||||
#define ZIPX_PATH_MAX 4096
|
||||
|
||||
typedef enum {
|
||||
ZIPX_PHASE_SCAN = 0,
|
||||
ZIPX_PHASE_EXTRACT = 1,
|
||||
ZIPX_PHASE_PUBLISH = 2,
|
||||
ZIPX_PHASE_CLEANUP = 3
|
||||
} zipx_phase_t;
|
||||
|
||||
typedef enum {
|
||||
ZIPX_OK = 0,
|
||||
ZIPX_ERR_CANCELED,
|
||||
ZIPX_ERR_OPEN, /* cannot open the archive */
|
||||
ZIPX_ERR_FORMAT, /* corrupt central directory / truncated */
|
||||
ZIPX_ERR_UNSUPPORTED,/* multipart or unsupported compression method */
|
||||
ZIPX_ERR_UNSAFE_NAME,/* traversal, absolute path, control chars, NUL */
|
||||
ZIPX_ERR_SPECIAL, /* symlink / device / fifo / socket entry */
|
||||
ZIPX_ERR_DUPLICATE, /* repeated entry or file/dir name clash inside zip */
|
||||
ZIPX_ERR_LIMIT_ENTRIES,
|
||||
ZIPX_ERR_LIMIT_FILE,
|
||||
ZIPX_ERR_LIMIT_TOTAL,
|
||||
ZIPX_ERR_LIMIT_RATIO,
|
||||
ZIPX_ERR_LIMIT_DEPTH,
|
||||
ZIPX_ERR_LIMIT_NAME,
|
||||
ZIPX_ERR_LIMIT_DICT, /* the archive's dictionary exceeds what we allow
|
||||
(RAR7 headers may ask for up to 64 GiB) */
|
||||
ZIPX_ERR_CONFLICT, /* target already exists for the chosen policy */
|
||||
ZIPX_ERR_PASSWORD, /* the archive is encrypted and the password is missing
|
||||
or wrong; the caller can prompt and retry */
|
||||
ZIPX_ERR_SPACE,
|
||||
ZIPX_ERR_IO,
|
||||
ZIPX_ERR_CRC,
|
||||
ZIPX_ERR_INTERNAL
|
||||
} zipx_status_t;
|
||||
|
||||
typedef enum {
|
||||
ZIPX_CONFLICT_FAIL = 0,
|
||||
ZIPX_CONFLICT_OVERWRITE = 1,
|
||||
ZIPX_CONFLICT_MERGE = 2
|
||||
} zipx_conflict_t;
|
||||
|
||||
typedef struct {
|
||||
uint64_t max_entries;
|
||||
uint64_t max_total_bytes;
|
||||
uint64_t max_file_bytes;
|
||||
uint32_t max_ratio; /* uncompressed/compressed, 0 disables */
|
||||
/* Entries whose uncompressed size is below this are never ratio-screened.
|
||||
Small highly-compressible entries are common in legitimate archives
|
||||
(zero-filled placeholders, sparse blobs) and are harmless because the
|
||||
actual bytes written are bounded by the declared size and by the real
|
||||
free-space check; the ratio screen only needs to catch entries large
|
||||
enough to matter. */
|
||||
uint64_t ratio_min_bytes;
|
||||
uint32_t max_depth;
|
||||
uint32_t max_name_len;
|
||||
uint32_t max_path_len;
|
||||
} zipx_limits_t;
|
||||
|
||||
typedef struct {
|
||||
int phase;
|
||||
uint64_t entries_total;
|
||||
uint64_t entries_done;
|
||||
uint64_t bytes_total;
|
||||
uint64_t bytes_done;
|
||||
const char *current; /* entry name being processed, may be NULL */
|
||||
} zipx_progress_t;
|
||||
|
||||
/* Returns non-zero when the caller wants the operation to stop. */
|
||||
typedef int (*zipx_cancel_fn)(void *userdata);
|
||||
typedef void (*zipx_progress_fn)(void *userdata, const zipx_progress_t *progress);
|
||||
|
||||
typedef struct {
|
||||
zipx_status_t status;
|
||||
int sys_errno;
|
||||
uint64_t entries_total;
|
||||
uint64_t entries_done;
|
||||
uint64_t bytes_total;
|
||||
uint64_t files_created;
|
||||
uint64_t dirs_created;
|
||||
char detail[ZIPX_PATH_MAX]; /* offending path or the staging directory */
|
||||
char message[192];
|
||||
} zipx_result_t;
|
||||
|
||||
const zipx_limits_t *zipx_default_limits(void);
|
||||
|
||||
/* Pre-built limit profiles. Use zipx_limits_profile() to look one up.
|
||||
ZIPX_LIMITS_DEFAULT is the safe profile shipped by zipx_default_limits().
|
||||
ZIPX_LIMITS_LARGE allows archives up to 2 TiB total / 1 TiB per file and
|
||||
a 1000:1 compression ratio. The caller is responsible for verifying that
|
||||
the PS5 has enough free disk space. */
|
||||
#define ZIPX_LIMITS_DEFAULT 0
|
||||
#define ZIPX_LIMITS_LARGE 1
|
||||
|
||||
const zipx_limits_t *zipx_limits_profile(int profile);
|
||||
const char *zipx_status_string(zipx_status_t status);
|
||||
|
||||
/* Extract zip_path into dst_dir.
|
||||
`password` may be NULL or empty when the archive is not encrypted; it is
|
||||
used for both ZIP encryption schemes, traditional PKWARE ("ZipCrypto") and
|
||||
WinZip AES. A missing or wrong password is reported as ZIPX_ERR_PASSWORD,
|
||||
which the caller is expected to turn into a prompt and retry.
|
||||
Returns ZIPX_OK or an error code; *result is always filled in.
|
||||
On any failure the staging directory is removed and dst_dir is left as it
|
||||
was, except for objects already published with the overwrite policy. */
|
||||
zipx_status_t zipx_extract(const char *zip_path, const char *dst_dir,
|
||||
zipx_conflict_t conflict,
|
||||
const zipx_limits_t *limits,
|
||||
zipx_cancel_fn cancel,
|
||||
zipx_progress_fn progress,
|
||||
void *userdata,
|
||||
const char *password,
|
||||
zipx_result_t *result);
|
||||
@@ -0,0 +1,100 @@
|
||||
/* Bits of the zipx_* contract that are not specific to a container format.
|
||||
|
||||
The limit profiles and the status-to-text mapping describe the *engine
|
||||
family*, not ZIP, so they live here rather than inside zip_extract.c. All
|
||||
three engines (ZIP, RAR, 7z) link this one object; keeping them in the ZIP
|
||||
file would force the RAR and 7z test builds to drag in minizip-ng and zlib
|
||||
for the sake of three functions. */
|
||||
|
||||
#include "zip_extract.h"
|
||||
|
||||
/* Default limits.
|
||||
*
|
||||
* Tuned to cover real-world PS5 workloads without prompting:
|
||||
* - PS5 system backup archives (~200-300 GiB total, individual chunks
|
||||
* well under 64 GiB)
|
||||
* - 3A-game archives with a single ~300 GiB uncompressed file
|
||||
*
|
||||
* Safety against decompression bombs is delegated to:
|
||||
* 1. `check_space()` (statvfs-based real disk space check) before extract
|
||||
* 2. `max_ratio` below (declared compression ratio cap)
|
||||
* The size caps here are an early-fail UX guard, not a security boundary.
|
||||
*/
|
||||
static const zipx_limits_t k_default_limits = {
|
||||
.max_entries = 200000,
|
||||
.max_total_bytes = 2ULL * 1024 * 1024 * 1024 * 1024,
|
||||
.max_file_bytes = 512ULL * 1024 * 1024 * 1024,
|
||||
.max_ratio = 500,
|
||||
/* Only entries that would individually materialise >=1 GiB are screened
|
||||
by ratio; anything smaller is harmless (bounded by declared size + the
|
||||
real free-space check) and is commonly highly compressible in
|
||||
legitimate archives. */
|
||||
.ratio_min_bytes = 1ULL * 1024 * 1024 * 1024,
|
||||
.max_depth = 32,
|
||||
.max_name_len = 255,
|
||||
.max_path_len = 1024
|
||||
};
|
||||
|
||||
/* Large profile for archives that exceed the default cap.
|
||||
*
|
||||
* - max_file_bytes = 1 TiB (single uncompressed file)
|
||||
* - max_total_bytes = 4 TiB (whole archive)
|
||||
* - max_ratio = 1000 (relaxed ratio cap; check_space still applies)
|
||||
*
|
||||
* Requires the user to opt in via the web UI (large=1) before these take
|
||||
* effect. Default limits must always be strictly smaller than large so the
|
||||
* large profile is unambiguously a relaxation.
|
||||
*/
|
||||
static const zipx_limits_t k_large_limits = {
|
||||
.max_entries = 500000,
|
||||
.max_total_bytes = 4ULL * 1024 * 1024 * 1024 * 1024,
|
||||
.max_file_bytes = 1ULL * 1024 * 1024 * 1024 * 1024,
|
||||
.max_ratio = 1000,
|
||||
.ratio_min_bytes = 1ULL * 1024 * 1024 * 1024,
|
||||
.max_depth = 32,
|
||||
.max_name_len = 255,
|
||||
.max_path_len = 1024
|
||||
};
|
||||
|
||||
const zipx_limits_t *
|
||||
zipx_default_limits(void) {
|
||||
return &k_default_limits;
|
||||
}
|
||||
|
||||
const zipx_limits_t *
|
||||
zipx_limits_profile(int profile) {
|
||||
switch(profile) {
|
||||
case ZIPX_LIMITS_LARGE:
|
||||
return &k_large_limits;
|
||||
case ZIPX_LIMITS_DEFAULT:
|
||||
default:
|
||||
return &k_default_limits;
|
||||
}
|
||||
}
|
||||
|
||||
const char *
|
||||
zipx_status_string(zipx_status_t status) {
|
||||
switch(status) {
|
||||
case ZIPX_OK: return "ok";
|
||||
case ZIPX_ERR_CANCELED: return "canceled";
|
||||
case ZIPX_ERR_OPEN: return "cannot open archive";
|
||||
case ZIPX_ERR_FORMAT: return "corrupt archive";
|
||||
case ZIPX_ERR_UNSUPPORTED: return "unsupported archive";
|
||||
case ZIPX_ERR_UNSAFE_NAME: return "unsafe entry name";
|
||||
case ZIPX_ERR_SPECIAL: return "unsupported entry type";
|
||||
case ZIPX_ERR_DUPLICATE: return "duplicate entry name";
|
||||
case ZIPX_ERR_LIMIT_ENTRIES: return "too many entries";
|
||||
case ZIPX_ERR_LIMIT_FILE: return "entry too large";
|
||||
case ZIPX_ERR_LIMIT_TOTAL: return "archive contents too large";
|
||||
case ZIPX_ERR_LIMIT_RATIO: return "compression ratio too high";
|
||||
case ZIPX_ERR_LIMIT_DEPTH: return "path too deep";
|
||||
case ZIPX_ERR_LIMIT_NAME: return "path too long";
|
||||
case ZIPX_ERR_LIMIT_DICT: return "dictionary too large";
|
||||
case ZIPX_ERR_CONFLICT: return "target already exists";
|
||||
case ZIPX_ERR_PASSWORD: return "password required or wrong";
|
||||
case ZIPX_ERR_SPACE: return "not enough space";
|
||||
case ZIPX_ERR_IO: return "read or write failed";
|
||||
case ZIPX_ERR_CRC: return "crc mismatch";
|
||||
default: return "internal error";
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,438 @@
|
||||
/* zipx_volstream -- present a multi-file archive volume set as one stream.
|
||||
part of ps5-web-file-manager
|
||||
|
||||
See zipx_volstream.h for the two split layouts this supports. */
|
||||
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
|
||||
#include "mz.h"
|
||||
#include "mz_os.h"
|
||||
#include "mz_strm.h"
|
||||
#include "mz_strm_os.h"
|
||||
|
||||
#include "zipx_volstream.h"
|
||||
|
||||
#define VOL_INT32_MAX 0x7fffffffLL
|
||||
|
||||
typedef struct {
|
||||
mz_stream stream; /* first member: callbacks cast the handle to this */
|
||||
char **paths; /* ordered part paths */
|
||||
int32_t count;
|
||||
int64_t *prefix; /* count + 1 entries, prefix[count] == total */
|
||||
int64_t total;
|
||||
int32_t mode; /* ZIPX_VOL_MODE_* */
|
||||
int32_t disk; /* active part index */
|
||||
int64_t pos; /* absolute position inside the concatenated set */
|
||||
int32_t os_part; /* part currently held by os, -1 when nothing is open */
|
||||
void *os;
|
||||
int32_t opened;
|
||||
int32_t error;
|
||||
} zipx_volstream_t;
|
||||
|
||||
/* ------------------------------------------------------------------------- */
|
||||
|
||||
static int32_t
|
||||
vol_use_part(zipx_volstream_t *v, int32_t part) {
|
||||
if(v->os_part == part) {
|
||||
return MZ_OK;
|
||||
}
|
||||
if(v->os_part >= 0) {
|
||||
mz_stream_close(v->os);
|
||||
v->os_part = -1;
|
||||
}
|
||||
if(mz_stream_open(v->os, v->paths[part], MZ_OPEN_MODE_READ) != MZ_OK) {
|
||||
v->error = MZ_OPEN_ERROR;
|
||||
return MZ_OPEN_ERROR;
|
||||
}
|
||||
v->os_part = part;
|
||||
return MZ_OK;
|
||||
}
|
||||
|
||||
/* Part holding an absolute offset, walking from `hint` (parts are laid out in
|
||||
order and reads are sequential, so this stays O(1) amortised). */
|
||||
static int32_t
|
||||
vol_part_of(zipx_volstream_t *v, int64_t pos, int32_t hint) {
|
||||
int32_t i;
|
||||
|
||||
if(pos < 0 || pos >= v->total) {
|
||||
return -1;
|
||||
}
|
||||
i = hint;
|
||||
if(i < 0) {
|
||||
i = 0;
|
||||
}
|
||||
if(i > v->count - 1) {
|
||||
i = v->count - 1;
|
||||
}
|
||||
while(i > 0 && pos < v->prefix[i]) {
|
||||
i--;
|
||||
}
|
||||
while(i < v->count - 1 && pos >= v->prefix[i + 1]) {
|
||||
i++;
|
||||
}
|
||||
return i;
|
||||
}
|
||||
|
||||
static int32_t
|
||||
vol_open(void *stream, const char *path, int32_t mode) {
|
||||
zipx_volstream_t *v = (zipx_volstream_t *)stream;
|
||||
int64_t sum = 0;
|
||||
int32_t i;
|
||||
|
||||
(void)path;
|
||||
(void)mode;
|
||||
if(!v || v->count <= 0 || !v->paths) {
|
||||
return MZ_PARAM_ERROR;
|
||||
}
|
||||
v->prefix = (int64_t *)calloc((size_t)v->count + 1, sizeof(*v->prefix));
|
||||
if(!v->prefix) {
|
||||
v->error = MZ_MEM_ERROR;
|
||||
return MZ_MEM_ERROR;
|
||||
}
|
||||
for(i = 0; i < v->count; i++) {
|
||||
int64_t size = mz_os_get_file_size(v->paths[i]);
|
||||
|
||||
if(size < 0) {
|
||||
v->error = MZ_OPEN_ERROR;
|
||||
return MZ_OPEN_ERROR;
|
||||
}
|
||||
v->prefix[i] = sum;
|
||||
sum += size;
|
||||
}
|
||||
v->prefix[v->count] = sum;
|
||||
v->total = sum;
|
||||
|
||||
if(!v->os) {
|
||||
v->os = mz_stream_os_create();
|
||||
if(!v->os) {
|
||||
v->error = MZ_MEM_ERROR;
|
||||
return MZ_MEM_ERROR;
|
||||
}
|
||||
}
|
||||
/* The end-of-central-directory record sits on the last disk, so a split
|
||||
disk set starts there; a byte split is read from its first byte. */
|
||||
v->disk = (v->mode == ZIPX_VOL_MODE_DISK) ? v->count - 1 : 0;
|
||||
v->pos = v->prefix[v->disk];
|
||||
v->os_part = -1;
|
||||
v->opened = 1;
|
||||
v->error = MZ_OK;
|
||||
return MZ_OK;
|
||||
}
|
||||
|
||||
static int32_t
|
||||
vol_is_open(void *stream) {
|
||||
zipx_volstream_t *v = (zipx_volstream_t *)stream;
|
||||
|
||||
return (v && v->opened) ? MZ_OK : MZ_OPEN_ERROR;
|
||||
}
|
||||
|
||||
static int32_t
|
||||
vol_read(void *stream, void *buf, int32_t size) {
|
||||
zipx_volstream_t *v = (zipx_volstream_t *)stream;
|
||||
int32_t done = 0;
|
||||
int32_t hint;
|
||||
|
||||
if(!v || !v->opened || !buf) {
|
||||
return MZ_PARAM_ERROR;
|
||||
}
|
||||
if(size <= 0) {
|
||||
return 0;
|
||||
}
|
||||
hint = v->os_part >= 0 ? v->os_part : 0;
|
||||
while(done < size) {
|
||||
int32_t part = vol_part_of(v, v->pos, hint);
|
||||
int64_t in_part;
|
||||
int64_t avail;
|
||||
int64_t want;
|
||||
int32_t got;
|
||||
|
||||
if(part < 0) {
|
||||
break; /* end of the logical archive */
|
||||
}
|
||||
hint = part;
|
||||
if(vol_use_part(v, part) != MZ_OK) {
|
||||
break;
|
||||
}
|
||||
in_part = v->pos - v->prefix[part];
|
||||
avail = (v->prefix[part + 1] - v->prefix[part]) - in_part;
|
||||
if(avail <= 0) { /* empty part: step over it */
|
||||
v->pos = v->prefix[part + 1];
|
||||
continue;
|
||||
}
|
||||
want = (int64_t)(size - done);
|
||||
if(want > avail) {
|
||||
want = avail;
|
||||
}
|
||||
if(want > VOL_INT32_MAX) {
|
||||
want = VOL_INT32_MAX;
|
||||
}
|
||||
if(mz_stream_tell(v->os) != in_part) {
|
||||
if(mz_stream_seek(v->os, in_part, MZ_SEEK_SET) != MZ_OK) {
|
||||
v->error = MZ_SEEK_ERROR;
|
||||
break;
|
||||
}
|
||||
}
|
||||
got = mz_stream_read(v->os, (uint8_t *)buf + done, (int32_t)want);
|
||||
if(got <= 0) {
|
||||
if(got < 0) {
|
||||
v->error = MZ_READ_ERROR;
|
||||
}
|
||||
break;
|
||||
}
|
||||
done += got;
|
||||
v->pos += got;
|
||||
if(got < (int32_t)want) {
|
||||
break; /* short read: let the caller come back */
|
||||
}
|
||||
}
|
||||
if(done == 0 && v->error != MZ_OK) {
|
||||
return v->error;
|
||||
}
|
||||
return done;
|
||||
}
|
||||
|
||||
static int32_t
|
||||
vol_write(void *stream, const void *buf, int32_t size) {
|
||||
(void)stream;
|
||||
(void)buf;
|
||||
(void)size;
|
||||
return MZ_SUPPORT_ERROR;
|
||||
}
|
||||
|
||||
static int64_t
|
||||
vol_tell(void *stream) {
|
||||
zipx_volstream_t *v = (zipx_volstream_t *)stream;
|
||||
|
||||
if(!v || !v->opened) {
|
||||
return -1;
|
||||
}
|
||||
if(v->mode == ZIPX_VOL_MODE_DISK) {
|
||||
return v->pos - v->prefix[v->disk];
|
||||
}
|
||||
return v->pos;
|
||||
}
|
||||
|
||||
static int32_t
|
||||
vol_seek(void *stream, int64_t offset, int32_t origin) {
|
||||
zipx_volstream_t *v = (zipx_volstream_t *)stream;
|
||||
int64_t target;
|
||||
|
||||
if(!v || !v->opened) {
|
||||
return MZ_PARAM_ERROR;
|
||||
}
|
||||
if(origin == MZ_SEEK_SET) {
|
||||
target = offset;
|
||||
if(v->mode == ZIPX_VOL_MODE_DISK) {
|
||||
target += v->prefix[v->disk]; /* offsets are disk relative there */
|
||||
}
|
||||
} else if(origin == MZ_SEEK_CUR) {
|
||||
target = v->pos + offset;
|
||||
} else if(origin == MZ_SEEK_END) {
|
||||
target = v->total + offset;
|
||||
} else {
|
||||
return MZ_PARAM_ERROR;
|
||||
}
|
||||
if(target < 0) {
|
||||
target = 0;
|
||||
}
|
||||
if(target > v->total) {
|
||||
target = v->total;
|
||||
}
|
||||
v->pos = target;
|
||||
return MZ_OK;
|
||||
}
|
||||
|
||||
static int32_t
|
||||
vol_close(void *stream) {
|
||||
zipx_volstream_t *v = (zipx_volstream_t *)stream;
|
||||
|
||||
if(!v) {
|
||||
return MZ_PARAM_ERROR;
|
||||
}
|
||||
if(v->os && v->os_part >= 0) {
|
||||
mz_stream_close(v->os);
|
||||
v->os_part = -1;
|
||||
}
|
||||
v->opened = 0;
|
||||
return MZ_OK;
|
||||
}
|
||||
|
||||
static int32_t
|
||||
vol_error(void *stream) {
|
||||
zipx_volstream_t *v = (zipx_volstream_t *)stream;
|
||||
|
||||
return v ? v->error : MZ_PARAM_ERROR;
|
||||
}
|
||||
|
||||
static int32_t
|
||||
vol_get_prop(void *stream, int32_t prop, int64_t *value) {
|
||||
zipx_volstream_t *v = (zipx_volstream_t *)stream;
|
||||
|
||||
if(!v || !value) {
|
||||
return MZ_PARAM_ERROR;
|
||||
}
|
||||
if(v->mode != ZIPX_VOL_MODE_DISK) {
|
||||
/* A byte split keeps absolute offsets, so the disk properties must look
|
||||
unsupported: minizip-ng then leaves every offset alone. */
|
||||
return MZ_PARAM_ERROR;
|
||||
}
|
||||
if(prop == MZ_STREAM_PROP_DISK_NUMBER) {
|
||||
*value = v->disk;
|
||||
return MZ_OK;
|
||||
}
|
||||
if(prop == MZ_STREAM_PROP_DISK_SIZE) {
|
||||
*value = v->prefix[v->disk + 1] - v->prefix[v->disk];
|
||||
return MZ_OK;
|
||||
}
|
||||
return MZ_PARAM_ERROR;
|
||||
}
|
||||
|
||||
static int32_t
|
||||
vol_set_prop(void *stream, int32_t prop, int64_t value) {
|
||||
zipx_volstream_t *v = (zipx_volstream_t *)stream;
|
||||
|
||||
if(!v) {
|
||||
return MZ_PARAM_ERROR;
|
||||
}
|
||||
if(prop != MZ_STREAM_PROP_DISK_NUMBER || v->mode != ZIPX_VOL_MODE_DISK) {
|
||||
return MZ_PARAM_ERROR;
|
||||
}
|
||||
if(value < 0) {
|
||||
/* minizip-ng passes -1 for entries that live on the same disk as the
|
||||
central directory, which is the final volume of the set. */
|
||||
v->disk = v->count - 1;
|
||||
v->pos = v->prefix[v->disk];
|
||||
return MZ_OK;
|
||||
}
|
||||
if(value >= v->count) {
|
||||
return MZ_PARAM_ERROR;
|
||||
}
|
||||
v->disk = (int32_t)value;
|
||||
v->pos = v->prefix[v->disk];
|
||||
return MZ_OK;
|
||||
}
|
||||
|
||||
/* ------------------------------------------------------------------------- */
|
||||
|
||||
static void vol_destroy(void **stream);
|
||||
|
||||
static mz_stream_vtbl vol_vtbl = {
|
||||
vol_open, vol_is_open, vol_read, vol_write, vol_tell, vol_seek, vol_close,
|
||||
vol_error, NULL, vol_destroy, vol_get_prop, vol_set_prop,
|
||||
};
|
||||
|
||||
static void *
|
||||
vol_create(void) {
|
||||
zipx_volstream_t *v = (zipx_volstream_t *)calloc(1, sizeof(*v));
|
||||
|
||||
if(!v) {
|
||||
return NULL;
|
||||
}
|
||||
v->stream.vtbl = &vol_vtbl;
|
||||
v->os_part = -1;
|
||||
return (void *)&v->stream;
|
||||
}
|
||||
|
||||
static void
|
||||
vol_destroy(void **stream) {
|
||||
zipx_volstream_t *v;
|
||||
int32_t i;
|
||||
|
||||
if(!stream || !*stream) {
|
||||
return;
|
||||
}
|
||||
v = (zipx_volstream_t *)*stream;
|
||||
vol_close(&v->stream);
|
||||
if(v->os) {
|
||||
mz_stream_os_delete(&v->os);
|
||||
}
|
||||
if(v->paths) {
|
||||
for(i = 0; i < v->count; i++) {
|
||||
free(v->paths[i]);
|
||||
}
|
||||
free(v->paths);
|
||||
}
|
||||
free(v->prefix);
|
||||
free(v);
|
||||
*stream = NULL;
|
||||
}
|
||||
|
||||
/* ------------------------------------------------------------------------- */
|
||||
|
||||
void *
|
||||
zipx_volstream_create(int32_t mode) {
|
||||
void *stream = vol_create();
|
||||
zipx_volstream_t *v = (zipx_volstream_t *)stream;
|
||||
|
||||
if(!v) {
|
||||
return NULL;
|
||||
}
|
||||
v->mode = (mode == ZIPX_VOL_MODE_DISK) ? ZIPX_VOL_MODE_DISK :
|
||||
ZIPX_VOL_MODE_CONCAT;
|
||||
return stream;
|
||||
}
|
||||
|
||||
void
|
||||
zipx_volstream_delete(void **stream) {
|
||||
mz_stream_delete(stream); /* routes through vtbl->destroy */
|
||||
}
|
||||
|
||||
int32_t
|
||||
zipx_volstream_set_parts(void *stream, const char *const *paths, int32_t count) {
|
||||
zipx_volstream_t *v = (zipx_volstream_t *)stream;
|
||||
int32_t i;
|
||||
|
||||
if(!v || !paths || count <= 0) {
|
||||
return MZ_PARAM_ERROR;
|
||||
}
|
||||
v->paths = (char **)calloc((size_t)count, sizeof(*v->paths));
|
||||
if(!v->paths) {
|
||||
return MZ_MEM_ERROR;
|
||||
}
|
||||
v->count = count;
|
||||
for(i = 0; i < count; i++) {
|
||||
size_t len = strlen(paths[i]) + 1;
|
||||
|
||||
v->paths[i] = (char *)malloc(len);
|
||||
if(!v->paths[i]) {
|
||||
return MZ_MEM_ERROR;
|
||||
}
|
||||
memcpy(v->paths[i], paths[i], len);
|
||||
}
|
||||
return MZ_OK;
|
||||
}
|
||||
|
||||
int64_t
|
||||
zipx_volstream_total(void *stream) {
|
||||
zipx_volstream_t *v = (zipx_volstream_t *)stream;
|
||||
|
||||
return v ? v->total : -1;
|
||||
}
|
||||
|
||||
const char *
|
||||
zipx_volstream_describe(void *stream, char *buf, unsigned int size) {
|
||||
zipx_volstream_t *v = (zipx_volstream_t *)stream;
|
||||
const char *base;
|
||||
|
||||
if(!buf || size == 0) {
|
||||
return "";
|
||||
}
|
||||
if(!v || v->count <= 0) {
|
||||
snprintf(buf, size, "(no volumes)");
|
||||
return buf;
|
||||
}
|
||||
base = strrchr(v->paths[0], '/');
|
||||
#ifdef _WIN32
|
||||
{
|
||||
const char *alt = strrchr(v->paths[0], '\\');
|
||||
if(alt && (!base || alt > base)) {
|
||||
base = alt;
|
||||
}
|
||||
}
|
||||
#endif
|
||||
base = base ? base + 1 : v->paths[0];
|
||||
snprintf(buf, size, "%s (%d volumes)", base, (int)v->count);
|
||||
return buf;
|
||||
}
|
||||
@@ -0,0 +1,56 @@
|
||||
/* zipx_volstream -- present a multi-file archive volume set as one stream.
|
||||
part of ps5-web-file-manager
|
||||
|
||||
Two split layouts exist in the wild and they need different behaviour:
|
||||
|
||||
ZIPX_VOL_MODE_CONCAT (byte split)
|
||||
`name.zip.001`, `name.zip.002`, ... (7-Zip) and `name.part1.zip`,
|
||||
`name.part2.zip` (WinRAR). Each part is a byte slice of one archive, so
|
||||
byte N of the logical archive is byte N of the concatenation and every
|
||||
offset stored inside the archive is already absolute. Offsets are passed
|
||||
through untouched and the disk properties are reported as unsupported so
|
||||
minizip-ng keeps using absolute offsets.
|
||||
|
||||
ZIPX_VOL_MODE_DISK (zip split disks)
|
||||
`name.z01`, `name.z02`, ..., `name.zip` (Info-ZIP / PKZIP style). The
|
||||
central directory stores the offset of a local header relative to the
|
||||
disk it starts on, so minizip-ng switches the active disk through
|
||||
MZ_STREAM_PROP_DISK_NUMBER before seeking. Seek/tell are relative to the
|
||||
active disk here, which is exactly what mz_zip_entry_seek_local_header
|
||||
expects, and the stream starts on the last disk because that is where the
|
||||
end-of-central-directory record lives. */
|
||||
|
||||
#ifndef ZIPX_VOLSTREAM_H
|
||||
#define ZIPX_VOLSTREAM_H
|
||||
|
||||
#include <stdint.h>
|
||||
|
||||
#include "zipx_volume.h"
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
/* Creates a stream handle; pass it to mz_stream_open() afterwards. */
|
||||
void *zipx_volstream_create(int32_t mode);
|
||||
|
||||
/* Deletes a handle created above (safe with *stream == NULL). */
|
||||
void zipx_volstream_delete(void **stream);
|
||||
|
||||
/* Copies the ordered part paths into the handle. Must be called before the
|
||||
stream is opened. Returns MZ_OK (0) or MZ_MEM_ERROR (-4). */
|
||||
int32_t zipx_volstream_set_parts(void *stream, const char *const *paths,
|
||||
int32_t count);
|
||||
|
||||
/* Total logical size (sum of the part sizes), or -1 when not resolved. */
|
||||
int64_t zipx_volstream_total(void *stream);
|
||||
|
||||
/* Human readable description of the set, e.g. "name.z01 (3 volumes)".
|
||||
Writes into buf and returns buf. */
|
||||
const char *zipx_volstream_describe(void *stream, char *buf, unsigned int size);
|
||||
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
#endif
|
||||
|
||||
#endif
|
||||
@@ -0,0 +1,600 @@
|
||||
/* zipx_volume -- see zipx_volume.h for what this groups and why. */
|
||||
|
||||
#include <ctype.h>
|
||||
#include <dirent.h>
|
||||
#include <stdarg.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <strings.h>
|
||||
#include <sys/stat.h>
|
||||
|
||||
#include "zipx_volume.h"
|
||||
|
||||
typedef struct {
|
||||
long num;
|
||||
char *path;
|
||||
} vol_part_t;
|
||||
|
||||
/* ------------------------------------------------------------------------- */
|
||||
/* path helpers */
|
||||
|
||||
static const char *
|
||||
file_base(const char *path) {
|
||||
const char *slash = strrchr(path, '/');
|
||||
const char *back = strrchr(path, '\\');
|
||||
|
||||
if(back && (!slash || back > slash)) {
|
||||
slash = back;
|
||||
}
|
||||
return slash ? slash + 1 : path;
|
||||
}
|
||||
|
||||
/* Directory part without a trailing separator; "" for a bare filename. */
|
||||
static void
|
||||
file_dir(const char *path, char *buf, size_t size) {
|
||||
const char *base = file_base(path);
|
||||
size_t len = (size_t)(base - path);
|
||||
|
||||
while(len > 0 && (path[len - 1] == '/' || path[len - 1] == '\\')) {
|
||||
len--;
|
||||
}
|
||||
if(len >= size) {
|
||||
len = size - 1;
|
||||
}
|
||||
memcpy(buf, path, len);
|
||||
buf[len] = 0;
|
||||
}
|
||||
|
||||
static int
|
||||
ends_with_ci(const char *s, const char *suffix) {
|
||||
size_t ls = strlen(s);
|
||||
size_t lf = strlen(suffix);
|
||||
|
||||
if(lf > ls) {
|
||||
return 0;
|
||||
}
|
||||
return strcasecmp(s + ls - lf, suffix) == 0;
|
||||
}
|
||||
|
||||
static int
|
||||
file_exists(const char *path) {
|
||||
struct stat st;
|
||||
|
||||
return stat(path, &st) == 0;
|
||||
}
|
||||
|
||||
static char *
|
||||
vol_join(const char *dir, const char *name) {
|
||||
size_t need = strlen(dir) + strlen(name) + 2;
|
||||
char *out = (char *)malloc(need);
|
||||
|
||||
if(!out) {
|
||||
return NULL;
|
||||
}
|
||||
if(dir[0]) {
|
||||
snprintf(out, need, "%s/%s", dir, name);
|
||||
} else {
|
||||
snprintf(out, need, "%s", name);
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
static char *
|
||||
err_printf(const char *fmt, ...) {
|
||||
va_list ap;
|
||||
char *buf;
|
||||
int need;
|
||||
|
||||
va_start(ap, fmt);
|
||||
need = vsnprintf(NULL, 0, fmt, ap);
|
||||
va_end(ap);
|
||||
if(need < 0) {
|
||||
return NULL;
|
||||
}
|
||||
buf = (char *)malloc((size_t)need + 1);
|
||||
if(!buf) {
|
||||
return NULL;
|
||||
}
|
||||
va_start(ap, fmt);
|
||||
vsnprintf(buf, (size_t)need + 1, fmt, ap);
|
||||
va_end(ap);
|
||||
return buf;
|
||||
}
|
||||
|
||||
/* ------------------------------------------------------------------------- */
|
||||
/* part collection */
|
||||
|
||||
/* Matches "PREFIX<digits>SUFFIX"; returns 1 and the value on a match. */
|
||||
static int
|
||||
match_numbered(const char *entry, const char *prefix, const char *suffix,
|
||||
long *num) {
|
||||
size_t plen = strlen(prefix);
|
||||
size_t slen = strlen(suffix);
|
||||
const char *p;
|
||||
char *end = NULL;
|
||||
long value;
|
||||
|
||||
if(strncmp(entry, prefix, plen) != 0) {
|
||||
return 0;
|
||||
}
|
||||
p = entry + plen;
|
||||
if(!isdigit((unsigned char)*p)) {
|
||||
return 0;
|
||||
}
|
||||
value = strtol(p, &end, 10);
|
||||
if(end == p) {
|
||||
return 0;
|
||||
}
|
||||
if(slen > 0) {
|
||||
if(strcmp(end, suffix) != 0) {
|
||||
return 0;
|
||||
}
|
||||
} else if(*end != 0) {
|
||||
return 0;
|
||||
}
|
||||
*num = value;
|
||||
return 1;
|
||||
}
|
||||
|
||||
static int
|
||||
part_compare(const void *a, const void *b) {
|
||||
const vol_part_t *pa = (const vol_part_t *)a;
|
||||
const vol_part_t *pb = (const vol_part_t *)b;
|
||||
|
||||
if(pa->num < pb->num) {
|
||||
return -1;
|
||||
}
|
||||
if(pa->num > pb->num) {
|
||||
return 1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
static void
|
||||
free_parts(vol_part_t *parts, int count) {
|
||||
int i;
|
||||
|
||||
for(i = 0; i < count; i++) {
|
||||
free(parts[i].path);
|
||||
parts[i].path = NULL;
|
||||
}
|
||||
}
|
||||
|
||||
/* Collects every entry in `dir` matching PREFIX<digits>SUFFIX, ordered by the
|
||||
number. Returns the count, or -1 when the directory cannot be listed (with
|
||||
*err set) or the set is larger than `max`. */
|
||||
static int
|
||||
scan_parts(const char *dir, const char *prefix, const char *suffix,
|
||||
vol_part_t *parts, int max, char **err) {
|
||||
const char *target = dir[0] ? dir : ".";
|
||||
DIR *d = opendir(target);
|
||||
struct dirent *ent;
|
||||
int count = 0;
|
||||
|
||||
if(!d) {
|
||||
if(err && !*err) {
|
||||
*err = err_printf("cannot list the directory '%s' that holds the other "
|
||||
"volumes", target);
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
while((ent = readdir(d)) != NULL) {
|
||||
long num = 0;
|
||||
|
||||
if(!match_numbered(ent->d_name, prefix, suffix, &num)) {
|
||||
continue;
|
||||
}
|
||||
if(count >= max) {
|
||||
if(err && !*err) {
|
||||
*err = err_printf("volume set has more than %d parts, the supported "
|
||||
"maximum", max);
|
||||
}
|
||||
free_parts(parts, count);
|
||||
closedir(d);
|
||||
return -1;
|
||||
}
|
||||
parts[count].num = num;
|
||||
parts[count].path = vol_join(dir, ent->d_name);
|
||||
if(!parts[count].path) {
|
||||
free_parts(parts, count + 1);
|
||||
closedir(d);
|
||||
return -1;
|
||||
}
|
||||
count++;
|
||||
}
|
||||
closedir(d);
|
||||
qsort(parts, (size_t)count, sizeof(*parts), part_compare);
|
||||
return count;
|
||||
}
|
||||
|
||||
/* Verifies the parts are numbered 1..count with no gap (and no duplicate),
|
||||
filling *err with the exact missing name when they are not. */
|
||||
static int
|
||||
check_contiguous(vol_part_t *parts, int count, const char *dir,
|
||||
const char *prefix, const char *suffix, int width,
|
||||
char **err) {
|
||||
int i;
|
||||
|
||||
for(i = 0; i < count; i++) {
|
||||
if(parts[i].num != (long)(i + 1)) {
|
||||
if(err && !*err) {
|
||||
char name[512];
|
||||
char *full;
|
||||
|
||||
snprintf(name, sizeof(name), "%s%0*ld%s", prefix, width,
|
||||
(long)(i + 1), suffix);
|
||||
full = vol_join(dir, name);
|
||||
*err = err_printf("volume set is incomplete: '%s' is missing",
|
||||
full ? full : name);
|
||||
free(full);
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* ------------------------------------------------------------------------- */
|
||||
/* volume set bookkeeping */
|
||||
|
||||
static void
|
||||
volume_reset(zipx_volume_t *vol) {
|
||||
memset(vol, 0, sizeof(*vol));
|
||||
vol->index = -1;
|
||||
}
|
||||
|
||||
static int
|
||||
volume_take(zipx_volume_t *vol, vol_part_t *parts, int count, int mode,
|
||||
const char *selected) {
|
||||
int i;
|
||||
|
||||
vol->paths = (char **)calloc((size_t)count, sizeof(*vol->paths));
|
||||
if(!vol->paths) {
|
||||
return -1;
|
||||
}
|
||||
for(i = 0; i < count; i++) {
|
||||
vol->paths[i] = parts[i].path;
|
||||
parts[i].path = NULL; /* ownership moves into vol */
|
||||
}
|
||||
vol->count = count;
|
||||
vol->mode = mode;
|
||||
vol->is_set = 1;
|
||||
vol->index = -1;
|
||||
for(i = 0; i < count; i++) {
|
||||
if(selected && strcmp(vol->paths[i], selected) == 0) {
|
||||
vol->index = i;
|
||||
break;
|
||||
}
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* ------------------------------------------------------------------------- */
|
||||
/* detection: name.zip.001 / name.7z.001 / name.rar.001 */
|
||||
|
||||
static int
|
||||
detect_digit_suffix(const char *dir, const char *base, const char *selected,
|
||||
zipx_volume_t *vol, char **err) {
|
||||
static const char *const known[] = { "zip", "7z", "rar", NULL };
|
||||
const char *dot = strrchr(base, '.');
|
||||
vol_part_t parts[ZIPX_VOL_MAX_PARTS];
|
||||
char stem[2048];
|
||||
char prefix[2100];
|
||||
size_t stem_len;
|
||||
int count;
|
||||
int i;
|
||||
int known_ext = 0;
|
||||
|
||||
if(!dot || dot == base || !dot[1]) {
|
||||
return 0;
|
||||
}
|
||||
for(i = 1; dot[i]; i++) {
|
||||
if(!isdigit((unsigned char)dot[i])) {
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
stem_len = (size_t)(dot - base);
|
||||
if(stem_len >= sizeof(stem)) {
|
||||
return 0;
|
||||
}
|
||||
memcpy(stem, base, stem_len);
|
||||
stem[stem_len] = 0;
|
||||
for(i = 0; known[i]; i++) {
|
||||
char tail[8];
|
||||
|
||||
snprintf(tail, sizeof(tail), ".%s", known[i]);
|
||||
if(ends_with_ci(stem, tail)) {
|
||||
known_ext = 1;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if(!known_ext) {
|
||||
return 0; /* "backup.001" style names are not archive volumes */
|
||||
}
|
||||
snprintf(prefix, sizeof(prefix), "%s.", stem);
|
||||
count = scan_parts(dir, prefix, "", parts, ZIPX_VOL_MAX_PARTS, err);
|
||||
if(count < 0) {
|
||||
return -1;
|
||||
}
|
||||
if(count <= 1) {
|
||||
free_parts(parts, count > 0 ? count : 0);
|
||||
if(count == 1 && err && !*err) {
|
||||
*err = err_printf("'%s' is the first volume of a split archive but no "
|
||||
"other volumes ('%s.002', ...) are present", base,
|
||||
stem);
|
||||
}
|
||||
return count == 1 ? -1 : 0;
|
||||
}
|
||||
if(check_contiguous(parts, count, dir, prefix, "", 3, err)) {
|
||||
free_parts(parts, count);
|
||||
return -1;
|
||||
}
|
||||
if(volume_take(vol, parts, count, ZIPX_VOL_MODE_CONCAT, selected)) {
|
||||
free_parts(parts, count);
|
||||
return -1;
|
||||
}
|
||||
return 1;
|
||||
}
|
||||
|
||||
/* ------------------------------------------------------------------------- */
|
||||
/* detection: name.z01 ... name.zip */
|
||||
|
||||
static int
|
||||
detect_z_suffix(const char *dir, const char *base, const char *selected,
|
||||
zipx_volume_t *vol, char **err) {
|
||||
const char *dot = strrchr(base, '.');
|
||||
vol_part_t parts[ZIPX_VOL_MAX_PARTS];
|
||||
char stem[2048];
|
||||
char prefix[2100];
|
||||
size_t stem_len;
|
||||
int count;
|
||||
|
||||
if(!dot || dot == base) {
|
||||
return 0;
|
||||
}
|
||||
if((dot[1] != 'z' && dot[1] != 'Z') || !isdigit((unsigned char)dot[2])) {
|
||||
return 0;
|
||||
}
|
||||
{
|
||||
int i;
|
||||
|
||||
for(i = 2; dot[i]; i++) {
|
||||
if(!isdigit((unsigned char)dot[i])) {
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
stem_len = (size_t)(dot - base);
|
||||
if(stem_len >= sizeof(stem)) {
|
||||
return 0;
|
||||
}
|
||||
memcpy(stem, base, stem_len);
|
||||
stem[stem_len] = 0;
|
||||
|
||||
snprintf(prefix, sizeof(prefix), "%s.z", stem);
|
||||
count = scan_parts(dir, prefix, "", parts, ZIPX_VOL_MAX_PARTS - 1, err);
|
||||
if(count < 0) {
|
||||
return -1;
|
||||
}
|
||||
if(count == 0) {
|
||||
return 0;
|
||||
}
|
||||
if(check_contiguous(parts, count, dir, prefix, "", 2, err)) {
|
||||
free_parts(parts, count);
|
||||
return -1;
|
||||
}
|
||||
/* The central directory always lives in "name.zip", the final volume. */
|
||||
{
|
||||
char name[2100];
|
||||
char *last;
|
||||
|
||||
snprintf(name, sizeof(name), "%s.zip", stem);
|
||||
last = vol_join(dir, name);
|
||||
if(!last) {
|
||||
free_parts(parts, count);
|
||||
return -1;
|
||||
}
|
||||
if(!file_exists(last)) {
|
||||
if(err && !*err) {
|
||||
*err = err_printf("volume set is incomplete: the last volume '%s' that "
|
||||
"holds the archive index is missing", name);
|
||||
}
|
||||
free(last);
|
||||
free_parts(parts, count);
|
||||
return -1;
|
||||
}
|
||||
parts[count].num = (long)count + 1;
|
||||
parts[count].path = last;
|
||||
count++;
|
||||
}
|
||||
if(volume_take(vol, parts, count, ZIPX_VOL_MODE_DISK, selected)) {
|
||||
free_parts(parts, count);
|
||||
return -1;
|
||||
}
|
||||
return 1;
|
||||
}
|
||||
|
||||
/* ------------------------------------------------------------------------- */
|
||||
/* detection: the final "name.zip" of a name.z01 ... name.zip set */
|
||||
|
||||
static int
|
||||
detect_zip_tail(const char *dir, const char *base, const char *selected,
|
||||
zipx_volume_t *vol, char **err) {
|
||||
vol_part_t parts[ZIPX_VOL_MAX_PARTS];
|
||||
char stem[2048];
|
||||
char prefix[2100];
|
||||
size_t stem_len;
|
||||
int count;
|
||||
|
||||
if(!ends_with_ci(base, ".zip")) {
|
||||
return 0;
|
||||
}
|
||||
stem_len = strlen(base) - 4;
|
||||
if(stem_len == 0 || stem_len >= sizeof(stem)) {
|
||||
return 0;
|
||||
}
|
||||
memcpy(stem, base, stem_len);
|
||||
stem[stem_len] = 0;
|
||||
|
||||
snprintf(prefix, sizeof(prefix), "%s.z", stem);
|
||||
count = scan_parts(dir, prefix, "", parts, ZIPX_VOL_MAX_PARTS - 1, err);
|
||||
if(count < 0) {
|
||||
return -1;
|
||||
}
|
||||
if(count == 0) {
|
||||
return 0; /* an ordinary single volume archive */
|
||||
}
|
||||
if(check_contiguous(parts, count, dir, prefix, "", 2, err)) {
|
||||
free_parts(parts, count);
|
||||
return -1;
|
||||
}
|
||||
{
|
||||
char *last = vol_join(dir, base);
|
||||
|
||||
if(!last) {
|
||||
free_parts(parts, count);
|
||||
return -1;
|
||||
}
|
||||
parts[count].num = (long)count + 1;
|
||||
parts[count].path = last;
|
||||
count++;
|
||||
}
|
||||
if(volume_take(vol, parts, count, ZIPX_VOL_MODE_DISK, selected)) {
|
||||
free_parts(parts, count);
|
||||
return -1;
|
||||
}
|
||||
return 1;
|
||||
}
|
||||
|
||||
/* ------------------------------------------------------------------------- */
|
||||
/* detection: name.part1.zip ... */
|
||||
|
||||
static int
|
||||
detect_part_suffix(const char *dir, const char *base, const char *selected,
|
||||
zipx_volume_t *vol, char **err) {
|
||||
vol_part_t parts[ZIPX_VOL_MAX_PARTS];
|
||||
char stem[2048];
|
||||
char prefix[2100];
|
||||
char trimmed[2048];
|
||||
const char *dot;
|
||||
size_t len;
|
||||
int count;
|
||||
|
||||
if(!ends_with_ci(base, ".zip")) {
|
||||
return 0;
|
||||
}
|
||||
len = strlen(base) - 4;
|
||||
if(len == 0 || len >= sizeof(trimmed)) {
|
||||
return 0;
|
||||
}
|
||||
memcpy(trimmed, base, len);
|
||||
trimmed[len] = 0;
|
||||
dot = strrchr(trimmed, '.');
|
||||
if(!dot || dot == trimmed || strncasecmp(dot, ".part", 5) != 0 ||
|
||||
!isdigit((unsigned char)dot[5])) {
|
||||
return 0;
|
||||
}
|
||||
{
|
||||
size_t stem_len = (size_t)(dot - trimmed);
|
||||
|
||||
if(stem_len >= sizeof(stem)) {
|
||||
return 0;
|
||||
}
|
||||
memcpy(stem, trimmed, stem_len);
|
||||
stem[stem_len] = 0;
|
||||
}
|
||||
snprintf(prefix, sizeof(prefix), "%s.part", stem);
|
||||
count = scan_parts(dir, prefix, ".zip", parts, ZIPX_VOL_MAX_PARTS, err);
|
||||
if(count < 0) {
|
||||
return -1;
|
||||
}
|
||||
if(count <= 1) {
|
||||
free_parts(parts, count > 0 ? count : 0);
|
||||
if(count == 1 && err && !*err) {
|
||||
*err = err_printf("'%s' is a volume of a split archive but the other "
|
||||
"volumes ('%s.part1.zip', ...) are missing", base, stem);
|
||||
}
|
||||
return count == 1 ? -1 : 0;
|
||||
}
|
||||
if(check_contiguous(parts, count, dir, prefix, ".zip", 1, err)) {
|
||||
free_parts(parts, count);
|
||||
return -1;
|
||||
}
|
||||
if(volume_take(vol, parts, count, ZIPX_VOL_MODE_DISK, selected)) {
|
||||
free_parts(parts, count);
|
||||
return -1;
|
||||
}
|
||||
return 1;
|
||||
}
|
||||
|
||||
/* ------------------------------------------------------------------------- */
|
||||
|
||||
int
|
||||
zipx_volume_detect(const char *path, zipx_volume_t *out, char **err) {
|
||||
char dir[4096];
|
||||
const char *base;
|
||||
int rc;
|
||||
|
||||
if(err) {
|
||||
*err = NULL;
|
||||
}
|
||||
if(!path || !out) {
|
||||
return -1;
|
||||
}
|
||||
volume_reset(out);
|
||||
base = file_base(path);
|
||||
file_dir(path, dir, sizeof(dir));
|
||||
|
||||
rc = detect_digit_suffix(dir, base, path, out, err);
|
||||
if(rc != 0) {
|
||||
return rc;
|
||||
}
|
||||
rc = detect_z_suffix(dir, base, path, out, err);
|
||||
if(rc != 0) {
|
||||
return rc;
|
||||
}
|
||||
rc = detect_zip_tail(dir, base, path, out, err);
|
||||
if(rc != 0) {
|
||||
return rc;
|
||||
}
|
||||
return detect_part_suffix(dir, base, path, out, err);
|
||||
}
|
||||
|
||||
void
|
||||
zipx_volume_free(zipx_volume_t *vol) {
|
||||
int i;
|
||||
|
||||
if(!vol) {
|
||||
return;
|
||||
}
|
||||
if(vol->paths) {
|
||||
for(i = 0; i < vol->count; i++) {
|
||||
free(vol->paths[i]);
|
||||
}
|
||||
free(vol->paths);
|
||||
}
|
||||
memset(vol, 0, sizeof(*vol));
|
||||
vol->index = -1;
|
||||
}
|
||||
|
||||
int
|
||||
zipx_volume_is_first(const char *path) {
|
||||
const char *base = file_base(path);
|
||||
const char *dot = strrchr(base, '.');
|
||||
|
||||
if(!dot) {
|
||||
return 0;
|
||||
}
|
||||
if(strcmp(dot, ".001") == 0) {
|
||||
return 1;
|
||||
}
|
||||
if((dot[1] == 'z' || dot[1] == 'Z') && dot[2] == '0' && dot[3] == '1' &&
|
||||
dot[4] == 0) {
|
||||
return 1;
|
||||
}
|
||||
if(ends_with_ci(base, ".part1.zip")) {
|
||||
return 1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,58 @@
|
||||
/* zipx_volume -- group a multi-file archive volume set into an ordered list.
|
||||
part of ps5-web-file-manager
|
||||
|
||||
Supports the naming conventions seen in the wild:
|
||||
|
||||
name.zip.001, name.zip.002, ... byte split (7-Zip "split to volumes")
|
||||
name.part1.zip, name.part2.zip byte split (WinRAR zip volumes)
|
||||
name.z01, name.z02, ..., name.zip zip split disks (Info-ZIP / PKZIP)
|
||||
|
||||
A caller can hand in any member of the set (the user usually clicks one file
|
||||
in the browser) and gets back the full ordered list plus the split layout so
|
||||
the engine can pick the matching stream behaviour. */
|
||||
|
||||
#ifndef ZIPX_VOLUME_H
|
||||
#define ZIPX_VOLUME_H
|
||||
|
||||
#include <stdint.h>
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
#define ZIPX_VOL_MAX_PARTS 512
|
||||
|
||||
/* How the set is split. A byte split (`name.zip.001`, `name.7z.001`,
|
||||
`name.part1.zip`) is the concatenation of its parts with absolute offsets;
|
||||
zip split disks (`name.z01` + `name.zip`) store per-disk offsets instead.
|
||||
The constants live here because they describe the *set*, and every consumer
|
||||
of zipx_volume_t needs them. */
|
||||
#define ZIPX_VOL_MODE_CONCAT 0
|
||||
#define ZIPX_VOL_MODE_DISK 1
|
||||
|
||||
typedef struct {
|
||||
char **paths; /* ordered part paths, owned by this struct */
|
||||
int count;
|
||||
int index; /* position of the path that was handed in (-1 unknown) */
|
||||
int mode; /* ZIPX_VOL_MODE_CONCAT or ZIPX_VOL_MODE_DISK */
|
||||
int is_set; /* 1 when the path is part of a multi-file set */
|
||||
} zipx_volume_t;
|
||||
|
||||
/* Inspects `path`: 1 when it belongs to a multi-file set (out is filled),
|
||||
0 when it is an ordinary single file (out is cleared), -1 on a hard error
|
||||
(*err, when non-NULL, receives a malloc'd message the caller must free;
|
||||
it is also set for the 0 case when a sibling set looks broken, so callers
|
||||
can surface "volumes are incomplete" instead of a generic open failure). */
|
||||
int zipx_volume_detect(const char *path, zipx_volume_t *out, char **err);
|
||||
|
||||
void zipx_volume_free(zipx_volume_t *vol);
|
||||
|
||||
/* True when `path` looks like the first volume of a set ("x.zip.001",
|
||||
"x.z01", "x.part1.zip"), used by the UI to label the entry. */
|
||||
int zipx_volume_is_first(const char *path);
|
||||
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
#endif
|
||||
|
||||
#endif
|
||||
@@ -0,0 +1,15 @@
|
||||
#!/usr/bin/env bash
|
||||
# Thin wrapper -- the actual driver is tests/bench_driver.py.
|
||||
#
|
||||
# /usr/bin/bash tests/bench-sevenz.sh [--big] [--runs N]
|
||||
#
|
||||
# bash is deliberately not used for timing here. In this sandbox every `date`
|
||||
# costs ~350 ms, so a t0/t1 pair injects ~700 ms of overhead into a
|
||||
# measurement whose real value is ~600 ms, and `time`'s user/sys accounting
|
||||
# does not see into the native child at all (it reported 31 ms of CPU for a
|
||||
# run that demonstrably decodes 82 MiB). Python reads a monotonic clock
|
||||
# around a single spawn per sample instead.
|
||||
|
||||
set -e
|
||||
ROOT="$(cd "$(dirname "$0")/.." && pwd -W 2>/dev/null || pwd)"
|
||||
exec python "$ROOT/tests/bench_driver.py" "$@"
|
||||
@@ -0,0 +1,321 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Performance baseline for the extraction engines.
|
||||
|
||||
python tests/bench_driver.py # 82 MiB fixture, fast
|
||||
python tests/bench_driver.py --big # 320 MiB fixture, accurate
|
||||
python tests/bench_driver.py --format zip
|
||||
python tests/bench_driver.py --format rar
|
||||
python tests/bench_driver.py --runs 5
|
||||
|
||||
Times our facades (src/zip_extract.c, src/rar_extract.c,
|
||||
src/sevenz_extract.c) against the external references that matter: the
|
||||
vendored SDK's own SzArEx path, and the official 7-Zip binary -- the latter
|
||||
being what upstream v1.8 gets by shelling out to a helper, so it doubles as
|
||||
the "how fast could we be" ceiling.
|
||||
|
||||
The three formats are packed from one shared payload, so the rows are
|
||||
comparable across formats and not just within one engine.
|
||||
|
||||
Why Python drives the measurement: in this sandbox a single `date` costs
|
||||
~350 ms, so the usual `t0=$(date)` / `t1=$(date)` pair adds ~700 ms of pure
|
||||
overhead to a measurement whose real value is ~600 ms, and `time`'s user/sys
|
||||
accounting does not see into the native child at all. Python spawns each
|
||||
child once and reads a monotonic clock around it, which leaves a small,
|
||||
constant "spawn tax" that is measured and subtracted (see the report).
|
||||
|
||||
The candidate binaries print their own in-process timing, which excludes the
|
||||
spawn tax entirely -- that is the most trustworthy figure for our side.
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import glob
|
||||
import os
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
|
||||
REPO = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
BUILD = os.path.join(REPO, ".build", "bench")
|
||||
SZ_BUILD = os.path.join(REPO, ".build", "sevenz-test")
|
||||
HT_BUILD = os.path.join(REPO, ".build", "host-test")
|
||||
SEVENZ_BIN = os.path.join(REPO, ".build", "7zdl", "extra", "x64", "7za.exe")
|
||||
BENCH_BIN = os.path.join(BUILD, "bench_extract.exe")
|
||||
SDK_BIN = os.path.join(SZ_BUILD, "sevenz_e2e.exe")
|
||||
# RAR is write-only in WinRAR (7-Zip can read the format but not create it),
|
||||
# so a RAR fixture needs rar.exe. Missing means the rar fixture is skipped,
|
||||
# never that the run fails.
|
||||
RAR_BIN = next((p for p in (
|
||||
os.environ.get("WFM_RAR"),
|
||||
r"C:\Program Files\WinRAR\rar.exe",
|
||||
r"C:\Program Files (x86)\WinRAR\rar.exe",
|
||||
"/usr/bin/rar", "/usr/local/bin/rar") if p and os.path.exists(p)), None)
|
||||
|
||||
# Object lists mirror what tests/run-sevenz-tests.sh and tests/run-tests.sh
|
||||
# build; those scripts must have run once before this can link.
|
||||
#
|
||||
# These lists are the one thing here that can rot: adding a source file to
|
||||
# either engine (sevenz_header.c for -mhe=on, mz_crypt_wfm.c plus the two
|
||||
# restored minizip-ng crypto streams for encrypted ZIP) silently leaves the
|
||||
# link with undefined symbols. If `g++` below fails on an unresolved symbol
|
||||
# that clearly lives in src/ or third_party/, check here first.
|
||||
VENDOR_7Z = ["7zAlloc", "7zArcIn", "7zBuf", "7zBuf2", "7zCrc", "7zCrcOpt",
|
||||
"7zDec", "7zFile", "7zStream", "Aes", "AesOpt", "Alloc", "Bcj2",
|
||||
"Bra", "Bra86", "BraIA64", "CpuArch", "Delta", "DllSecur",
|
||||
"Lzma2Dec", "LzmaDec", "Lzma2DecMt", "MtDec", "Threads", "Ppmd7",
|
||||
"Ppmd7Dec", "Sha256", "Sha256Opt", "SwapBytes"]
|
||||
ZLIB = ["adler32", "crc32", "deflate", "inffast", "inflate", "inftrees",
|
||||
"trees", "zutil"]
|
||||
MINIZIP = ["mz_crypt", "mz_crypt_wfm", "mz_os", "mz_os_posix", "mz_strm",
|
||||
"mz_strm_mem", "mz_strm_os_posix", "mz_strm_pkcrypt",
|
||||
"mz_strm_wzaes", "mz_strm_zlib", "mz_zip"]
|
||||
EXTRA_LIBS = ["-lole32", "-loleaut32", "-luuid", "-ladvapi32", "-luser32",
|
||||
"-lshell32"]
|
||||
|
||||
|
||||
def objs(base, names):
|
||||
return [os.path.join(base, n + ".o") for n in names]
|
||||
|
||||
|
||||
def build_bench():
|
||||
"""Link tests/bench_extract.c against the prebuilt engine objects."""
|
||||
if os.path.exists(BENCH_BIN):
|
||||
return True
|
||||
|
||||
# The RAR engine is C++ (vendored UnRAR), so bench_extract.c is compiled
|
||||
# with gcc and the link goes through g++.
|
||||
unrar = sorted(glob.glob(os.path.join(HT_BUILD, "unrar7_*.o")))
|
||||
needed = (objs(SZ_BUILD, ["sevenz_extract", "sevenz_chain",
|
||||
"sevenz_header", "sevenz_mt", "sevenz_volstream",
|
||||
"zipx_common", "zipx_volume"])
|
||||
+ objs(HT_BUILD, ["zip_extract", "zipx_volstream", "rar_extract"])
|
||||
+ objs(HT_BUILD, ZLIB) + objs(HT_BUILD, MINIZIP)
|
||||
+ objs(SZ_BUILD, VENDOR_7Z) + unrar)
|
||||
missing = [p for p in needed if not os.path.exists(p)]
|
||||
if missing:
|
||||
print("missing engine objects, run these first:")
|
||||
print(" /usr/bin/bash tests/run-sevenz-tests.sh")
|
||||
print(" /usr/bin/bash tests/run-tests.sh --rebuild")
|
||||
print("first missing: %s" % missing[0])
|
||||
return False
|
||||
|
||||
includes = ["-I" + os.path.join(REPO, "third_party", "7z"),
|
||||
"-I" + os.path.join(REPO, "third_party", "minizip-ng", "include"),
|
||||
"-I" + os.path.join(REPO, "third_party", "zlib", "include"),
|
||||
"-I" + os.path.join(REPO, "src"),
|
||||
"-I" + os.path.join(REPO, "tests", "compat"),
|
||||
"-include", os.path.join(REPO, "tests", "posix_compat.h")]
|
||||
obj = os.path.join(BUILD, "bench_extract.o")
|
||||
print("== compiling bench harness ==")
|
||||
if subprocess.run(["gcc", "-O2", "-w"] + includes
|
||||
+ ["-c", "-o", obj,
|
||||
os.path.join(REPO, "tests", "bench_extract.c")],
|
||||
cwd=REPO).returncode != 0:
|
||||
return False
|
||||
|
||||
libs = list(EXTRA_LIBS)
|
||||
if os.name == "nt":
|
||||
# Windows unrar system.cpp references SetSuspendState (PowrProf).
|
||||
libs.append("-lpowrprof")
|
||||
if subprocess.run(["g++", "-O2", "-o", BENCH_BIN, obj] + needed + libs,
|
||||
cwd=REPO).returncode != 0:
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def make_fixture(big, fmt):
|
||||
"""Build the archive if absent: repeated source text plus a real PE file.
|
||||
|
||||
The blend matters -- compression throughput depends heavily on match
|
||||
length, so a pure-text corpus would flatter every decoder equally and a
|
||||
pure-random one would measure nothing but copying.
|
||||
"""
|
||||
stem = "big4" if big else "big"
|
||||
archive = os.path.join(BUILD, "%s.%s" % (stem, fmt))
|
||||
if os.path.exists(archive):
|
||||
return archive
|
||||
|
||||
os.makedirs(BUILD, exist_ok=True)
|
||||
tp = os.path.join(REPO, "third_party")
|
||||
parts = []
|
||||
for sub in ("zlib/src", "7z", "minizip-ng/src"):
|
||||
d = os.path.join(tp, sub)
|
||||
if os.path.isdir(d):
|
||||
for name in sorted(os.listdir(d)):
|
||||
if name.endswith((".c", ".h")):
|
||||
parts.append(os.path.join(d, name))
|
||||
if not parts:
|
||||
print("no source available to build a fixture from")
|
||||
return None
|
||||
|
||||
repeat = 240 if big else 60
|
||||
src = os.path.join(BUILD, "payload_src.bin")
|
||||
with open(src, "wb") as out:
|
||||
for _ in range(repeat):
|
||||
for path in parts:
|
||||
with open(path, "rb") as fh:
|
||||
out.write(fh.read())
|
||||
|
||||
payload = src
|
||||
pe = r"C:\Windows\System32\ntoskrnl.exe"
|
||||
if os.path.exists(pe):
|
||||
binary = os.path.join(BUILD, "payload_bin.bin")
|
||||
with open(binary, "wb") as out:
|
||||
for _ in range(30 if big else 3):
|
||||
with open(pe, "rb") as fh:
|
||||
out.write(fh.read())
|
||||
payload = os.path.join(BUILD, "payload_mix.bin")
|
||||
with open(payload, "wb") as out:
|
||||
for _ in range(4 if big else 1):
|
||||
for path in (src, binary):
|
||||
with open(path, "rb") as fh:
|
||||
out.write(fh.read())
|
||||
|
||||
print("== creating fixture (first run only) ==")
|
||||
if fmt == "rar":
|
||||
# RAR needs WinRAR's rar.exe; 7-Zip cannot write the format.
|
||||
rar = RAR_BIN
|
||||
if not rar:
|
||||
print("WinRAR (rar.exe) not found; cannot create a RAR fixture")
|
||||
return None
|
||||
add = ["a", "-m3", "-idq", "-ep1"]
|
||||
subprocess.run([rar] + add + [archive, payload], cwd=REPO,
|
||||
stdout=subprocess.DEVNULL)
|
||||
else:
|
||||
if fmt == "7z":
|
||||
add = ["a", "-t7z", "-m0=lzma2", "-mx=5", "-ms=on"]
|
||||
else:
|
||||
add = ["a", "-tzip", "-mx=5", "-mm=Deflate"]
|
||||
subprocess.run([SEVENZ_BIN] + add + [archive, payload], cwd=REPO,
|
||||
stdout=subprocess.DEVNULL)
|
||||
return archive
|
||||
|
||||
|
||||
def sample(argv, runs, workdir):
|
||||
"""Run argv `runs` times; return (best wall ms, best in-process ms|None).
|
||||
|
||||
Every run writes to its own path (`workdir0`, `workdir1`, ...) that does
|
||||
not exist yet. Reusing one output directory is not an option: publishing
|
||||
into a tree left by the previous run charges the rename step for the
|
||||
collision, and that alone moved the same archive from 0.77 s to 1.28 s.
|
||||
The placeholder `{out}` in argv marks where the run directory goes.
|
||||
|
||||
Nothing is deleted between runs either -- clearing these trees is a bulk
|
||||
delete this host blocks -- so `.build/bench/W*` does accumulate and is
|
||||
worth clearing by hand now and then.
|
||||
"""
|
||||
best = None
|
||||
internal = None
|
||||
for i in range(runs):
|
||||
# Forward slashes on purpose: the engines derive the destination's
|
||||
# parent with a '/' scan (they only ever see POSIX paths on the PS5),
|
||||
# so a Windows-style relative or absolute path is rejected outright.
|
||||
run_dir = ("%s%d" % (workdir, i)).replace("\\", "/")
|
||||
cmd = [arg.replace("{out}", run_dir) for arg in argv]
|
||||
started = time.perf_counter()
|
||||
proc = subprocess.run(cmd, stdout=subprocess.PIPE,
|
||||
stderr=subprocess.STDOUT, cwd=REPO, timeout=1800)
|
||||
elapsed = (time.perf_counter() - started) * 1000.0
|
||||
if proc.returncode != 0:
|
||||
sys.stdout.write(proc.stdout.decode("utf-8", "replace")[:300])
|
||||
return None, None
|
||||
if best is None or elapsed < best:
|
||||
best = elapsed
|
||||
match = re.search(rb"wall\s*:\s*([0-9.]+)\s*s", proc.stdout)
|
||||
if match:
|
||||
value = float(match.group(1)) * 1000.0
|
||||
if internal is None or value < internal:
|
||||
internal = value
|
||||
return best, internal
|
||||
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("--big", action="store_true",
|
||||
help="4x fixture (~320 MiB) for accurate ratios")
|
||||
ap.add_argument("--runs", type=int, default=3)
|
||||
ap.add_argument("--format", choices=["7z", "zip", "rar"], default="7z",
|
||||
help="which engine to benchmark (default 7z)")
|
||||
args = ap.parse_args()
|
||||
|
||||
# For ZIP there is no useful SDK reference: the vendored SDK is 7z-only,
|
||||
# so the comparison is just ours versus the official binary.
|
||||
for path in (SEVENZ_BIN, SDK_BIN if args.format == "7z" else SEVENZ_BIN):
|
||||
if not os.path.exists(path):
|
||||
print("missing: %s" % path)
|
||||
return 1
|
||||
if not build_bench():
|
||||
return 1
|
||||
|
||||
archive = make_fixture(args.big, args.format)
|
||||
if not archive:
|
||||
return 1
|
||||
raw = os.path.getsize(archive)
|
||||
|
||||
print()
|
||||
print("archive : %s (%.0f MiB packed), best of %d runs"
|
||||
% (os.path.basename(archive), raw / 1048576.0, args.runs))
|
||||
print()
|
||||
|
||||
rows = []
|
||||
wall, inner = sample([BENCH_BIN, archive, "{out}"], args.runs,
|
||||
os.path.join(BUILD, "W7"))
|
||||
if wall:
|
||||
rows.append(("ours / " + args.format, wall, inner))
|
||||
|
||||
if args.format == "7z":
|
||||
wall, _ = sample([SDK_BIN, archive, "{out}"], args.runs,
|
||||
os.path.join(BUILD, "W8"))
|
||||
if wall:
|
||||
rows.append(("sdk SzArEx", wall, None))
|
||||
variants = (("7za 1 thread", ["-mmt=off"]),
|
||||
("7za 8 threads", ["-mmt=8"]),
|
||||
("7za all cores", []))
|
||||
elif args.format == "rar":
|
||||
# 7-Zip reads RAR, so it is a valid cross-check on the same archive.
|
||||
variants = (("7za 1 thread", ["-mmt=off"]),
|
||||
("7za all cores", []))
|
||||
else:
|
||||
variants = (("7za 1 thread", ["-mmt=off"]),)
|
||||
|
||||
for label, extra in variants:
|
||||
wall, _ = sample([SEVENZ_BIN, "x", "-y", "-aoa"] + extra
|
||||
+ ["-o{out}", archive], args.runs,
|
||||
os.path.join(BUILD, "W9"))
|
||||
if wall:
|
||||
rows.append((label, wall, None))
|
||||
|
||||
if args.format == "rar" and RAR_BIN:
|
||||
# rar.exe treats the trailing separator as "this is the target
|
||||
# directory"; without it the path is parsed as a file mask and the
|
||||
# command reports that there is nothing to extract.
|
||||
wall, _ = sample([RAR_BIN, "x", "-y", "-o+", archive, "{out}/"],
|
||||
args.runs, os.path.join(BUILD, "W10"))
|
||||
if wall:
|
||||
rows.append(("winrar x", wall, None))
|
||||
|
||||
print(" %-18s %10s %12s" % ("configuration", "external", "internal"))
|
||||
for label, wall, inner in rows:
|
||||
print(" %-18s %9.0f ms %12s"
|
||||
% (label, wall,
|
||||
"%9.0f ms" % inner if inner else " -"))
|
||||
|
||||
tax = None
|
||||
ours = [r for r in rows if r[0].startswith("ours")]
|
||||
if ours and ours[0][2]:
|
||||
tax = ours[0][1] - ours[0][2]
|
||||
ref = ours[0][2]
|
||||
print()
|
||||
print(" spawn tax (ours external - internal): %.0f ms" % tax)
|
||||
print()
|
||||
print(" %-18s %12s %9s" % ("configuration", "net", "vs ours"))
|
||||
for label, wall, inner in rows:
|
||||
net = inner if inner else max(wall - tax, 1.0)
|
||||
print(" %-18s %9.0f ms %8.2fx" % (label, net, ref / net))
|
||||
print()
|
||||
print(" net = external minus spawn tax; >1x means faster than ours")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -0,0 +1,134 @@
|
||||
/* Wall-clock benchmark for the extraction paths, ZIP, RAR and 7z.
|
||||
*
|
||||
* bench_extract <archive> <out-dir> [password]
|
||||
*
|
||||
* Reports how long the facade takes end to end -- decode, staging writes,
|
||||
* publish -- which is exactly what a PS5 user waits for. Note there is no
|
||||
* fsync in the extract path at all (see the comment in zip_extract.c's
|
||||
* publish_entry): the pipeline is "sync nothing, rename everything". It is
|
||||
* deliberately separate from the correctness drivers: those assert on bytes,
|
||||
* this one only prints numbers, and it is not part of the test matrix.
|
||||
*
|
||||
* Keeping it in-tree matters because "is our engine fast?" is a question that
|
||||
* will come up again, and the answer should be a command anyone can rerun
|
||||
* rather than a number somebody remembers. The format is picked from the
|
||||
* suffix so the same binary covers all three engines.
|
||||
*/
|
||||
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <time.h>
|
||||
|
||||
#include "sevenz_extract.h"
|
||||
|
||||
#ifndef BENCH_NO_RAR
|
||||
#include "rar_extract.h"
|
||||
#endif
|
||||
|
||||
static int
|
||||
has_suffix(const char *path, const char *suffix) {
|
||||
size_t path_len;
|
||||
size_t suffix_len;
|
||||
|
||||
if(!path || !suffix) return 0;
|
||||
path_len = strlen(path);
|
||||
suffix_len = strlen(suffix);
|
||||
if(path_len < suffix_len) return 0;
|
||||
for(size_t i = 0; i < suffix_len; i++) {
|
||||
char a = path[path_len - suffix_len + i];
|
||||
char b = suffix[i];
|
||||
if(a >= 'A' && a <= 'Z') a = (char)(a - 'A' + 'a');
|
||||
if(b >= 'A' && b <= 'Z') b = (char)(b - 'A' + 'a');
|
||||
if(a != b) return 0;
|
||||
}
|
||||
return 1;
|
||||
}
|
||||
|
||||
static double
|
||||
now_seconds(void) {
|
||||
struct timespec ts;
|
||||
|
||||
if(timespec_get(&ts, TIME_UTC) != TIME_UTC) return 0.0;
|
||||
return (double)ts.tv_sec + (double)ts.tv_nsec / 1000000000.0;
|
||||
}
|
||||
|
||||
int
|
||||
main(int argc, char **argv) {
|
||||
zipx_result_t result;
|
||||
zipx_status_t status;
|
||||
const char *archive;
|
||||
const char *out_dir;
|
||||
const char *password;
|
||||
const char *format;
|
||||
double started;
|
||||
double elapsed;
|
||||
double mebibytes;
|
||||
|
||||
if(argc < 3) {
|
||||
fprintf(stderr, "usage: %s <archive> <out-dir> [password]\n", argv[0]);
|
||||
return 2;
|
||||
}
|
||||
archive = argv[1];
|
||||
out_dir = argv[2];
|
||||
password = argc > 3 ? argv[3] : NULL;
|
||||
|
||||
if(has_suffix(archive, ".7z") || has_suffix(archive, ".7z.001") ||
|
||||
has_suffix(archive, ".001")) {
|
||||
format = "7z";
|
||||
}
|
||||
#ifndef BENCH_SEVENZ_ONLY
|
||||
else if(has_suffix(archive, ".zip") || has_suffix(archive, ".zip.001") ||
|
||||
has_suffix(archive, ".z01")) {
|
||||
format = "zip";
|
||||
}
|
||||
#endif
|
||||
#ifndef BENCH_NO_RAR
|
||||
else if(has_suffix(archive, ".rar") || has_suffix(archive, ".part1.rar") ||
|
||||
has_suffix(archive, ".r00")) {
|
||||
format = "rar";
|
||||
}
|
||||
#endif
|
||||
else {
|
||||
fprintf(stderr, "unsupported benchmark format: %s\n", archive);
|
||||
return 2;
|
||||
}
|
||||
|
||||
memset(&result, 0, sizeof(result));
|
||||
started = now_seconds();
|
||||
#ifndef BENCH_NO_RAR
|
||||
if(!strcmp(format, "rar")) {
|
||||
status = rar_extract(archive, out_dir, ZIPX_CONFLICT_OVERWRITE,
|
||||
zipx_default_limits(), NULL, NULL, NULL, NULL,
|
||||
&result);
|
||||
} else
|
||||
#endif
|
||||
#ifndef BENCH_SEVENZ_ONLY
|
||||
if(!strcmp(format, "zip")) {
|
||||
status = zipx_extract(archive, out_dir, ZIPX_CONFLICT_OVERWRITE,
|
||||
zipx_default_limits(), NULL, NULL, NULL, NULL, &result);
|
||||
} else
|
||||
#endif
|
||||
{
|
||||
status = sevenz_extract(archive, out_dir, ZIPX_CONFLICT_OVERWRITE,
|
||||
zipx_default_limits(), NULL, NULL, NULL,
|
||||
password, &result);
|
||||
}
|
||||
elapsed = now_seconds() - started;
|
||||
|
||||
mebibytes = (double)result.bytes_total / (1024.0 * 1024.0);
|
||||
|
||||
printf("format : %s\n", format);
|
||||
printf("status : %s\n", zipx_status_string(status));
|
||||
printf("entries : %llu\n", (unsigned long long)result.entries_total);
|
||||
printf("unpacked : %.1f MiB\n", mebibytes);
|
||||
printf("wall : %.3f s\n", elapsed);
|
||||
if(elapsed > 0.0) {
|
||||
printf("through : %.1f MiB/s\n", mebibytes / elapsed);
|
||||
}
|
||||
if(status != ZIPX_OK) {
|
||||
printf("detail : %s\n", result.detail[0] ? result.detail : "(none)");
|
||||
printf("message : %s\n", result.message[0] ? result.message : "(none)");
|
||||
}
|
||||
return status == ZIPX_OK ? 0 : 1;
|
||||
}
|
||||
@@ -0,0 +1,170 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Cross-format extraction benchmark: same payload, three containers.
|
||||
|
||||
python tests/bench_formats.py # one 329 MiB file, 3 runs
|
||||
python tests/bench_formats.py --source .build/bench/payload_bin.bin \\
|
||||
--stem bin --runs 5 # 39 MiB of machine code
|
||||
python tests/bench_formats.py --source .build/bench/manyfiles_src --stem mf
|
||||
|
||||
bench_driver.py answers "how do we compare with 7-Zip for one format"; this
|
||||
answers "which container should a user expect to unpack fastest", which is a
|
||||
different question and needs the three archives to hold the same bytes.
|
||||
|
||||
Method notes that took a while to get right, so they are pinned here:
|
||||
|
||||
* Fresh output directory per run. Publishing on top of the previous run's
|
||||
tree took the same 7z archive from 0.77 s to 1.28 s.
|
||||
* Forward slashes in every path. The engines derive the destination parent
|
||||
with a '/' scan, so "C:\\...\\W0" is rejected as an invalid destination.
|
||||
* Best of N, because a single run varies widely on this host: extraction
|
||||
creates hundreds of MiB that the on-access scanner inspects and the page
|
||||
cache has to write back, and neither is under our control.
|
||||
* Corpus matters as much as container. Deflate decodes faster than LZMA2 on
|
||||
ordinary data, but on highly repetitive text LZMA2 finds long matches
|
||||
where deflate only finds 32 KiB ones, and the order flips. Run the same
|
||||
archive set on more than one source before concluding anything.
|
||||
|
||||
Packed sizes are printed next to the times on purpose: a container that packs
|
||||
5x smaller also reads 5x less from the card, which is why the ranking on the
|
||||
PS5 -- where storage is the slow part -- can differ from the ranking here.
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import os
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
|
||||
REPO = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
BUILD = os.path.join(REPO, ".build", "bench")
|
||||
BENCH_BIN = os.path.join(BUILD, "bench_extract.exe")
|
||||
SEVENZ_BIN = os.path.join(REPO, ".build", "7zdl", "extra", "x64", "7za.exe")
|
||||
RAR_BIN = next((p for p in (
|
||||
os.environ.get("WFM_RAR"),
|
||||
r"C:\Program Files\WinRAR\rar.exe",
|
||||
r"C:\Program Files (x86)\WinRAR\rar.exe",
|
||||
"/usr/bin/rar", "/usr/local/bin/rar") if p and os.path.exists(p)), None)
|
||||
DEFAULT_SOURCE = os.path.join(BUILD, "payload4.bin")
|
||||
|
||||
# Nominal level 5 in both packers, which is the GUI default of each:
|
||||
# 7-Zip -mx=5, WinRAR -m3 ("Normal"). They are not equivalent amounts of
|
||||
# work -- LZMA2 at level 5 is a far stronger compressor than deflate at 5 --
|
||||
# but they are what a user who never opens the advanced panel ends up with.
|
||||
FORMATS = ("zip", "7z", "rar")
|
||||
|
||||
|
||||
def pack(stem, payload):
|
||||
"""Create stem.{zip,7z,rar} from payload if they are not there yet."""
|
||||
made = []
|
||||
for fmt in FORMATS:
|
||||
archive = os.path.join(BUILD, "%s.%s" % (stem, fmt))
|
||||
if os.path.exists(archive):
|
||||
continue
|
||||
if fmt == "rar":
|
||||
if not RAR_BIN:
|
||||
print("skip %s: WinRAR (rar.exe) not found" % fmt)
|
||||
continue
|
||||
cmd = [RAR_BIN, "a", "-m3", "-idq"]
|
||||
if os.path.isfile(payload):
|
||||
cmd.append("-ep1")
|
||||
cmd += [archive, payload]
|
||||
else:
|
||||
cmd = [SEVENZ_BIN, "a", "-t" + fmt] + (
|
||||
["-m0=lzma2", "-mx=5", "-ms=on"] if fmt == "7z"
|
||||
else ["-mx=5", "-mm=Deflate"]) + [archive, payload]
|
||||
print("== packing %s ==" % os.path.basename(archive))
|
||||
if subprocess.run(cmd, cwd=REPO, stdout=subprocess.DEVNULL).returncode:
|
||||
print("packing failed")
|
||||
return None
|
||||
made.append(archive)
|
||||
return [os.path.join(BUILD, "%s.%s" % (stem, f)) for f in FORMATS]
|
||||
|
||||
|
||||
def time_archive(archive, runs):
|
||||
"""Best in-process wall time over `runs` fresh-directory extractions."""
|
||||
tag = os.path.splitext(os.path.basename(archive))[0]
|
||||
times = []
|
||||
unpacked = entries = 0
|
||||
for i in range(runs):
|
||||
# A path that does not exist yet: the engines publish with a rename,
|
||||
# and a rename into a tree that already has the file costs extra.
|
||||
out = os.path.join(BUILD, "F-%s-%d" % (tag, i)).replace("\\", "/")
|
||||
started = time.perf_counter()
|
||||
proc = subprocess.run([BENCH_BIN, archive, out], cwd=REPO,
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=subprocess.STDOUT, timeout=1800)
|
||||
external = time.perf_counter() - started
|
||||
if proc.returncode != 0:
|
||||
sys.stdout.write(proc.stdout.decode("utf-8", "replace")[:400])
|
||||
return None
|
||||
text = proc.stdout.decode("utf-8", "replace")
|
||||
match = re.search(r"wall\s*:\s*([0-9.]+)", text)
|
||||
if not match:
|
||||
return None
|
||||
times.append(float(match.group(1)))
|
||||
unpacked = float(re.search(r"unpacked\s*:\s*([0-9.]+)", text).group(1))
|
||||
entries = int(re.search(r"entries\s*:\s*(\d+)", text).group(1))
|
||||
return {"times": sorted(times), "unpacked": unpacked, "entries": entries}
|
||||
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("--runs", type=int, default=3)
|
||||
ap.add_argument("--source", default=DEFAULT_SOURCE,
|
||||
help="file or directory to pack (default: %s)" % DEFAULT_SOURCE)
|
||||
ap.add_argument("--stem", help="archive base name (default: source basename)")
|
||||
args = ap.parse_args()
|
||||
|
||||
if not os.path.exists(BENCH_BIN):
|
||||
print("missing %s -- run: python tests/bench_driver.py" % BENCH_BIN)
|
||||
return 1
|
||||
|
||||
payload = args.source
|
||||
if not os.path.isabs(payload):
|
||||
payload = os.path.join(REPO, payload)
|
||||
if not os.path.exists(payload):
|
||||
print("missing payload: %s" % payload)
|
||||
return 1
|
||||
stem = args.stem or os.path.splitext(os.path.basename(payload))[0]
|
||||
|
||||
archives = pack(stem, payload)
|
||||
if not archives:
|
||||
return 1
|
||||
|
||||
rows = []
|
||||
for archive in archives:
|
||||
if not os.path.exists(archive):
|
||||
continue
|
||||
result = time_archive(archive, args.runs)
|
||||
if result is None:
|
||||
print("failed: %s" % archive)
|
||||
continue
|
||||
rows.append((os.path.splitext(archive)[1][1:], archive, result))
|
||||
|
||||
if not rows:
|
||||
return 1
|
||||
|
||||
print()
|
||||
print("best of %d runs, fresh output directory each time" % args.runs)
|
||||
print("payload: %d entries, %.1f MiB unpacked"
|
||||
% (rows[0][2]["entries"], rows[0][2]["unpacked"]))
|
||||
print()
|
||||
print(" %-5s %10s %9s %12s %12s" %
|
||||
("fmt", "packed", "best", "median", "throughput"))
|
||||
best = min(r[2]["times"][0] for r in rows)
|
||||
for fmt, archive, result in sorted(rows, key=lambda r: r[2]["times"][0]):
|
||||
packed = os.path.getsize(archive) / 1048576.0
|
||||
times = result["times"]
|
||||
median = times[len(times) // 2]
|
||||
rate = result["unpacked"] / times[0]
|
||||
print(" %-5s %8.1f M %7.0f ms %9.0f ms %8.0f MiB/s"
|
||||
% (fmt, packed, times[0] * 1000.0, median * 1000.0, rate))
|
||||
print()
|
||||
for fmt, archive, result in sorted(rows, key=lambda r: r[2]["times"][0]):
|
||||
print(" %-5s vs fastest: %.2fx" % (fmt, result["times"][0] / best))
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -0,0 +1,161 @@
|
||||
/* Prints every extraction progress report, with a monotonic timestamp.
|
||||
*
|
||||
* bench_progress <archive> <out-dir> [--mode print|empty|none]
|
||||
*
|
||||
* bench_extract answers "how long"; this answers "did the UI move while it
|
||||
* took that long". A report that sits at entries=0/N for minutes reads to the
|
||||
* user as a hang even though bytes are still flowing, so being able to see the
|
||||
* phase/entries/bytes sequence is what turns "the progress bar freezes" into a
|
||||
* named phase. Format is picked from the suffix, same as bench_extract.
|
||||
*
|
||||
* --mode exists to answer a second question: "does reporting itself cost
|
||||
* time?". The engine's report() helper runs on the hot path -- unrar calls it
|
||||
* once per decompressed chunk -- and it reads the clock before it decides
|
||||
* whether to throttle, so the cost is paid even when no report goes out.
|
||||
* print a callback that formats and prints every report (diagnostic)
|
||||
* empty a callback that returns immediately (engine cost, no consumer)
|
||||
* none no callback at all (the engine skips report() entirely)
|
||||
* Comparing wall time across the three separates "our bookkeeping" from
|
||||
* "the file I/O we cannot avoid".
|
||||
*/
|
||||
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <time.h>
|
||||
|
||||
#include "zip_extract.h"
|
||||
#include "rar_extract.h"
|
||||
#include "sevenz_extract.h"
|
||||
|
||||
static double g_t0;
|
||||
|
||||
static int
|
||||
has_suffix(const char *path, const char *suffix) {
|
||||
size_t path_len, suffix_len, i;
|
||||
|
||||
if(!path || !suffix) return 0;
|
||||
path_len = strlen(path);
|
||||
suffix_len = strlen(suffix);
|
||||
if(path_len < suffix_len) return 0;
|
||||
for(i = 0; i < suffix_len; i++) {
|
||||
char a = path[path_len - suffix_len + i];
|
||||
char b = suffix[i];
|
||||
|
||||
if(a >= 'A' && a <= 'Z') a = (char)(a - 'A' + 'a');
|
||||
if(b >= 'A' && b <= 'Z') b = (char)(b - 'A' + 'a');
|
||||
if(a != b) return 0;
|
||||
}
|
||||
return 1;
|
||||
}
|
||||
|
||||
static double
|
||||
now_seconds(void) {
|
||||
struct timespec ts;
|
||||
|
||||
if(timespec_get(&ts, TIME_UTC) != TIME_UTC) return 0.0;
|
||||
return (double)ts.tv_sec + (double)ts.tv_nsec / 1000000000.0;
|
||||
}
|
||||
|
||||
static const char *
|
||||
phase_name(int phase) {
|
||||
switch(phase) {
|
||||
case ZIPX_PHASE_SCAN: return "scan";
|
||||
case ZIPX_PHASE_EXTRACT: return "extract";
|
||||
case ZIPX_PHASE_PUBLISH: return "publish";
|
||||
case ZIPX_PHASE_CLEANUP: return "cleanup";
|
||||
default: return "?";
|
||||
}
|
||||
}
|
||||
|
||||
/* Costs exactly what the engine's report() costs, without a consumer. */
|
||||
static void
|
||||
on_progress_empty(void *userdata, const zipx_progress_t *p) {
|
||||
(void)userdata;
|
||||
(void)p;
|
||||
}
|
||||
|
||||
static void
|
||||
on_progress(void *userdata, const zipx_progress_t *p) {
|
||||
(void)userdata;
|
||||
printf("%8.3f %-7s entries=%llu/%llu bytes=%llu/%llu %s\n",
|
||||
now_seconds() - g_t0, phase_name(p->phase),
|
||||
(unsigned long long)p->entries_done,
|
||||
(unsigned long long)p->entries_total,
|
||||
(unsigned long long)p->bytes_done,
|
||||
(unsigned long long)p->bytes_total,
|
||||
p->current ? p->current : "");
|
||||
fflush(stdout);
|
||||
}
|
||||
|
||||
int
|
||||
main(int argc, char **argv) {
|
||||
zipx_result_t result;
|
||||
zipx_status_t status;
|
||||
const char *archive;
|
||||
const char *out_dir;
|
||||
const char *format;
|
||||
const char *mode = "print";
|
||||
zipx_progress_fn on_report = on_progress;
|
||||
double elapsed;
|
||||
int i;
|
||||
|
||||
if(argc < 3) {
|
||||
fprintf(stderr, "usage: %s <archive> <out-dir> [--mode print|empty|none]\n",
|
||||
argv[0]);
|
||||
return 2;
|
||||
}
|
||||
archive = argv[1];
|
||||
out_dir = argv[2];
|
||||
|
||||
for(i = 3; i < argc; i++) {
|
||||
if(!strcmp(argv[i], "--mode") && i + 1 < argc) {
|
||||
mode = argv[++i];
|
||||
} else {
|
||||
fprintf(stderr, "unknown argument: %s\n", argv[i]);
|
||||
return 2;
|
||||
}
|
||||
}
|
||||
if(!strcmp(mode, "empty")) {
|
||||
on_report = on_progress_empty;
|
||||
} else if(!strcmp(mode, "none")) {
|
||||
on_report = NULL;
|
||||
} else if(strcmp(mode, "print")) {
|
||||
fprintf(stderr, "unknown --mode: %s\n", mode);
|
||||
return 2;
|
||||
}
|
||||
|
||||
if(has_suffix(archive, ".rar") || has_suffix(archive, ".part1.rar") ||
|
||||
has_suffix(archive, ".r00")) {
|
||||
format = "rar";
|
||||
} else if(has_suffix(archive, ".zip") || has_suffix(archive, ".z01") ||
|
||||
has_suffix(archive, ".zip.001")) {
|
||||
format = "zip";
|
||||
} else {
|
||||
format = "7z";
|
||||
}
|
||||
|
||||
memset(&result, 0, sizeof(result));
|
||||
g_t0 = now_seconds();
|
||||
|
||||
if(!strcmp(format, "rar")) {
|
||||
status = rar_extract(archive, out_dir, ZIPX_CONFLICT_OVERWRITE,
|
||||
zipx_default_limits(), NULL, on_report, NULL, NULL,
|
||||
&result);
|
||||
} else if(!strcmp(format, "zip")) {
|
||||
status = zipx_extract(archive, out_dir, ZIPX_CONFLICT_OVERWRITE,
|
||||
zipx_default_limits(), NULL, on_report, NULL, NULL,
|
||||
&result);
|
||||
} else {
|
||||
status = sevenz_extract(archive, out_dir, ZIPX_CONFLICT_OVERWRITE,
|
||||
zipx_default_limits(), NULL, on_report, NULL,
|
||||
NULL, &result);
|
||||
}
|
||||
|
||||
elapsed = now_seconds() - g_t0;
|
||||
printf("---- done: format=%s mode=%s status=%s entries=%llu bytes=%llu wall=%.3fs\n",
|
||||
format, mode, zipx_status_string(status),
|
||||
(unsigned long long)result.entries_total,
|
||||
(unsigned long long)result.bytes_total, elapsed);
|
||||
return status == ZIPX_OK ? 0 : 1;
|
||||
}
|
||||
@@ -0,0 +1,40 @@
|
||||
/* Standalone big-file e2e: extract one large zip64 archive on the host and
|
||||
* report the engine result. Verification (size + sha256/cmp) is done by the
|
||||
* caller with shell tools.
|
||||
*
|
||||
* cc -O2 -Isrc -Ithird_party/minizip-ng/include \
|
||||
* -include tests/posix_compat.h \
|
||||
* -o bigfile_e2e bigfile_e2e.c zip_extract.o <minizip+zlib objs>
|
||||
*
|
||||
* ./bigfile_e2e <archive.zip> <out-dir>
|
||||
*/
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include "zip_extract.h"
|
||||
|
||||
int
|
||||
main(int argc, char **argv) {
|
||||
zipx_result_t res;
|
||||
zipx_status_t st;
|
||||
|
||||
if(argc != 3) {
|
||||
fprintf(stderr, "usage: %s <archive.zip> <out-dir>\n", argv[0]);
|
||||
return 2;
|
||||
}
|
||||
st = zipx_extract(argv[1], argv[2], ZIPX_CONFLICT_FAIL,
|
||||
zipx_limits_profile(ZIPX_LIMITS_LARGE),
|
||||
NULL, NULL, NULL, NULL, &res);
|
||||
printf("status=%d (%s)\n", (int)st, zipx_status_string(st));
|
||||
printf("sys_errno=%d entries=%llu/%llu files=%llu dirs=%llu\n",
|
||||
res.sys_errno, (unsigned long long)res.entries_done,
|
||||
(unsigned long long)res.entries_total,
|
||||
(unsigned long long)res.files_created,
|
||||
(unsigned long long)res.dirs_created);
|
||||
printf("bytes_total=%llu detail=%s\n",
|
||||
(unsigned long long)res.bytes_total, res.detail);
|
||||
if(res.message[0]) {
|
||||
printf("message=%s\n", res.message);
|
||||
}
|
||||
return st == ZIPX_OK ? 0 : 1;
|
||||
}
|
||||
@@ -0,0 +1,44 @@
|
||||
/* Host test shim: provides <sys/statvfs.h> for MinGW hosts.
|
||||
Only used when building the test suite (see run-tests.sh). */
|
||||
|
||||
#ifndef WFM_TEST_SYS_STATVFS_H
|
||||
#define WFM_TEST_SYS_STATVFS_H
|
||||
|
||||
#include <windows.h>
|
||||
|
||||
struct statvfs {
|
||||
unsigned long f_bsize;
|
||||
unsigned long f_frsize;
|
||||
unsigned long f_blocks;
|
||||
unsigned long f_bfree;
|
||||
unsigned long f_bavail;
|
||||
};
|
||||
|
||||
static int
|
||||
statvfs(const char *path, struct statvfs *buf) {
|
||||
ULARGE_INTEGER total;
|
||||
ULARGE_INTEGER free_bytes;
|
||||
char full[MAX_PATH];
|
||||
char root[8];
|
||||
|
||||
/* Resolve to an absolute path first: the drive-letter extraction below
|
||||
only works for "X:\..." style paths, and callers may pass relative
|
||||
paths (e.g. the standalone bigfile_e2e driver). */
|
||||
if(!GetFullPathNameA(path, (DWORD)sizeof(full), full, NULL)) {
|
||||
return -1;
|
||||
}
|
||||
snprintf(root, sizeof(root), "%.3s", full);
|
||||
if(!GetDiskFreeSpaceExA(root, &free_bytes, &total, NULL)) {
|
||||
return -1;
|
||||
}
|
||||
memset(buf, 0, sizeof(*buf));
|
||||
/* `unsigned long` is 32-bit on Windows: store free space scaled by 4096
|
||||
so archives up to 16 TiB don't overflow (real 64-bit hosts are LP64
|
||||
and unaffected; PS5 SDK is LP64 too). */
|
||||
buf->f_bsize = 4096;
|
||||
buf->f_frsize = 4096;
|
||||
buf->f_bavail = free_bytes.QuadPart / 4096;
|
||||
return 0;
|
||||
}
|
||||
|
||||
#endif
|
||||