diff --git a/.github/workflows/tests.yml b/.github/workflows/tests.yml index 93bfc89..5a6e810 100644 --- a/.github/workflows/tests.yml +++ b/.github/workflows/tests.yml @@ -38,9 +38,13 @@ jobs: - run: cargo test --all-features --locked - - run: cargo test --locked bounded:: + - run: "cargo test --locked bounded::" - - run: cargo test --locked --no-default-features bounded:: + - run: "cargo test --locked --no-default-features bounded::" + + - run: cargo test --release --locked --features incremental-experiment --lib --bins + + - run: cargo test --release --locked --no-default-features --features incremental-experiment --lib # Full TESTING.md correctness pass: pixel-for-pixel and ICC-profile # comparison against a pinned heif-dec validator over the libheif corpus, @@ -72,7 +76,7 @@ jobs: - name: Install system dependencies run: | sudo apt-get update - sudo apt-get install -y cmake pkg-config ffmpeg \ + sudo apt-get install -y cmake pkg-config ffmpeg ruby \ libde265-dev libx265-dev libaom-dev libdav1d-dev \ libopenjp2-7-dev libjpeg-dev libbrotli-dev zlib1g-dev @@ -120,10 +124,45 @@ jobs: - name: Full correctness pass run: scripts/heic_tests.sh verify --full --require-exts heic,avif + - name: Incremental decoder oracle comparisons + run: | + prefix="$PWD/.heic-test-assets/libpng-install" + cmake -S scripts/incremental -B .heic-test-runs/incremental-oracle \ + -DHEIF_SOURCE_DIR="$PWD/.heic-test-assets/libheif" \ + -DHEIF_BUILD_DIR="$PWD/.heic-test-runs/validator-build" \ + -DPNG_PNG_INCLUDE_DIR="$prefix/include" \ + -DPNG_LIBRARY="$prefix/lib/libpng16.a" + cmake --build .heic-test-runs/incremental-oracle --parallel + ruby scripts/incremental/generate.rb \ + .heic-test-runs/incremental-oracle/primary-oracle \ + .heic-test-assets/libheif/fuzzing/data/corpus/colors-no-alpha.heic + cargo test --release --locked --features incremental-experiment --test incremental-memory + cargo test --release --locked --no-default-features --features incremental-experiment --test incremental-memory + cargo build --release --locked --features incremental-experiment --bins + ruby scripts/incremental/verify.rb \ + .heic-test-runs/incremental-oracle/primary-oracle \ + target/release/incremental-allocation target/release/incremental-compare \ + .heic-test-assets/incremental-corpus/*.heic \ + .heic-test-assets/libheif/fuzzing/data/corpus/colors-no-alpha.heic \ + .heic-test-assets/libheif/fuzzing/data/corpus/colors-no-alpha-thumbnail.heic \ + .heic-test-assets/libheif/fuzzing/data/corpus/hevc32.heif \ + .heic-test-assets/stress-corpus/nowpp_photo_small.heic \ + .heic-test-assets/stress-corpus/wpp_narrow.heic \ + .heic-test-assets/stress-corpus/wpp_narrow2.heic + for name in direct tall pipeline; do + prefix="$PWD/.heic-test-runs/incremental/$name" + ANNEX_B_OUTPUT="$prefix.hevc" target/release/incremental-check ".heic-test-assets/incremental-corpus/$name.heic" + ffmpeg -v error -y -i "$prefix.hevc" -frames:v 1 -pix_fmt yuv420p -f rawvideo "$prefix.yuv" + REFERENCE_YUV="$prefix.yuv" target/release/incremental-check ".heic-test-assets/incremental-corpus/$name.heic" + done + - name: Upload verify report if: failure() uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 with: name: verify-report - path: .heic-test-runs/verify/run/report.txt + path: | + .heic-test-runs/verify/run/report.txt + .heic-test-runs/incremental/*.json* + .heic-test-assets/incremental-corpus/ if-no-files-found: ignore diff --git a/API.md b/API.md index e4fbe13..0a5d983 100644 --- a/API.md +++ b/API.md @@ -356,3 +356,40 @@ fn decode(path: &Path) -> Result<(), heic_decoder::DecodeError> { Ok(()) } ``` + +## Experimental incremental bounded decoding + +Enable the `incremental-experiment` Cargo feature to use +`decode_incremental_experiment(BoundedInput, BoundedDecodeOptions)`. It returns +`(BoundedRgbImage, [u64; 3])`; the diagnostic counters are reconstructed rows, +SAO edge components, and SAO band components across all coded items. This +feature and entry point are experimental. Existing decode functions retain +their behavior and do not select this path automatically. + +The incremental path supports opaque 8-bit 4:2:0 HEVC still images with one +IDR slice per coded item, including non-grid images and uniform grids whose +individual tiles exceed the output cap. WPP, HEVC tiles, multiple slices, +higher bit depths, alpha, other chroma formats, and unsupported color profiles +return errors. The output is display-oriented sRGB RGB8, using area averaging +when reduced, with a maximum side from 1 through 6000. Known depth, semantic +matte, and gain-map auxiliaries may accompany the SDR base image; this does +not apply their effects or provide HDR output. + +Reconstruction, deblocking, and SAO finish in rolling bands before conversion +and reduction. Large coded images use a bounded two-thread pipeline. The +working state grows with coded width and CTU size, while output storage is +capped by `max_side`; it does not retain a full source raster or use temporary +files or repeated reconstruction. A caller's borrowed compressed input is +separate from decoder-owned memory. Admission is conservative and may reject +budgets smaller than its reserved workspace, including a worker stack. +Allocator overhead and process RSS are not the same as requested allocation +payloads. Use Path input when avoiding a caller-owned compressed input buffer +is also important. + +This first subset retains limits of 16384 per coded side, 256 million display +pixels, bounded metadata, and a 64 KiB slice-header prefix. It supports one +clean-aperture crop before orientation, with chroma interpolation for odd +origins on a single coded image or one-tile grid. Odd-origin crops across +multiple grid items and repeated crops or crops after orientation are +unsupported. Exif orientation applies when no effective container orientation +is present. `decoder-tracing` disables bounded decoding. diff --git a/Cargo.toml b/Cargo.toml index b46dac8..951d073 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -8,6 +8,7 @@ license-file = "LICENSE" repository = "https://github.com/ente-io/heic-decoder" [features] +incremental-experiment = [] default = ["std", "parallel-grid"] decoder-tracing = ["std"] image-integration = ["dep:image"] @@ -40,3 +41,34 @@ scuffle-h265 = "0.2.2" [dev-dependencies] qcms = "0.3.0" + +[[bin]] +name = "incremental-allocation" +path = "src/bin/incremental-allocation.rs" +required-features = ["incremental-experiment"] + +[[bin]] +name = "incremental-check" +path = "src/bin/incremental-check.rs" +required-features = ["incremental-experiment"] + +[[bin]] +name = "capability-inspect" +path = "src/bin/capability-inspect.rs" +required-features = ["incremental-experiment"] + +[[bin]] +name = "incremental-compare" +path = "src/bin/incremental-compare.rs" +required-features = ["incremental-experiment"] + +[[bin]] +name = "incremental-bench" +path = "src/bin/incremental-bench.rs" +required-features = ["incremental-experiment"] + +[[test]] +name = "incremental-memory" +path = "tests/incremental-memory.rs" +harness = false +required-features = ["incremental-experiment"] diff --git a/TESTING.md b/TESTING.md index cb41589..bc8c542 100644 --- a/TESTING.md +++ b/TESTING.md @@ -161,3 +161,121 @@ libheif corpus and the ente fixtures. Generated reports and PNG artifacts are under `.heic-test-runs/`. Use `--keep-artifacts` with `verify` when debugging a pixel mismatch. + +## Incremental bounded decoder + +The opt-in `incremental-experiment` feature is tested separately because +`--all-features` also enables decoder tracing, which disables bounded decode. +The standalone allocation test imposes both live-heap and individual-request +ceilings across decoding and conversion threads. It covers Path/Bytes parity, +source-height growth, odd crops on both axes, grid clipping, tile color-profile +inheritance, threaded grid tiles, and early budget rejection. The malformed +reference fixture uses `.bin` and is tested under a 256 KiB ceiling separately +from the positive `*.heic` oracle comparisons. + +The external primary-image oracle is only a test executable; it adds no native +dependency to the Rust decoder. After the existing harness builds libheif, +build and run it with CMake, libpng development files, Ruby, and FFmpeg with +libx265. No incremental image assets are committed. The generator starts +from the existing external libheif corpus and creates the geometry, metadata, +and height-growth cases missing from that corpus in the ignored +`.heic-test-assets/incremental-corpus` directory. All reference PNGs come from +live strict libheif decoding. Its manifest records source, oracle, FFmpeg, +and input identities; encoded bytes can differ across libx265 versions. + +```bash +cmake -S scripts/incremental -B .heic-test-runs/incremental-oracle \ + -DHEIF_SOURCE_DIR="$PWD/.heic-test-assets/libheif" \ + -DHEIF_BUILD_DIR="$PWD/.heic-test-runs/validator-build" +cmake --build .heic-test-runs/incremental-oracle --parallel +ruby scripts/incremental/generate.rb \ + .heic-test-runs/incremental-oracle/primary-oracle \ + .heic-test-assets/libheif/fuzzing/data/corpus/colors-no-alpha.heic +cargo test --release --locked --features incremental-experiment --lib --bins --test incremental-memory +cargo test --release --locked --no-default-features --features incremental-experiment --lib --test incremental-memory +cargo build --release --locked --features incremental-experiment --bins +ruby scripts/incremental/verify.rb \ + .heic-test-runs/incremental-oracle/primary-oracle \ + target/release/incremental-allocation target/release/incremental-compare \ + .heic-test-assets/incremental-corpus/*.heic +``` + +Supply `PNG_PNG_INCLUDE_DIR` and `PNG_LIBRARY` when using a custom libpng, +as the CI workflow does. The runner compares original-size (up to side 6000) +and side-65 outputs, writes identities and metrics under +`.heic-test-runs/incremental`, and fails on any decoder, oracle, geometry, or +pixel-comparison failure. `INCREMENTAL_TEST_ROOT` overrides the output folder. +The generator accepts an optional output directory; set +`INCREMENTAL_FIXTURES_DIR` to that directory for allocation tests. Missing +generated inputs fail the test with setup instructions rather than skipping it. +Every supplied input is required to decode; unsupported inputs do not count +as successful comparisons. CI also exercises the six currently supported +files in the libheif/stress corpus. The separate normal suite keeps its exact +RGB/ICC comparisons and existing expected-failure accounting. + +The oracle enables libheif strict decoding and rejects warnings. RGB display +comparisons normalize embedded ICC to sRGB and apply an independent floating +point area average. The `rgb8-rounding` profile allows at most one value per +RGB channel per pixel; alpha remains exact. The `exact` profile allows no +sample changes. Dimensions and raster lengths are always checked. Average +error, PSNR, signed bias and local error are diagnostics, never substitutes +for the per-sample gate. These profiles do not authorize different tone +mapping, resampling filters, or high-bit-depth rounding. Native reconstruction +must retain exact sample checks against an independent decoder. ICC +normalization currently shares moxcms with the implementation; separate qcms +unit tests cover that boundary. + +The pinned libheif has a bilinear chroma-border indexing defect. Odd-grid +regressions retain an interior aperture so they still exercise the final +chroma neighbor without depending on its erroneous outer-border conversion. +Every supplied input uses the unmodified live strict oracle; there are no +stored corrected PNGs or input-specific comparison exceptions. This leaves +the affected outer-border RGB conversion outside independent libheif coverage. +Do not raise the global tolerance or silently classify discrepancies as +oracle defects. + +A separate high-frequency odd-crop probe remains outside the passing corpus: +at side 6000, one green sample is 72 versus the strict oracle's 70, despite +exact native YUV agreement with FFmpeg. The sample lies inside a band, and +the display-conversion cause remains unresolved. The one-value RGB gate is +unchanged; this probe is not counted as a passing input. Preserve it for +the decoder compatibility follow-up. + +For a supplied large supported image, measure allocator requests separately +from uninstrumented decode timing: + +```bash +target/release/incremental-allocation bounded image.heic 6000 128 +target/release/incremental-allocation bytes image.heic 6000 128 +target/release/incremental-bench bounded image.heic +target/release/incremental-bench normal image.heic +``` + +Run timing trials serially, after warm-up, in alternating order. Normal mode +returns the full raster; bounded mode includes capped output and reduction. +No claim of equal work or universal speedup follows from those timings. +Use a process memory tool separately for RSS. The allocator probe reports +requested heap bytes and excludes the caller's borrowed input allocation. + +`incremental-check` is a deliberately unbounded verification tool comparing +native reconstructed samples against the full Rust decoder. For an 8-bit +4:2:0 fixture with no conformance-window crop, provide independently decoded +planar YUV to require exact independent agreement too: + +```bash +ANNEX_B_OUTPUT=reference.hevc target/release/incremental-check image.heic +ffmpeg -v error -y -i reference.hevc -frames:v 1 -pix_fmt yuv420p -f rawvideo reference.yuv +REFERENCE_YUV=reference.yuv target/release/incremental-check image.heic +``` + +This verification tool is not part of the bounded path or its memory evidence. +`capability-inspect` prints coded-item SPS/PPS features. `scripts/incremental/wrap.rb` +packages a supplied one-IDR Annex-B 8-bit 4:2:0 stream into a direct HEIC or +repeated-tile grid; its dimensions must match the stream. `CROP`, `ROTATION` +and `MIRROR` environment variables add fixture transforms without re-encoding. +`GRID=1` creates a single-tile grid, `CANVAS` clips its visible dimensions, +`TILE_NCLX` and `PRIMARY_NCLX` set color metadata, and `REFERENCE_COUNT` +constructs malformed reference-count regression inputs. +The 200 MP photographic fixture used during development was externally +sourced and re-encoded as Main Still Picture Level 8.5, with WPP disabled; +it is not committed and is not a native camera HEIC compatibility claim. diff --git a/scripts/incremental/CMakeLists.txt b/scripts/incremental/CMakeLists.txt new file mode 100644 index 0000000..9cb3a81 --- /dev/null +++ b/scripts/incremental/CMakeLists.txt @@ -0,0 +1,11 @@ +cmake_minimum_required(VERSION 3.16) +project(incremental_oracle LANGUAGES CXX) +find_package(PNG REQUIRED) +find_library(HEIF_LIBRARY NAMES heif PATHS "${HEIF_BUILD_DIR}/libheif" NO_DEFAULT_PATH REQUIRED) +add_executable(primary-oracle primary-oracle.cc) +target_compile_features(primary-oracle PRIVATE cxx_std_17) +target_include_directories(primary-oracle PRIVATE "${HEIF_SOURCE_DIR}/libheif/api" "${HEIF_BUILD_DIR}") +target_link_libraries(primary-oracle PRIVATE "${HEIF_LIBRARY}" PNG::PNG) +if(CMAKE_CXX_COMPILER_ID MATCHES "GNU|Clang") + target_compile_options(primary-oracle PRIVATE -Wall -Wextra -Werror) +endif() diff --git a/scripts/incremental/generate.rb b/scripts/incremental/generate.rb new file mode 100644 index 0000000..2551ae0 --- /dev/null +++ b/scripts/incremental/generate.rb @@ -0,0 +1,74 @@ +require 'digest' +require 'fileutils' +require 'json' +require 'open3' +require 'rbconfig' + +abort 'Usage: generate.rb ORACLE LIBHEIF_CORPUS_IMAGE [OUTPUT_DIRECTORY]' unless (2..3).cover?(ARGV.size) +oracle, source = ARGV.take(2).map { |path| File.expand_path(path) } +root = File.expand_path('../..', __dir__) +output = File.expand_path(ARGV[2] || File.join(root, '.heic-test-assets/incremental-corpus')) +FileUtils.mkdir_p(output) +FileUtils.rm_f(File.join(output, 'manifest.json')) + +def execute(*args) + stdout, stderr, status = Open3.capture3(*args) + raise "Command failed: #{args.inspect}\n#{stderr}\n#{stdout}" unless status.success? + stdout +end + +def oracle_png(oracle, input, output) + metadata = JSON.parse(execute(oracle, input, output, 'strict')) + raise 'Oracle recovered from errors' unless metadata.fetch('strict') && metadata.fetch('warnings').zero? +end + +seed = File.join(output, 'source.png') +oracle_png(oracle, source, seed) +encodes = [['direct', 256, 256], ['tall', 256, 2048], ['pipeline', 1024, 1088], ['edge', 256, 256], ['vertical', 256, 256]] +encodes.each do |name, width, height| + filters = "scale=#{width}:#{height},format=yuv420p" + filters += ",geq=lum='lum(X,Y)':cb='16+mod(X*13+Y*7,224)':cr='16+mod(X*5+Y*17,224)'" if %w[edge vertical].include?(name) + filters += ',transpose=clock' if name == 'vertical' + parameters = 'keyint=1:wpp=0:repeat-headers=1:pools=none' + parameters += ':lossless=1' if name == 'vertical' + execute('ffmpeg', '-v', 'error', '-y', '-i', seed, '-vf', filters, '-frames:v', '1', + '-c:v', 'libx265', '-preset', 'fast', '-crf', '28', '-x265-params', parameters, + '-f', 'hevc', File.join(output, "#{name}.hevc")) +end + +cases = [ + ['direct', 'direct', 1, 1, {}], + ['tall', 'tall', 1, 1, {}], + ['pipeline', 'pipeline', 1, 1, {}], + ['grid', 'direct', 2, 2, {}], + ['grid-pipeline', 'pipeline', 2, 2, {}], + ['crop', 'tall', 1, 1, { 'CROP' => '252x2040' }], + ['oriented', 'direct', 1, 1, { 'CROP' => '130x190', 'ROTATION' => '1', 'MIRROR' => '1' }], + ['odd-short', 'direct', 1, 1, { 'CROP' => '130x190' }], + ['odd-tall', 'tall', 1, 1, { 'CROP' => '130x1982' }], + ['odd-pipeline', 'pipeline', 1, 1, { 'CROP' => '898x1022' }], + ['odd-width-grid', 'edge', 1, 1, { 'GRID' => '1', 'CANVAS' => '255x256', 'CROP' => '253x254' }], + ['odd-height-grid', 'vertical', 1, 1, { 'GRID' => '1', 'CANVAS' => '256x255', 'CROP' => '254x253' }], + ['undefined-grid-nclx', 'direct', 2, 2, { 'TILE_NCLX' => '1/13/1/0', 'PRIMARY_NCLX' => '2/2/2/1' }], + ['invalid-grid-references', 'direct', 2, 2, { 'REFERENCE_COUNT' => '65535' }] +] +cases.each do |name, encoded, columns, rows, options| + _, width, height = encodes.find { |entry| entry[0] == encoded } + extension = name == 'invalid-grid-references' ? 'bin' : 'heic' + path = File.join(output, "#{name}.#{extension}") + environment = %w[GRID CANVAS CROP ROTATION MIRROR TILE_NCLX PRIMARY_NCLX REFERENCE_COUNT].to_h { |key| [key, nil] }.merge(options) + execute(environment, RbConfig.ruby, File.join(__dir__, 'wrap.rb'), File.join(output, "#{encoded}.hevc"), + path, width.to_s, height.to_s, columns.to_s, rows.to_s) + oracle_png(oracle, path, path.sub(/\.heic$/, '.png')) if extension == 'heic' + puts "Generated #{path}" +end +manifest = { + source: { path: source, sha256: Digest::SHA256.file(source).hexdigest }, + oracle_sha256: Digest::SHA256.file(oracle).hexdigest, + ffmpeg: execute('ffmpeg', '-version').lines.first.strip, + inputs: cases.to_h do |name, *_| + path = File.join(output, "#{name}.#{name == 'invalid-grid-references' ? 'bin' : 'heic'}") + [File.basename(path), Digest::SHA256.file(path).hexdigest] + end +} +File.write(File.join(output, 'manifest.json'), JSON.pretty_generate(manifest)) diff --git a/scripts/incremental/primary-oracle.cc b/scripts/incremental/primary-oracle.cc new file mode 100644 index 0000000..4b7ad93 --- /dev/null +++ b/scripts/incremental/primary-oracle.cc @@ -0,0 +1,92 @@ +#include +#include +#include +#include +#include +#include +#include +#include +#include + +void check(heif_error error) { + if (error.code != heif_error_Ok) { + throw std::runtime_error(error.message); + } +} + +int main(int argc, char **argv) { + try { + if (argc < 2 || argc > 4) { + throw std::runtime_error("Usage: primary-oracle INPUT [PNG|-] [strict|recover]"); + } + auto context = std::unique_ptr( + heif_context_alloc(), heif_context_free); + if (!context) { + throw std::runtime_error("Cannot allocate decoder context"); + } + check(heif_context_read_from_file(context.get(), argv[1], nullptr)); + heif_item_id primary = 0; + check(heif_context_get_primary_image_ID(context.get(), &primary)); + heif_image_handle *raw_handle = nullptr; + check(heif_context_get_primary_image_handle(context.get(), &raw_handle)); + auto handle = std::unique_ptr( + raw_handle, heif_image_handle_release); + auto options = std::unique_ptr( + heif_decoding_options_alloc(), heif_decoding_options_free); + if (!options) { + throw std::runtime_error("Cannot allocate decoder options"); + } + options->strict_decoding = argc < 4 || std::string(argv[3]) != "recover"; + heif_image *raw_image = nullptr; + check(heif_decode_image(handle.get(), &raw_image, heif_colorspace_RGB, + heif_chroma_interleaved_RGBA, options.get())); + auto image = std::unique_ptr( + raw_image, heif_image_release); + int warnings = heif_image_get_decoding_warnings(image.get(), 0, nullptr, 0); + int width = heif_image_get_width(image.get(), heif_channel_interleaved); + int height = heif_image_get_height(image.get(), heif_channel_interleaved); + std::cout << "{\"primary_id\":" << primary << ",\"width\":" << width + << ",\"height\":" << height << ",\"warnings\":" << warnings + << ",\"strict\":" << (options->strict_decoding ? "true" : "false") << "}\n"; + if (argc >= 3 && std::string(argv[2]) != "-") { + std::vector icc(heif_image_get_raw_color_profile_size(image.get())); + if (!icc.empty()) { + check(heif_image_get_raw_color_profile(image.get(), icc.data())); + } + int stride = 0; + const uint8_t *plane = heif_image_get_plane_readonly(image.get(), heif_channel_interleaved, &stride); + FILE *file = std::fopen(argv[2], "wb"); + if (!file) { + throw std::runtime_error("Cannot open PNG output"); + } + png_structp png = png_create_write_struct(PNG_LIBPNG_VER_STRING, nullptr, nullptr, nullptr); + if (!png) { + std::fclose(file); + throw std::runtime_error("Cannot allocate PNG writer"); + } + png_infop info = png_create_info_struct(png); + if (!info || setjmp(png_jmpbuf(png))) { + png_destroy_write_struct(&png, info ? &info : nullptr); + std::fclose(file); + throw std::runtime_error("Cannot write PNG"); + } + png_init_io(png, file); + png_set_IHDR(png, info, width, height, 8, PNG_COLOR_TYPE_RGBA, + PNG_INTERLACE_NONE, PNG_COMPRESSION_TYPE_DEFAULT, PNG_FILTER_TYPE_DEFAULT); + if (!icc.empty()) { + png_set_iCCP(png, info, "ICC", PNG_COMPRESSION_TYPE_BASE, icc.data(), icc.size()); + } + png_write_info(png, info); + for (int y = 0; y < height; ++y) { + png_write_row(png, plane + static_cast(y) * stride); + } + png_write_end(png, info); + png_destroy_write_struct(&png, &info); + std::fclose(file); + } + return warnings == 0 ? 0 : 2; + } catch (const std::exception &error) { + std::cerr << error.what() << '\n'; + return 1; + } +} diff --git a/scripts/incremental/verify.rb b/scripts/incremental/verify.rb new file mode 100644 index 0000000..a73fd21 --- /dev/null +++ b/scripts/incremental/verify.rb @@ -0,0 +1,53 @@ +require 'digest' +require 'fileutils' +require 'json' +require 'open3' +require 'timeout' + +abort 'Usage: verify.rb ORACLE BOUNDED COMPARATOR INPUT...' if ARGV.size < 4 +oracle, bounded, comparator, *inputs = ARGV.map { |p| File.expand_path(p) } +root = File.expand_path('../..', __dir__) +output = File.expand_path(ENV.fetch('INCREMENTAL_TEST_ROOT', File.join(root, '.heic-test-runs/incremental'))) +FileUtils.mkdir_p(output) +def execute(*args) + Open3.popen3(*args, pgroup: true) do |input, out, err, thread| + input.close + stdout = Thread.new { out.read } + stderr = Thread.new { err.read } + begin + status = Timeout.timeout(120) { thread.value } + raise "Command failed (#{status}): #{args.inspect}\n#{stderr.value}\n#{stdout.value}" unless status.success? + stdout.value + rescue Timeout::Error + Process.kill('KILL', -thread.pid) + thread.value + raise "Timed out: #{args.inspect}" + end + end +end + +manifest = { + binaries: [oracle, bounded, comparator].to_h { |p| [p, Digest::SHA256.file(p).hexdigest] }, + inputs: inputs.to_h { |p| [p, Digest::SHA256.file(p).hexdigest] }, + profile: 'rgb8-rounding' +} +File.write(File.join(output, 'manifest.json'), JSON.pretty_generate(manifest)) +File.open(File.join(output, 'results.jsonl'), 'w') do |report| + inputs.each do |path| + dir = File.join(output, Digest::SHA256.hexdigest(path)) + FileUtils.mkdir_p(dir) + png = File.join(dir, 'reference.png') + metadata = JSON.parse(execute(oracle, path, png, 'strict')) + raise 'Oracle recovered from errors' unless metadata.fetch('strict') && metadata.fetch('warnings').zero? + reference = 'strict-libheif-primary' + [6000, 65].each do |side| + raw = File.join(dir, 'actual.rgb') + decoded = JSON.parse(execute(bounded, 'bounded', path, side.to_s, '128', raw)) + metrics = JSON.parse(execute(comparator, 'png', png, raw, decoded.fetch('width').to_s, decoded.fetch('height').to_s, side.to_s, 'rgb8-rounding')) + report.puts(JSON.generate({ path: path, side: side, reference: reference, decoded: decoded, metrics: metrics })) + report.flush + FileUtils.rm_f(raw) + end + puts "PASS #{path} (#{reference})" + end +end diff --git a/scripts/incremental/wrap.rb b/scripts/incremental/wrap.rb new file mode 100644 index 0000000..36ff40d --- /dev/null +++ b/scripts/incremental/wrap.rb @@ -0,0 +1,90 @@ +input, output, width, height, columns, rows = ARGV +abort 'Usage: wrap.rb ANNEX_B OUTPUT WIDTH HEIGHT [COLUMNS] [ROWS]' unless input && output && width && height +width = Integer(width) +height = Integer(height) +columns = Integer(columns || '1') +rows = Integer(rows || '1') +raise 'Invalid fixture dimensions' unless width.positive? && height.positive? && (1..256).cover?(columns) && (1..256).cover?(rows) +nals = File.binread(input).split(/\x00\x00\x00?\x01/n).reject(&:empty?) +parameters = (32..34).map do |kind| + nals.find { |nal| (nal.getbyte(0) >> 1 & 63) == kind } || raise('Missing parameter set') +end +pictures = nals.select { |nal| [19, 20].include?(nal.getbyte(0) >> 1 & 63) } +raise 'One IDR required' unless pictures.size == 1 +payload = pictures.map { |nal| [nal.bytesize].pack('N') + nal }.join +sps = parameters[1].byteslice(2..).gsub(/\x00\x00\x03/n, "\x00\x00".b) +hvcc = [1].pack('C') + sps.byteslice(1, 12) + [0xf0, 0, 0xfc, 0xfd, 0xf8, 0xf8, 0, 0, 0x0f, 3].pack('C*') +parameters.each_with_index { |nal, i| hvcc += [0x80 + 32 + i, 1, nal.bytesize].pack('Cnn') + nal } +def box(kind, data) + [data.bytesize + 8].pack('N') + kind + data +end + +def full(kind, version, data) + box(kind, [version << 24].pack('N') + data) +end + +count = columns * rows +grid = count > 1 || ENV['GRID'] == '1' +total = count + (grid ? 1 : 0) +primary = grid ? total : 1 +canvas_width, canvas_height = ENV.fetch('CANVAS', "#{width * columns}x#{height * rows}").split('x').map { |n| Integer(n) } +raise 'Invalid canvas' unless canvas_width.positive? && canvas_height.positive? && canvas_width <= width * columns && canvas_height <= height * rows +ispe = ->(w, h) { full('ispe', 0, [w, h].pack('NN')) } +properties = [box('hvcC', hvcc), ispe.call(width, height), full('pixi', 0, [3, 8, 8, 8].pack('C*'))] +properties << ispe.call(canvas_width, canvas_height) if grid +tile_color = nil +primary_color = nil +[['TILE_NCLX', :tile], ['PRIMARY_NCLX', :primary]].each do |key, target| + next unless ENV[key] + primaries, transfer, matrix, range = ENV.fetch(key).split('/').map { |n| Integer(n) } + properties << box('colr', 'nclx' + [primaries, transfer, matrix, range << 7].pack('nnnC')) + target == :tile ? tile_color = properties.length : primary_color = properties.length +end +extra = [] +if ENV['CROP'] + w, h = ENV.fetch('CROP').split('x').map { |n| Integer(n) } + properties << box('clap', [w, 1, h, 1, 0, 1, 0, 1].pack('N*')) + extra << properties.length +end +if ENV['ROTATION'] + properties << box('irot', [Integer(ENV.fetch('ROTATION'))].pack('C')) + extra << properties.length +end +if ENV['MIRROR'] + properties << box('imir', [Integer(ENV.fetch('MIRROR'))].pack('C')) + extra << properties.length +end +items = (1..total).map do |id| + kind = id == primary && grid ? 'grid' : 'hvc1' + full('infe', 2, [id, 0].pack('nn') + kind + "\0") +end.join +ipma = [total].pack('N') + (1..total).map do |id| + associations = id == primary && grid ? [0x84] : [0x81, 2, 3] + associations << tile_color if tile_color && !(id == primary && grid) + associations << primary_color if primary_color && id == primary + associations += extra.map { |n| n | 0x80 } if id == primary + [id, associations.length].pack('nC') + associations.pack('C*') +end.join +targets = ENV['REFERENCE_COUNT'] ? Array.new(Integer(ENV.fetch('REFERENCE_COUNT')), 1) : (1..count).to_a +raise 'Invalid reference count' unless (1..65_535).cover?(targets.length) +references = grid ? full('iref', 0, box('dimg', [primary, targets.length, *targets].pack('n*'))) : ''.b +griddata = grid ? [0, 1, rows - 1, columns - 1, canvas_width, canvas_height].pack('CCCCNN') : ''.b +ftyp = box('ftyp', "heic\0\0\0\0mif1heic".b) +make_meta = lambda do |offset| + locations = [0x44, 0, total].pack('CCn') + (1..total).each do |id| + length = id == primary && grid ? griddata.bytesize : payload.bytesize + locations += [id, 0, 1].pack('nnn') + [offset, length].pack('NN') + offset += length + end + full('meta', 0, + full('hdlr', 0, "\0\0\0\0pict".b + "\0" * 13) + + full('pitm', 0, [primary].pack('n')) + + full('iinf', 0, [total].pack('n') + items) + + full('iloc', 0, locations) + + box('iprp', box('ipco', properties.join) + full('ipma', 0, ipma)) + references) +end +meta = make_meta.call(0) +meta = make_meta.call(ftyp.bytesize + meta.bytesize + 8) +File.binwrite(output, ftyp + meta + box('mdat', payload * count + griddata)) +puts "#{output}: #{canvas_width}x#{canvas_height}, #{count} coded items" diff --git a/src/bin/capability-inspect.rs b/src/bin/capability-inspect.rs new file mode 100644 index 0000000..a0e75fb --- /dev/null +++ b/src/bin/capability-inspect.rs @@ -0,0 +1,64 @@ +extern crate alloc; +extern crate heic_decoder as api; +#[path = "../heic-decoder/mod.rs"] +mod heic_decoder; + +use api::isobmff::{HeicPrimaryItemDataWithGrid, HevcDecoderConfigurationBox}; +use heic_decoder::hevc::bitstream::parse_single_nal_bounded; +use heic_decoder::hevc::params::{parse_pps_bounded, parse_sps_bounded}; + +fn inspect(id: u32, config: &HevcDecoderConfigurationBox, data: &[u8]) { + let nals: Vec<_> = config + .nal_arrays + .iter() + .flat_map(|a| a.nal_units.iter()) + .collect(); + let parameter = |kind| nals.iter().find(|n| n[0] >> 1 & 63 == kind).unwrap(); + let sps_nal = parse_single_nal_bounded(parameter(33)).unwrap(); + let sps = parse_sps_bounded(&sps_nal.payload).unwrap(); + let pps_nal = parse_single_nal_bounded(parameter(34)).unwrap(); + let pps = parse_pps_bounded(&pps_nal.payload).unwrap(); + let mut types = Vec::new(); + let mut pos = 0; + while pos < data.len() { + let n = usize::from(config.nal_length_size); + let len = data[pos..pos + n] + .iter() + .fold(0usize, |v, b| v << 8 | usize::from(*b)); + pos += n; + types.push(data[pos] >> 1 & 63); + pos += len; + } + println!( + "{{\"item\":{id},\"width\":{},\"height\":{},\"depth\":{},\"chroma\":{},\"ctu\":{},\"wpp\":{},\"tiles\":{},\"dependent_slices\":{},\"pcm\":{},\"sao\":{},\"transform_skip\":{},\"lossless_bypass\":{},\"nals\":{:?}}}", + sps.pic_width_in_luma_samples, + sps.pic_height_in_luma_samples, + sps.bit_depth_y(), + sps.chroma_format_idc, + sps.ctb_size(), + pps.entropy_coding_sync_enabled_flag, + pps.tiles_enabled_flag, + pps.dependent_slice_segments_enabled_flag, + sps.pcm_enabled_flag, + sps.sample_adaptive_offset_enabled_flag, + pps.transform_skip_enabled_flag, + pps.transquant_bypass_enabled_flag, + types + ); +} + +fn main() { + let path = std::env::args().nth(1).unwrap(); + let bytes = std::fs::read(path).unwrap(); + match api::isobmff::extract_primary_heic_item_data_with_grid(&bytes).unwrap() { + HeicPrimaryItemDataWithGrid::Coded(item) => { + let props = api::isobmff::parse_primary_heic_item_preflight_properties(&bytes).unwrap(); + inspect(item.item_id, &props.hvcc, &item.payload); + } + HeicPrimaryItemDataWithGrid::Grid(grid) => { + for tile in grid.tiles { + inspect(tile.item_id, &tile.hvcc, &tile.payload); + } + } + } +} diff --git a/src/bin/incremental-allocation.rs b/src/bin/incremental-allocation.rs new file mode 100644 index 0000000..5607957 --- /dev/null +++ b/src/bin/incremental-allocation.rs @@ -0,0 +1,96 @@ +#[path = "support/allocation.rs"] +mod allocation; +use allocation::*; +use std::path::Path; +use std::sync::atomic::Ordering; +use std::time::Instant; + +fn main() { + let args: Vec = std::env::args().collect(); + if !(3..=6).contains(&args.len()) || !matches!(args[1].as_str(), "bounded" | "bytes" | "normal") + { + eprintln!( + "Usage: incremental-allocation bounded|bytes|normal INPUT [MAX_SIDE] [MIB] [RGB_OUTPUT]" + ); + std::process::exit(2); + } + let mode = args[1].as_str(); + let path = Path::new(&args[2]); + let side = args.get(3).and_then(|s| s.parse().ok()).unwrap_or(6000); + let mib = args + .get(4) + .and_then(|s| s.parse::().ok()) + .unwrap_or(128); + if let Ok(value) = std::env::var("DENY_REQUESTS_ABOVE") { + MAX_REQUEST.store(value.parse().unwrap(), Ordering::Relaxed); + } + let borrowed = if mode == "bytes" { + Some(std::fs::read(path).unwrap()) + } else { + None + }; + let baseline = LIVE.load(Ordering::Relaxed); + if !mode.starts_with("normal") { + MAX_LIVE.store(baseline + mib * 1024 * 1024, Ordering::Relaxed); + } + PEAK.store(baseline, Ordering::Relaxed); + ACTIVE.store(true, Ordering::Relaxed); + let start = Instant::now(); + let result = if mode == "normal" { + heic_decoder::decode_path_to_rgb8(path).map_err(|e| e.to_string()) + } else { + let input = borrowed + .as_ref() + .map_or(heic_decoder::BoundedInput::Path(path), |bytes| { + heic_decoder::BoundedInput::Bytes(bytes) + }); + heic_decoder::decode_incremental_experiment( + input, + heic_decoder::BoundedDecodeOptions { + max_side: side, + max_memory_bytes: mib * 1024 * 1024, + }, + ) + .map(|(image, stats)| { + eprintln!( + "decoded_rows={} sao_edge_components={} sao_band_components={}", + stats[0], stats[1], stats[2] + ); + image.image + }) + .map_err(|e| e.to_string()) + }; + let elapsed = start.elapsed(); + ACTIVE.store(false, Ordering::Relaxed); + let peak = PEAK.load(Ordering::Relaxed).saturating_sub(baseline); + eprintln!( + "denied_allocation_attempts={} borrowed_input_bytes={}", + DENIED.load(Ordering::Relaxed), + borrowed.as_ref().map_or(0, Vec::len) + ); + let largest = LARGEST.load(Ordering::Relaxed); + let workers = THREADS.load(Ordering::Relaxed).saturating_sub(1); + match result { + Ok(image) => { + println!( + "{{\"mode\":\"{mode}\",\"side\":{side},\"budget_mib\":{mib},\"width\":{},\"height\":{},\"milliseconds\":{:.3},\"peak_bytes\":{peak},\"largest_bytes\":{largest},\"allocating_workers\":{workers}}}", + image.width, + image.height, + elapsed.as_secs_f64() * 1000.0 + ); + if !mode.starts_with("normal") { + assert!(peak <= mib * 1024 * 1024); + } + if let Some(output) = args.get(5) { + std::fs::write(output, image.pixels).unwrap(); + } + } + Err(error) => { + println!( + "{{\"error\":\"{error}\",\"milliseconds\":{:.3},\"peak_bytes\":{peak},\"largest_bytes\":{largest}}}", + elapsed.as_secs_f64() * 1000.0 + ); + std::process::exit(1); + } + } +} diff --git a/src/bin/incremental-bench.rs b/src/bin/incremental-bench.rs new file mode 100644 index 0000000..9cf8b65 --- /dev/null +++ b/src/bin/incremental-bench.rs @@ -0,0 +1,37 @@ +use heic_decoder::{BoundedDecodeOptions, BoundedInput}; +use std::hint::black_box; +use std::path::Path; +use std::time::Instant; + +fn main() -> Result<(), Box> { + let args: Vec = std::env::args().collect(); + if !(3..=4).contains(&args.len()) { + return Err("Usage: incremental-bench bounded|normal INPUT [MAX_SIDE]".into()); + } + let mode = &args[1]; + let input = Path::new(&args[2]); + let max_side = args.get(3).map(|s| s.parse()).transpose()?.unwrap_or(6000); + let start = Instant::now(); + let image = match mode.as_str() { + "bounded" => { + heic_decoder::decode_incremental_experiment( + BoundedInput::Path(input), + BoundedDecodeOptions { + max_side, + ..Default::default() + }, + )? + .0 + .image + } + "normal" => heic_decoder::decode_path_to_rgb8(input)?, + _ => return Err("Mode must be bounded or normal".into()), + }; + black_box(&image.pixels); + let elapsed = start.elapsed().as_secs_f64() * 1000.0; + println!( + "{{\"mode\":\"{mode}\",\"width\":{},\"height\":{},\"milliseconds\":{elapsed:.6}}}", + image.width, image.height + ); + Ok(()) +} diff --git a/src/bin/incremental-check.rs b/src/bin/incremental-check.rs new file mode 100644 index 0000000..aabd35c --- /dev/null +++ b/src/bin/incremental-check.rs @@ -0,0 +1,86 @@ +extern crate alloc; +extern crate heic_decoder as api; +#[path = "../heic-decoder/mod.rs"] +mod heic_decoder; + +fn main() { + let paths: Vec<_> = std::env::args().skip(1).collect(); + assert!(!paths.is_empty(), "Usage: incremental-check INPUT..."); + let annex_b_output = std::env::var_os("ANNEX_B_OUTPUT"); + assert!(annex_b_output.is_none() || paths.len() == 1); + for path in paths { + let bytes = std::fs::read(&path).unwrap(); + let item = api::isobmff::extract_primary_heic_item_data(&bytes).unwrap(); + let props = api::isobmff::parse_primary_heic_item_preflight_properties(&bytes).unwrap(); + let config = &props.hvcc; + let mut all: Vec<&[u8]> = config + .nal_arrays + .iter() + .flat_map(|a| a.nal_units.iter().map(Vec::as_slice)) + .collect(); + let mut pos = 0; + while pos < item.payload.len() { + let n = usize::from(config.nal_length_size); + let len = item.payload[pos..pos + n] + .iter() + .fold(0usize, |v, b| (v << 8) | usize::from(*b)); + pos += n; + all.push(&item.payload[pos..pos + len]); + pos += len; + } + let parameter = |kind| *all.iter().find(|n| (n[0] >> 1) & 63 == kind).unwrap(); + if let Some(output) = &annex_b_output { + use std::io::Write; + let mut file = std::fs::File::create(output).unwrap(); + for nal in &all { + file.write_all(&[0, 0, 0, 1]).unwrap(); + file.write_all(nal).unwrap(); + } + } + let Some(slice) = all.iter().find(|n| matches!((n[0] >> 1) & 63, 19 | 20)) else { + continue; + }; + let prepare = || { + heic_decoder::hevc::bounded::Prepared::new( + parameter(32), + parameter(33), + parameter(34), + slice, + ) + .unwrap() + }; + let reference = prepare().decode().unwrap(); + let independent = std::env::var("REFERENCE_YUV") + .ok() + .map(|p| std::fs::read(p).unwrap()); + if let Some(bytes) = &independent { + assert_eq!( + bytes.len(), + reference.width as usize * reference.height as usize * 3 / 2 + ); + } + let result = prepare().decode_incremental(|start, band| { + for component in 0..3 { + let sub = if component == 0 {1} else {2}; + let (expected, stride) = reference.plane(component); + let (actual, _) = band.plane(component); + let offset = (start / sub) as usize * stride; + let len = (band.height / sub) as usize * stride; + if let Some(bytes) = &independent { + let plane_offset = match component { 0 => 0, 1 => reference.width as usize * reference.height as usize, _ => reference.width as usize * reference.height as usize * 5 / 4 }; + let expected = &bytes[plane_offset + offset..plane_offset + offset + len]; + if let Some(i) = actual[..len].iter().zip(expected).position(|(&a,&b)| a != u16::from(b)) { + panic!("independent YUV difference: component {component} ({},{}) actual={} expected={}", i%stride, start/sub + (i/stride) as u32, actual[i],expected[i]); + } + } + if let Some(i) = actual[..len] +.iter().zip(&expected[offset..offset+len]).position(|(a,b)| a != b) { + panic!("{path}: component {component} ({},{}) actual={} expected={}", i%stride, start/sub + (i/stride) as u32, actual[i],expected[offset+i]); + } + } + Ok(()) + }).unwrap(); + assert_eq!(result.rows, reference.height); + println!("{path}: {result:?}"); + } +} diff --git a/src/bin/incremental-compare.rs b/src/bin/incremental-compare.rs new file mode 100644 index 0000000..e00468e --- /dev/null +++ b/src/bin/incremental-compare.rs @@ -0,0 +1,306 @@ +use std::error::Error; +use std::fs::File; +use std::io::BufReader; + +struct Reference { + width: usize, + height: usize, + pixels: Vec, + transformed: bool, +} + +fn read_reference(path: &str) -> Result> { + let mut decoder = png::Decoder::new(BufReader::new(File::open(path)?)); + decoder.set_transformations(png::Transformations::EXPAND); + let mut reader = decoder.read_info()?; + let icc = reader.info().icc_profile.as_ref().map(|p| p.to_vec()); + let mut bytes = vec![0; reader.output_buffer_size().ok_or("PNG size overflow")?]; + let info = reader.next_frame(&mut bytes)?; + if info.bit_depth != png::BitDepth::Eight { + return Err("Expected an 8-bit reference".into()); + } + bytes.truncate(info.buffer_size()); + let pixels = match info.color_type { + png::ColorType::Rgb => bytes, + png::ColorType::Rgba => { + if bytes.chunks_exact(4).any(|p| p[3] != 255) { + return Err("Reference has nonopaque alpha".into()); + } + bytes + .chunks_exact(4) + .flat_map(|p| p[..3].iter().copied()) + .collect() + } + _ => return Err("Expected RGB reference".into()), + }; + let transformed = icc.is_some(); + let pixels = if let Some(icc) = icc { + let profile = moxcms::ColorProfile::new_from_slice(&icc)?; + let transform = profile.create_transform_8bit( + moxcms::Layout::Rgb, + &moxcms::ColorProfile::new_srgb(), + moxcms::Layout::Rgb, + Default::default(), + )?; + let mut output = vec![0; pixels.len()]; + transform.transform(&pixels, &mut output)?; + output + } else { + pixels + }; + Ok(Reference { + width: info.width as usize, + height: info.height as usize, + pixels, + transformed, + }) +} + +fn area_reference(source: &[u8], sw: usize, sh: usize, w: usize, h: usize) -> Vec { + if (sw, sh) == (w, h) { + return source.to_vec(); + } + let mut output = vec![0; w * h * 3]; + for y in 0..h { + let y0 = y as f64 * sh as f64 / h as f64; + let y1 = (y + 1) as f64 * sh as f64 / h as f64; + for x in 0..w { + let x0 = x as f64 * sw as f64 / w as f64; + let x1 = (x + 1) as f64 * sw as f64 / w as f64; + let mut sum = [0.0; 3]; + for sy in y0.floor() as usize..(y1.ceil() as usize).min(sh) { + let wy = y1.min((sy + 1) as f64) - y0.max(sy as f64); + for sx in x0.floor() as usize..(x1.ceil() as usize).min(sw) { + let weight = wy * (x1.min((sx + 1) as f64) - x0.max(sx as f64)); + for c in 0..3 { + sum[c] += f64::from(source[(sy * sw + sx) * 3 + c]) * weight; + } + } + } + for c in 0..3 { + output[(y * w + x) * 3 + c] = (sum[c] / ((x1 - x0) * (y1 - y0))).round() as u8; + } + } + } + output +} + +fn metrics(expected: &[u8], actual: &[u8], width: usize, channels: usize) -> String { + assert_eq!(expected.len(), actual.len()); + let mut histogram = [0u64; 256]; + let mut signed = [0i64; 3]; + let mut square = 0u64; + let mut pixel_count = 0u64; + let mut alpha_errors = 0u64; + let mut border_above_one = 0u64; + let height = expected.len() / channels / width; + let mut tile_sums = + vec![0u64; width.div_ceil(16) * (expected.len() / channels / width).div_ceil(16)]; + let mut tile_counts = vec![0u64; tile_sums.len()]; + for (i, (a, b)) in expected + .chunks_exact(channels) + .zip(actual.chunks_exact(channels)) + .enumerate() + { + let tile = (i / width / 16) * width.div_ceil(16) + (i % width / 16); + pixel_count += u64::from(a[..3] != b[..3]); + if channels == 4 { + alpha_errors += u64::from(a[3] != b[3]); + } + for c in 0..3 { + let error = a[c].abs_diff(b[c]); + if error > 1 + && (i % width == 0 + || i % width == width - 1 + || i / width == 0 + || i / width == height - 1) + { + border_above_one += 1; + } + histogram[usize::from(error)] += 1; + signed[c] += i64::from(b[c]) - i64::from(a[c]); + square += u64::from(error).pow(2); + tile_sums[tile] += u64::from(error); + tile_counts[tile] += 1; + } + } + let count: u64 = histogram.iter().sum(); + let sum: u64 = histogram + .iter() + .enumerate() + .map(|(e, &n)| e as u64 * n) + .sum(); + let max = histogram.iter().rposition(|&n| n > 0).unwrap_or(0); + let p999 = histogram + .iter() + .scan(0u64, |n, &v| { + *n += v; + Some(*n) + }) + .position(|n| n as f64 >= count as f64 * 0.999) + .unwrap_or(0); + let worst_tile_mae = tile_sums + .iter() + .zip(&tile_counts) + .map(|(&s, &n)| s as f64 / n as f64) + .fold(0.0, f64::max); + let psnr = if square == 0 { + "null".to_string() + } else { + format!( + "{:.8}", + 10.0 * (255.0f64.powi(2) * count as f64 / square as f64).log10() + ) + }; + let above_one: u64 = histogram[2..].iter().sum(); + let bias = signed.map(|s| s as f64 / (count / 3) as f64); + format!( + "{{\"max_error\":{max},\"mean_error\":{:.10},\"p999_error\":{p999},\"changed_pixels\":{pixel_count},\"above_one\":{above_one},\"border_above_one\":{border_above_one},\"psnr_db\":{psnr},\"worst_16x16_mae\":{worst_tile_mae:.10},\"channel_bias\":{bias:?},\"alpha_errors\":{alpha_errors},\"samples\":{count}}}", + sum as f64 / count as f64 + ) +} + +#[derive(Clone, Copy)] +enum Profile { + Exact, + Rgb8Rounding, +} + +impl Profile { + fn parse(value: &str) -> Result> { + match value { + "exact" => Ok(Self::Exact), + "rgb8-rounding" => Ok(Self::Rgb8Rounding), + _ => Err("Profile must be exact or rgb8-rounding".into()), + } + } + + fn accepts(self, expected: &[u8], actual: &[u8], channels: usize) -> bool { + if expected.is_empty() + || expected.len() != actual.len() + || !matches!(channels, 3 | 4) + || !expected.len().is_multiple_of(channels) + { + return false; + } + expected + .iter() + .zip(actual) + .enumerate() + .all(|(i, (&a, &b))| { + let allowance = match self { + Self::Exact => 0, + Self::Rgb8Rounding => u8::from(i % channels < 3), + }; + a.abs_diff(b) <= allowance + }) + } +} + +fn main() -> Result<(), Box> { + let args: Vec = std::env::args().collect(); + if args.len() < 7 { + return Err("Usage: comparator raw EXPECTED ACTUAL WIDTH CHANNELS PROFILE | png REFERENCE ACTUAL WIDTH HEIGHT SIDE PROFILE".into()); + } + if args[1] == "raw" { + let expected = std::fs::read(&args[2])?; + let actual = std::fs::read(&args[3])?; + let width: usize = args[4].parse()?; + let channels: usize = args[5].parse()?; + let profile = Profile::parse(&args[6])?; + if width == 0 + || !matches!(channels, 3 | 4) + || expected.is_empty() + || expected.len() != actual.len() + || !expected.len().is_multiple_of(width * channels) + { + return Err("Raw buffer geometry differs".into()); + } + println!("{}", metrics(&expected, &actual, width, channels)); + if !profile.accepts(&expected, &actual, channels) { + return Err("Pixel comparison failed".into()); + } + } else if args[1] == "png" && args.len() == 8 { + let profile = Profile::parse(&args[7])?; + let Reference { + width: sw, + height: sh, + pixels: source, + transformed, + } = read_reference(&args[2])?; + let actual = std::fs::read(&args[3])?; + let width: usize = args[4].parse()?; + let height: usize = args[5].parse()?; + let side: usize = args[6].parse()?; + let longest = sw.max(sh); + let expected_dimensions = if longest <= side { + (sw, sh) + } else { + ((sw * side / longest).max(1), (sh * side / longest).max(1)) + }; + if side == 0 || (width, height) != expected_dimensions || actual.len() != width * height * 3 + { + return Err(format!( + "Dimensions differ: reference={expected_dimensions:?}, actual={width}x{height}" + ) + .into()); + } + let expected = area_reference(&source, sw, sh, width, height); + eprintln!("reference={sw}x{sh} output={width}x{height} icc_transformed={transformed}"); + println!("{}", metrics(&expected, &actual, width, 3)); + if !profile.accepts(&expected, &actual, 3) { + return Err("Pixel comparison failed".into()); + } + } else { + return Err("Unknown comparison mode or argument count".into()); + } + Ok(()) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn rounding_and_local_corruption_remain_distinct() { + let zero = vec![0; 32 * 32 * 3]; + let one = vec![1; zero.len()]; + let mut corruption = zero.clone(); + corruption[0] = 255; + assert!(metrics(&zero, &one, 32, 3).contains("\"max_error\":1,")); + let stats = metrics(&zero, &corruption, 32, 3); + assert!(stats.contains("\"max_error\":255,")); + assert!(stats.contains("\"above_one\":1,")); + assert!(metrics(&[0, 0, 0, 255], &[0, 0, 0, 254], 1, 4).contains("\"alpha_errors\":1,")); + } + + #[test] + fn independent_area_average() { + assert_eq!( + area_reference(&[0, 0, 0, 100, 100, 100, 200, 200, 200], 3, 1, 2, 1), + [33, 33, 33, 167, 167, 167] + ); + } + + #[test] + fn display_rounding_does_not_relax_reconstruction_or_alpha() { + let expected = [100, 100, 100, 255]; + let rounded = [101, 99, 100, 255]; + assert!(Profile::Rgb8Rounding.accepts(&expected, &rounded, 4)); + assert!(!Profile::Exact.accepts(&expected, &rounded, 4)); + assert!(!Profile::Rgb8Rounding.accepts(&expected, &[100, 100, 100, 254], 4)); + assert!(!Profile::Rgb8Rounding.accepts(&expected, &rounded[..3], 4)); + } + + #[test] + fn local_defects_cannot_hide_in_an_average() { + let expected = vec![100; 64 * 64 * 3]; + let mut actual = expected.clone(); + actual[0] = 102; + assert!(!Profile::Rgb8Rounding.accepts(&expected, &actual, 3)); + actual[..64 * 3].fill(102); + assert!(!Profile::Rgb8Rounding.accepts(&expected, &actual, 3)); + let colors = [255, 0, 0, 0, 0, 255]; + assert!(!Profile::Rgb8Rounding.accepts(&colors, &[0, 0, 255, 255, 0, 0], 3)); + } +} diff --git a/src/bin/support/allocation.rs b/src/bin/support/allocation.rs new file mode 100644 index 0000000..df0ceb7 --- /dev/null +++ b/src/bin/support/allocation.rs @@ -0,0 +1,90 @@ +use std::alloc::{GlobalAlloc, Layout, System}; +use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering}; + +struct Tracking; +pub(crate) static LIVE: AtomicUsize = AtomicUsize::new(0); +pub(crate) static PEAK: AtomicUsize = AtomicUsize::new(0); +pub(crate) static ACTIVE: AtomicBool = AtomicBool::new(false); +pub(crate) static MAX_REQUEST: AtomicUsize = AtomicUsize::new(usize::MAX); +pub(crate) static MAX_LIVE: AtomicUsize = AtomicUsize::new(usize::MAX); +pub(crate) static LARGEST: AtomicUsize = AtomicUsize::new(0); +pub(crate) static DENIED: AtomicUsize = AtomicUsize::new(0); +pub(crate) static THREADS: AtomicUsize = AtomicUsize::new(0); +thread_local! { static COUNTED: std::cell::Cell = const { std::cell::Cell::new(false) }; } + +fn reserve(size: usize) -> bool { + let active = ACTIVE.load(Ordering::Relaxed); + let result = LIVE.fetch_update(Ordering::SeqCst, Ordering::SeqCst, |live| { + let next = live.checked_add(size)?; + if active + && (size > MAX_REQUEST.load(Ordering::Relaxed) + || next > MAX_LIVE.load(Ordering::Relaxed)) + { + None + } else { + Some(next) + } + }); + let Ok(previous) = result else { + DENIED.fetch_add(1, Ordering::Relaxed); + return false; + }; + if active { + COUNTED.with(|counted| { + if !counted.replace(true) { + THREADS.fetch_add(1, Ordering::Relaxed); + } + }); + PEAK.fetch_max(previous + size, Ordering::Relaxed); + LARGEST.fetch_max(size, Ordering::Relaxed); + } + true +} + +unsafe impl GlobalAlloc for Tracking { + unsafe fn alloc(&self, layout: Layout) -> *mut u8 { + if !reserve(layout.size()) { + return std::ptr::null_mut(); + } + let pointer = unsafe { System.alloc(layout) }; + if pointer.is_null() { + LIVE.fetch_sub(layout.size(), Ordering::SeqCst); + } + pointer + } + + unsafe fn alloc_zeroed(&self, layout: Layout) -> *mut u8 { + if !reserve(layout.size()) { + return std::ptr::null_mut(); + } + let pointer = unsafe { System.alloc_zeroed(layout) }; + if pointer.is_null() { + LIVE.fetch_sub(layout.size(), Ordering::SeqCst); + } + pointer + } + + unsafe fn dealloc(&self, pointer: *mut u8, layout: Layout) { + unsafe { System.dealloc(pointer, layout) }; + LIVE.fetch_sub(layout.size(), Ordering::SeqCst); + } + + unsafe fn realloc(&self, pointer: *mut u8, layout: Layout, size: usize) -> *mut u8 { + if !reserve(size) { + return std::ptr::null_mut(); + } + let result = unsafe { System.realloc(pointer, layout, size) }; + LIVE.fetch_sub( + if result.is_null() { + size + } else { + layout.size() + }, + Ordering::SeqCst, + ); + result + } +} + +#[global_allocator] +static ALLOCATOR: Tracking = Tracking; diff --git a/src/bounded.rs b/src/bounded.rs index 349b99d..2ffde39 100644 --- a/src/bounded.rs +++ b/src/bounded.rs @@ -2,6 +2,8 @@ mod codec; mod color; mod container; mod grid; +#[cfg(feature = "incremental-experiment")] +pub(crate) mod incremental; mod memory; mod resample; #[cfg(test)] diff --git a/src/bounded/codec.rs b/src/bounded/codec.rs index 0a19caa..0c6e41a 100644 --- a/src/bounded/codec.rs +++ b/src/bounded/codec.rs @@ -4,6 +4,21 @@ use super::container::Reader; use super::{BoundedDecodeError as Error, Result}; pub(super) fn prepare<'a>(config: &'a [u8], payload: &'a [u8]) -> Result> { + let (mut nals, length_size) = configuration(config)?; + let mut r = Reader::new(payload); + while r.pos < payload.len() { + let len = r.uint(length_size)? as usize; + nals.add(r.take(len)?, true)?; + } + Ok(Prepared::new( + nals.parameters[0].ok_or(Error::Malformed("missing VPS"))?, + nals.parameters[1].ok_or(Error::Malformed("missing SPS"))?, + nals.parameters[2].ok_or(Error::Malformed("missing PPS"))?, + nals.slice.ok_or(Error::Malformed("missing IDR slice"))?, + )?) +} + +pub(super) fn configuration(config: &[u8]) -> Result<(Nals<'_>, usize)> { let mut r = Reader::new(config); let header = r.take(23)?; if header[0] != 1 { @@ -27,28 +42,18 @@ pub(super) fn prepare<'a>(config: &'a [u8], payload: &'a [u8]) -> Result { - parameters: [Option<&'a [u8]>; 3], +pub(super) struct Nals<'a> { + pub(super) parameters: [Option<&'a [u8]>; 3], slice: Option<&'a [u8]>, count: usize, } impl<'a> Nals<'a> { - fn add(&mut self, nal: &'a [u8], in_payload: bool) -> Result<()> { + pub(super) fn add(&mut self, nal: &'a [u8], in_payload: bool) -> Result<()> { self.count += 1; if self.count > 1024 { return Err(Error::LimitExceeded("NAL count")); diff --git a/src/bounded/container.rs b/src/bounded/container.rs index c70fe86..793b6ad 100644 --- a/src/bounded/container.rs +++ b/src/bounded/container.rs @@ -273,7 +273,17 @@ fn unique<'a>(slot: &mut Option<&'a [u8]>, data: &'a [u8]) -> Result<()> { impl<'a> Index<'a> { pub(super) fn parse(metadata: &'a [u8], source_len: u64, budget: &Budget) -> Result { + Self::parse_primary(metadata, source_len, budget, false) + } + + pub(super) fn parse_primary( + metadata: &'a [u8], + source_len: u64, + budget: &Budget, + direct: bool, + ) -> Result { let mut reader = Reader::new(metadata); + if reader.full_box()? != (0, 0) { return Err(Error::Unsupported("meta version or flags")); } @@ -351,7 +361,9 @@ impl<'a> Index<'a> { let primary = items .binary_search_by_key(&primary_id, |item| item.id) .map_err(|_| Error::Malformed("primary item missing"))?; - if items[primary].kind != *b"grid" { + if items[primary].kind != *b"grid" + && !(direct && matches!(&items[primary].kind, b"hvc1" | b"hev1")) + { return Err(Error::Unsupported("primary item must be a HEIC grid")); } let idat = idat.unwrap_or_default(); @@ -596,6 +608,50 @@ impl<'a> Index<'a> { }) } + #[cfg(feature = "incremental-experiment")] + pub(super) fn read_range( + &self, + item: usize, + source: &mut Source<'_>, + mut position: usize, + mut output: &mut [u8], + ) -> Result<()> { + let location = self.items[item] + .location + .ok_or(Error::Malformed("missing item location"))?; + if position + .checked_add(output.len()) + .is_none_or(|end| end > location.length) + { + return Err(Error::Malformed("item range")); + } + let mut r = Reader::new(location.extents); + while !output.is_empty() && r.pos < r.data.len() { + r.uint(location.index_size)?; + let offset = location.base + r.uint(location.offset_size)?; + let length = r.uint(location.length_size)? as usize; + if position >= length { + position -= length; + continue; + } + let count = output.len().min(length - position); + let offset = offset + position as u64; + if location.method == 0 { + source.read(offset, &mut output[..count])?; + } else { + output[..count] + .copy_from_slice(&self.idat[offset as usize..offset as usize + count]); + } + output = &mut output[count..]; + position = 0; + } + if output.is_empty() { + Ok(()) + } else { + Err(Error::Malformed("item extent coverage")) + } + } + pub(super) fn read_item( &self, item: usize, diff --git a/src/bounded/grid.rs b/src/bounded/grid.rs index 8c1fcbc..43f17d6 100644 --- a/src/bounded/grid.rs +++ b/src/bounded/grid.rs @@ -24,6 +24,20 @@ pub(super) struct Properties<'a> { impl<'a> Properties<'a> { pub(super) fn read(index: &Index<'a>, item: usize, tile: bool) -> Result { + Self::read_with_compatibility(index, item, tile, false) + } + + #[cfg(feature = "incremental-experiment")] + pub(super) fn read_incremental(index: &Index<'a>, item: usize, tile: bool) -> Result { + Self::read_with_compatibility(index, item, tile, true) + } + + fn read_with_compatibility( + index: &Index<'a>, + item: usize, + tile: bool, + legacy_pixi: bool, + ) -> Result { let mut result = Self { config: None, dimensions: None, @@ -78,7 +92,9 @@ impl<'a> Properties<'a> { return Err(Error::Unsupported("pixi version or flags")); } let count = r.uint(1)? as usize; - if count != 3 || r.take(count)?.iter().any(|&n| n != 8) { + if (count != 3 && !(legacy_pixi && count == 1)) + || r.take(count)?.iter().any(|&n| n != 8) + { return Err(Error::Unsupported("8-bit RGB channels required")); } r.finish()?; @@ -293,90 +309,7 @@ impl<'a> Grid<'a> { { return Err(Error::Malformed("grid coverage")); } - let mut exif = None; - for reference in index.references.iter() { - let required_source = reference.from == primary_id - || tiles - .iter() - .any(|tile| index.items[tile.item].id == reference.from); - let touches = required_source - || reference.targets().any(|id| { - id == primary_id || tiles.iter().any(|tile| index.items[tile.item].id == id) - }); - if !touches { - continue; - } - match &reference.kind { - b"dimg" if reference.from == primary_id => {} - b"dimg" if !required_source => {} - b"auxl" => { - let item = index.item(reference.from)?; - let mut found = false; - for (essential, property) in index.properties(item) { - if property.kind != *b"auxC" { - continue; - } - let mut r = Reader::new(property.data); - if r.full_box()? != (0, 0) { - return Err(Error::Unsupported("auxC version or flags")); - } - let value = &r.data[r.pos..]; - let end = value - .iter() - .position(|&b| b == 0) - .ok_or(Error::Malformed("auxiliary type"))?; - let value = &value[..end]; - if crate::ALPHA_AUX_TYPES.contains(&value) { - return Err(Error::Unsupported("alpha auxiliary")); - } - let known = [ - b"urn:com:apple:photo:2020:aux:hdrgainmap".as_slice(), - b"urn:com:apple:photo:2020:aux:semanticskymatte", - b"urn:com:apple:photo:2018:aux:portraiteffectsmatte", - b"urn:com:apple:photo:2019:aux:semanticskinmatte", - b"urn:com:apple:photo:2019:aux:semantichairmatte", - b"urn:com:apple:photo:2019:aux:semanticteethmatte", - b"urn:com:apple:photo:2020:aux:semanticglassesmatte", - b"tag:apple.com,2023:photo:aux:styledeltamap", - b"tag:apple.com,2023:photo:aux:linearthumbnail", - b"urn:mpeg:hevc:2015:auxid:2", - b"urn:iso:std:iso:ts:21496:-1", - ] - .contains(&value); - if essential && !known { - return Err(Error::Unsupported("required auxiliary type")); - } - found = true; - } - if !found { - return Err(Error::Malformed("auxiliary type missing")); - } - } - b"cdsc" => { - let item = index.item(reference.from)?; - if index.items[item].is_exif && reference.targets().any(|id| id == primary_id) { - let len = index.items[item] - .location - .ok_or(Error::Malformed("Exif location"))? - .length; - if len > 1024 * 1024 { - return Err(Error::LimitExceeded("Exif bytes")); - } - let mut bytes = budget.zeroed(len, "Exif")?; - index.read_item(item, source, &mut bytes)?; - let orientation = crate::parse_exif_orientation_from_item_payload(&bytes) - .and_then(|n| u8::try_from(n).ok()) - .filter(|n| (1..=8).contains(n)); - if exif.is_some() && exif != orientation { - return Err(Error::Malformed("conflicting Exif orientation")); - } - exif = orientation; - } - } - b"thmb" => {} - _ => return Err(Error::Unsupported("required item dependency")), - } - } + let exif = validate_references(index, source, &tiles, budget)?; Ok(Self { width, height, @@ -387,3 +320,98 @@ impl<'a> Grid<'a> { }) } } + +pub(super) fn validate_references( + index: &Index<'_>, + source: &mut Source<'_>, + tiles: &[Tile<'_>], + budget: &Budget, +) -> Result> { + let primary_id = index.items[index.primary].id; + let mut exif = None; + for reference in index.references.iter() { + let required_source = reference.from == primary_id + || tiles + .iter() + .any(|tile| index.items[tile.item].id == reference.from); + let touches = required_source + || reference.targets().any(|id| { + id == primary_id || tiles.iter().any(|tile| index.items[tile.item].id == id) + }); + if !touches { + continue; + } + match &reference.kind { + b"dimg" + if reference.from == primary_id && index.items[index.primary].kind == *b"grid" => {} + b"dimg" if !required_source => {} + b"auxl" => { + let item = index.item(reference.from)?; + let mut found = false; + for (essential, property) in index.properties(item) { + if property.kind != *b"auxC" { + continue; + } + let mut r = Reader::new(property.data); + if r.full_box()? != (0, 0) { + return Err(Error::Unsupported("auxC version or flags")); + } + let value = &r.data[r.pos..]; + let end = value + .iter() + .position(|&b| b == 0) + .ok_or(Error::Malformed("auxiliary type"))?; + let value = &value[..end]; + if crate::ALPHA_AUX_TYPES.contains(&value) { + return Err(Error::Unsupported("alpha auxiliary")); + } + let known = [ + b"urn:com:apple:photo:2020:aux:hdrgainmap".as_slice(), + b"urn:com:apple:photo:2020:aux:semanticskymatte", + b"urn:com:apple:photo:2018:aux:portraiteffectsmatte", + b"urn:com:apple:photo:2019:aux:semanticskinmatte", + b"urn:com:apple:photo:2019:aux:semantichairmatte", + b"urn:com:apple:photo:2019:aux:semanticteethmatte", + b"urn:com:apple:photo:2020:aux:semanticglassesmatte", + b"tag:apple.com,2023:photo:aux:styledeltamap", + b"tag:apple.com,2023:photo:aux:linearthumbnail", + b"urn:mpeg:hevc:2015:auxid:2", + b"urn:iso:std:iso:ts:21496:-1", + ] + .contains(&value); + if essential && !known { + return Err(Error::Unsupported("required auxiliary type")); + } + found = true; + } + if !found { + return Err(Error::Malformed("auxiliary type missing")); + } + } + b"cdsc" => { + let item = index.item(reference.from)?; + if index.items[item].is_exif && reference.targets().any(|id| id == primary_id) { + let len = index.items[item] + .location + .ok_or(Error::Malformed("Exif location"))? + .length; + if len > 1024 * 1024 { + return Err(Error::LimitExceeded("Exif bytes")); + } + let mut bytes = budget.zeroed(len, "Exif")?; + index.read_item(item, source, &mut bytes)?; + let orientation = crate::parse_exif_orientation_from_item_payload(&bytes) + .and_then(|n| u8::try_from(n).ok()) + .filter(|n| (1..=8).contains(n)); + if exif.is_some() && exif != orientation { + return Err(Error::Malformed("conflicting Exif orientation")); + } + exif = orientation; + } + } + b"thmb" => {} + _ => return Err(Error::Unsupported("required item dependency")), + } + } + Ok(exif) +} diff --git a/src/bounded/incremental.rs b/src/bounded/incremental.rs new file mode 100644 index 0000000..b346b06 --- /dev/null +++ b/src/bounded/incremental.rs @@ -0,0 +1,807 @@ +use std::io::{BufReader, Read}; + +use crate::heic_decoder::hevc::DecodedFrame; +use crate::heic_decoder::hevc::bounded::{Geometry, Prepared}; + +use super::container::{Index, Reader, Source}; +use super::grid::{Grid, Properties, Tile}; +use super::memory::Budget; +use super::resample::{Accumulator, Layout, Region, zeroed}; +use super::{ + BoundedDecodeError as Error, BoundedDecodeOptions, BoundedInput, BoundedRgbImage, Result, + codec, color, +}; + +struct Slice { + offset: usize, + length: usize, + prefix: Vec, +} + +fn slice(index: &Index<'_>, source: &mut Source<'_>, item: usize, config: &[u8]) -> Result { + let (_, length_size) = codec::configuration(config)?; + let length = index.items[item] + .location + .ok_or(Error::Malformed("coded item location"))? + .length; + let mut position = 0; + let mut found = None; + let mut count = 0; + while position < length { + count += 1; + if count > 1024 { + return Err(Error::LimitExceeded("NAL count")); + } + let mut bytes = [0; 4]; + index.read_range(item, source, position, &mut bytes[..length_size])?; + let size = bytes[..length_size] + .iter() + .fold(0usize, |n, &b| (n << 8) | usize::from(b)); + position += length_size; + if size < 2 || size > length - position { + return Err(Error::Malformed("NAL length")); + } + index.read_range(item, source, position, &mut bytes[..2])?; + if bytes[0] & 0x81 != 0 || bytes[1] != 1 { + return Err(Error::Unsupported("NAL layer or temporal id")); + } + match bytes[0] >> 1 & 63 { + 19 | 20 => { + if found.is_some() { + return Err(Error::Unsupported("multiple slices")); + } + let mut prefix = zeroed(size.min(65536))?; + index.read_range(item, source, position, &mut prefix)?; + found = Some(Slice { + offset: position + 2, + length: size - 2, + prefix, + }); + } + 35 | 38..=40 => {} + _ => { + return Err(Error::Unsupported( + "experiment requires configuration parameter sets and one IDR", + )); + } + } + position += size; + } + found.ok_or(Error::Malformed("missing IDR")) +} + +fn prepared<'a>(config: &'a [u8], slice: &'a Slice) -> Result> { + let (nals, _) = codec::configuration(config)?; + let parameter = |i: usize| nals.parameters[i].ok_or(Error::Malformed("missing parameter set")); + Ok(Prepared::new( + parameter(0)?, + parameter(1)?, + parameter(2)?, + &slice.prefix, + )?) +} + +struct ItemRange<'a, 'b, 'c> { + index: &'a Index<'b>, + source: &'a mut Source<'c>, + item: usize, + position: usize, + remaining: usize, +} + +impl Read for ItemRange<'_, '_, '_> { + fn read(&mut self, output: &mut [u8]) -> std::io::Result { + let count = output.len().min(self.remaining); + self.index + .read_range(self.item, self.source, self.position, &mut output[..count]) + .map_err(std::io::Error::other)?; + self.position += count; + self.remaining -= count; + Ok(count) + } +} + +struct Rbsp { + input: R, + zeroes: u8, +} + +impl Read for Rbsp { + fn read(&mut self, output: &mut [u8]) -> std::io::Result { + let mut count = 0; + while count < output.len() { + let mut byte = [0]; + if self.input.read(&mut byte)? == 0 { + break; + } + let byte = byte[0]; + if self.zeroes == 2 && byte == 3 { + self.zeroes = 0; + continue; + } + self.zeroes = if byte == 0 { + (self.zeroes + 1).min(2) + } else { + 0 + }; + output[count] = byte; + count += 1; + } + Ok(count) + } +} + +fn plan<'a>(index: &Index<'a>, source: &mut Source<'_>, budget: &Budget) -> Result> { + let primary = index.primary; + let properties = Properties::read_incremental(index, primary, false)?; + let (width, height) = properties + .dimensions + .ok_or(Error::Malformed("primary dimensions"))?; + let is_grid = index.items[primary].kind == *b"grid"; + let (rows, columns) = if is_grid { + let len = index.items[primary] + .location + .ok_or(Error::Malformed("grid location"))? + .length; + if !matches!(len, 8 | 12) { + return Err(Error::Malformed("grid length")); + } + let mut bytes = [0; 12]; + index.read_item(primary, source, &mut bytes[..len])?; + let mut r = Reader::new(&bytes[..len]); + if r.uint(1)? != 0 { + return Err(Error::Unsupported("grid version")); + } + let flags = r.uint(1)?; + if flags > 1 { + return Err(Error::Unsupported("grid flags")); + } + let rows = r.uint(1)? as u32 + 1; + let columns = r.uint(1)? as u32 + 1; + let size = if flags == 0 { 2 } else { 4 }; + if r.uint(size)? != u64::from(width) || r.uint(size)? != u64::from(height) { + return Err(Error::Malformed("grid dimensions")); + } + r.finish()?; + (rows, columns) + } else { + (1, 1) + }; + if width == 0 + || height == 0 + || u64::from(width) * u64::from(height) > 256_000_000 + || rows * columns > 4096 + { + return Err(Error::LimitExceeded("experiment image dimensions")); + } + let tile_count = (rows * columns) as usize; + let reference = if is_grid { + let mut references = index + .references + .iter() + .filter(|r| r.from == index.items[primary].id && r.kind == *b"dimg"); + let reference = references + .next() + .ok_or(Error::Malformed("grid references"))?; + if references.next().is_some() { + return Err(Error::Malformed("duplicate grid references")); + } + if reference.targets().count() != tile_count { + return Err(Error::Malformed("tile count")); + } + Some(reference) + } else { + None + }; + let tile_ids = reference + .into_iter() + .flat_map(|reference| reference.targets()) + .chain((!is_grid).then_some(index.items[primary].id)); + let mut tiles = budget.buffer::>(tile_count, "tile index")?; + for id in tile_ids { + let item = index.item(id)?; + if !matches!(&index.items[item].kind, b"hvc1" | b"hev1") { + return Err(Error::Unsupported("direct HEVC item required")); + } + let props = Properties::read_incremental(index, item, is_grid)?; + let config = props.config.ok_or(Error::Malformed("missing hvcC"))?; + let _headers = budget.reserve(2 * 1024 * 1024, "header parsing")?; + let slice = slice(index, source, item, config)?; + let prepared = prepared(config, &slice)?; + let geometry = prepared.geometry; + let workspace_bytes = prepared.incremental_workspace()?; + if props.dimensions != Some((geometry.width, geometry.height)) { + return Err(Error::Malformed("ispe differs from SPS")); + } + if tiles + .first() + .is_some_and(|first| first.geometry != geometry || first.color != props.color) + { + return Err(Error::Unsupported("nonuniform grid")); + } + tiles.push(Tile { + item, + config, + geometry, + color: props.color, + payload_len: slice.length, + workspace_bytes, + })?; + } + let g = tiles[0].geometry; + if g.width * columns < width + || g.height * rows < height + || g.width * (columns - 1) >= width + || g.height * (rows - 1) >= height + { + return Err(Error::Malformed("tile coverage")); + } + let exif = super::grid::validate_references(index, source, &tiles, budget)?; + Ok(Grid { + width, + height, + columns, + tiles, + properties, + exif, + }) +} + +pub fn decode( + input: BoundedInput<'_>, + options: BoundedDecodeOptions, +) -> Result<(BoundedRgbImage, [u64; 3])> { + if options.max_side == 0 || options.max_side > 6000 || options.max_memory_bytes < 65536 { + return Err(Error::InvalidOptions("experiment dimensions or budget")); + } + if cfg!(feature = "decoder-tracing") { + return Err(Error::Unsupported("decoder tracing")); + } + let budget = Budget::new(options.max_memory_bytes); + let _runtime = budget.reserve(65536, "runtime")?; + let mut source = Source::new(input)?; + let metadata = source.metadata(&budget)?; + let index = Index::parse_primary(&metadata, source.len(), &budget, true)?; + let grid = plan(&index, &mut source, &budget)?; + let layout = Layout::new(&grid, options.max_side)?; + for (i, transform) in grid.properties.transforms.iter().flatten().enumerate() { + if i > 0 + && matches!( + transform, + crate::isobmff::PrimaryItemTransformProperty::CleanAperture(_) + ) + { + return Err(Error::Unsupported( + "repeated crop or crop after orientation", + )); + } + } + let bilinear = layout.left % 2 != 0 || layout.top % 2 != 0; + if bilinear && grid.tiles.len() != 1 { + return Err(Error::Unsupported("odd grid crop chroma interpolation")); + } + let _chroma_workspace = if bilinear { + Some(budget.reserve( + grid.tiles[0].geometry.coded_width as usize * 4 * 65, + "chroma interpolation band", + )?) + } else { + None + }; + let mut color = grid.tiles[0].color.clone(); + if grid + .properties + .color + .nclx + .as_ref() + .is_some_and(|n| !n.is_undefined()) + { + color.nclx = grid.properties.color.nclx.clone(); + } + if grid.properties.color.icc.is_some() { + color.icc = grid.properties.color.icc; + } + let transform = color::ColorTransform::new( + &color, + grid.tiles[0].geometry.primaries, + grid.tiles[0].geometry.transfer, + &budget, + )?; + let mut accumulator = Accumulator::new(layout, &budget)?; + let output_len = layout.display_width as usize * layout.display_height as usize * 3; + let workspace = grid.tiles.iter().map(|t| t.workspace_bytes).max().unwrap(); + let _workspace = budget.reserve(workspace, "rolling reconstruction and conversion")?; + let mut output = budget.zeroed(output_len, "RGB output")?; + let mut statistics = [0; 3]; + for (i, tile) in grid.tiles.iter().enumerate() { + let tx = i as u32 % grid.columns * tile.geometry.width; + let ty = i as u32 / grid.columns * tile.geometry.height; + let region = layout.region(tx, ty, tile.geometry.width, tile.geometry.height); + let mut rows = Rows::new(layout, region)?; + let slice = slice(&index, &mut source, tile.item, tile.config)?; + let prepared = prepared(tile.config, &slice)?; + if prepared.geometry != tile.geometry { + return Err(Error::Malformed("changed coded header")); + } + let range = ItemRange { + index: &index, + source: &mut source, + item: tile.item, + position: slice.offset, + remaining: slice.length, + }; + let mut rbsp = Rbsp { + input: BufReader::with_capacity(16384, range), + zeroes: 0, + }; + let mut conversion = Conversion::default(); + let mut bands = ChromaBands::new(if bilinear { + Some(prepared.output_band()?) + } else { + None + })?; + let mut conversion_geometry = tile.geometry; + if grid.tiles.len() == 1 { + conversion_geometry.width = conversion_geometry.width.min(grid.width); + conversion_geometry.height = conversion_geometry.height.min(grid.height); + } + let mut process = + |start: u32, frame: &DecodedFrame, halo: Option>| -> Result<()> { + let g = conversion_geometry; + let first = start.max(g.crop[2]); + let last = (start + frame.height).min(g.crop[2] + g.height); + if first >= last { + return Ok(()); + } + let pixels = conversion.convert( + frame, + g, + ConversionRegion { + start, + y: first - start, + height: last - first, + halo, + }, + &color, + &transform, + )?; + for row in 0..last - first { + let y = ty + first - g.crop[2] + row; + if y < layout.top + region.top || y >= layout.top + region.bottom { + continue; + } + let offset = row as usize * g.width as usize * 3; + rows.add( + y - layout.top, + tx, + &pixels[offset..offset + g.width as usize * 3], + &mut accumulator, + &mut output, + ); + } + Ok(()) + }; + let stats = if u64::from(tile.geometry.width) * u64::from(tile.geometry.height) + >= 1024 * 1024 + { + let spare = prepared.output_band()?; + std::thread::scope(|scope| -> Result<_> { + let (work_tx, work_rx) = std::sync::mpsc::sync_channel::<(u32, DecodedFrame)>(0); + let (free_tx, free_rx) = std::sync::mpsc::sync_channel(1); + free_tx + .send(spare) + .map_err(|_| Error::Unsupported("band queue initialization"))?; + let worker = std::thread::Builder::new() + .stack_size(512 * 1024) + .spawn_scoped(scope, move || -> Result<()> { + while let Ok((start, mut frame)) = work_rx.recv() { + bands.push(start, &mut frame, &mut process)?; + free_tx + .send(frame) + .map_err(|_| Error::Unsupported("band return queue"))?; + } + bands.finish(&mut process) + }) + .map_err(|_| Error::AllocationFailed)?; + let decoded = prepared.decode_stream(&mut rbsp, |start, frame| { + let mut spare = free_rx.recv().map_err(|_| { + crate::heic_decoder::HevcError::DecodingError("band worker stopped") + })?; + std::mem::swap(frame, &mut spare); + work_tx.send((start, spare)).map_err(|_| { + crate::heic_decoder::HevcError::DecodingError("band worker stopped") + }) + }); + drop(work_tx); + worker + .join() + .map_err(|_| Error::Unsupported("band worker panicked"))??; + Ok(decoded?) + })? + } else { + let mut callback_error = None; + let decoded = prepared.decode_stream(&mut rbsp, |start, frame| { + if let Err(error) = bands.push(start, frame, &mut process) { + callback_error = Some(error); + return Err(crate::heic_decoder::HevcError::DecodingError( + "row sink failed", + )); + } + Ok(()) + }); + if let Some(error) = callback_error { + return Err(error); + } + let stats = decoded?; + bands.finish(&mut process)?; + stats + }; + statistics[0] += u64::from(stats.rows); + statistics[1] += stats.sao_edge_components as u64; + statistics[2] += stats.sao_band_components as u64; + rows.finish(&mut accumulator, &mut output); + if (i as u32 + 1).is_multiple_of(grid.columns) { + accumulator.finish_tile_row(); + } + } + Ok(( + BoundedRgbImage { + image: crate::DecodedRgbImage { + width: layout.display_width, + height: layout.display_height, + pixels: output.into_vec(), + source_bit_depth: 8, + icc_profile: None, + }, + original_dimensions: layout.original_dimensions, + }, + statistics, + )) +} + +struct Rows { + layout: Layout, + region: Region, + y: u32, + sums: Vec<[u64; 3]>, + horizontal: Vec<[u32; 3]>, + spans: Vec, +} + +#[derive(Default, Clone)] +struct Span { + start: u32, + end: u32, + first: u32, + last: u32, +} + +impl Rows { + fn new(layout: Layout, region: Region) -> Result { + let width = if layout.is_unscaled() || region.len() == 0 { + 0 + } else { + (region.output_right - region.output_left) as usize + }; + let mut spans: Vec = zeroed(width)?; + for (i, span) in spans.iter_mut().enumerate() { + let x = u64::from(region.output_left) + i as u64; + let x0 = x * u64::from(layout.width); + let x1 = (x + 1) * u64::from(layout.width); + let start = (x0 / u64::from(layout.output_width)).max(u64::from(region.left)); + let end = x1 + .div_ceil(u64::from(layout.output_width)) + .min(u64::from(region.right)); + let weight = |sx: u64| { + (x1.min((sx + 1) * u64::from(layout.output_width)) + - x0.max(sx * u64::from(layout.output_width))) as u32 + }; + *span = Span { + start: start as u32, + end: end as u32, + first: weight(start), + last: weight(end - 1), + }; + } + Ok(Self { + layout, + region, + y: region.output_top, + sums: zeroed(width)?, + horizontal: zeroed(width)?, + spans, + }) + } + + fn add( + &mut self, + sy: u32, + tx: u32, + pixels: &[u8], + accumulator: &mut Accumulator, + output: &mut [u8], + ) { + let l = self.layout; + if self.region.len() == 0 { + return; + } + if l.is_unscaled() { + let start = (self.region.left + l.left - tx) as usize * 3; + let end = (self.region.right + l.left - tx) as usize * 3; + l.write_row(self.region.left, sy, &pixels[start..end], output); + return; + } + for (sum, span) in self.horizontal.iter_mut().zip(&self.spans) { + let start = (span.start + l.left - tx) as usize * 3; + let end = (span.end + l.left - tx) as usize * 3; + let input = &pixels[start..end]; + for c in 0..3 { + sum[c] = u32::from(input[c]) * span.first; + } + if input.len() > 3 { + for pixel in input[3..input.len() - 3].chunks_exact(3) { + for c in 0..3 { + sum[c] += u32::from(pixel[c]) * l.output_width; + } + } + for c in 0..3 { + sum[c] += u32::from(input[input.len() - 3 + c]) * span.last; + } + } + } + let oy_start = u64::from(sy) * u64::from(l.output_height) / u64::from(l.height); + let oy_end = (u64::from(sy + 1) * u64::from(l.output_height)).div_ceil(u64::from(l.height)); + for oy in oy_start as u32..oy_end as u32 { + if oy != self.y { + accumulator.merge_row(self.region, self.y, &self.sums, output); + self.sums.fill([0; 3]); + self.y = oy; + } + let wy = (u64::from(sy + 1) * u64::from(l.output_height)) + .min(u64::from(oy + 1) * u64::from(l.height)) + - (u64::from(sy) * u64::from(l.output_height)) + .max(u64::from(oy) * u64::from(l.height)); + for (sum, horizontal) in self.sums.iter_mut().zip(&self.horizontal) { + for c in 0..3 { + sum[c] += u64::from(horizontal[c]) * wy; + } + } + } + } + + fn finish(self, accumulator: &mut Accumulator, output: &mut [u8]) { + if self.region.len() != 0 && !self.layout.is_unscaled() { + accumulator.merge_row(self.region, self.y, &self.sums, output); + } + } +} + +#[derive(Default)] +struct Conversion { + rgb: Vec, + converted: Vec, +} + +#[derive(Clone, Copy)] +struct ChromaHalo<'a> { + before: [&'a [u16]; 2], + after: [&'a [u16]; 2], +} + +struct ConversionRegion<'a> { + start: u32, + y: u32, + height: u32, + halo: Option>, +} + +struct ChromaBands { + pending: Option, + start: Option, + before: [Vec; 2], +} + +impl ChromaBands { + fn new(pending: Option) -> Result { + let width = pending.as_ref().map_or(0, DecodedFrame::c_stride); + Ok(Self { + pending, + start: None, + before: [zeroed(width)?, zeroed(width)?], + }) + } + + fn push( + &mut self, + start: u32, + frame: &mut DecodedFrame, + emit: &mut impl FnMut(u32, &DecodedFrame, Option>) -> Result<()>, + ) -> Result<()> { + let Some(pending) = &mut self.pending else { + return emit(start, frame, None); + }; + let width = frame.c_stride(); + if let Some(previous_start) = self.start { + emit( + previous_start, + pending, + Some(ChromaHalo { + before: [&self.before[0], &self.before[1]], + after: [&frame.cb_plane[..width], &frame.cr_plane[..width]], + }), + )?; + let last = (pending.height as usize / 2 - 1) * width; + self.before[0].copy_from_slice(&pending.cb_plane[last..last + width]); + self.before[1].copy_from_slice(&pending.cr_plane[last..last + width]); + } else { + self.before[0].copy_from_slice(&frame.cb_plane[..width]); + self.before[1].copy_from_slice(&frame.cr_plane[..width]); + } + std::mem::swap(pending, frame); + self.start = Some(start); + Ok(()) + } + + fn finish( + &self, + emit: &mut impl FnMut(u32, &DecodedFrame, Option>) -> Result<()>, + ) -> Result<()> { + if let (Some(pending), Some(start)) = (&self.pending, self.start) { + let width = pending.c_stride(); + let last = (pending.height as usize / 2 - 1) * width; + emit( + start, + pending, + Some(ChromaHalo { + before: [&self.before[0], &self.before[1]], + after: [ + &pending.cb_plane[last..last + width], + &pending.cr_plane[last..last + width], + ], + }), + )?; + } + Ok(()) + } +} + +impl Conversion { + fn convert( + &mut self, + frame: &DecodedFrame, + geometry: Geometry, + region: ConversionRegion<'_>, + color: &super::grid::Color<'_>, + transform: &color::ColorTransform, + ) -> Result<&[u8]> { + let ConversionRegion { + start, + y, + height, + halo, + } = region; + let colr = crate::isobmff::PrimaryItemColorProperties { + nclx: color.nclx.clone(), + icc: None, + }; + let range = + crate::ycbcr_range_override_from_primary_colr(&colr).unwrap_or(if frame.full_range { + crate::YCbCrRange::Full + } else { + crate::YCbCrRange::Limited + }); + let matrix = crate::ycbcr_matrix_override_from_primary_colr(&colr).unwrap_or( + crate::YCbCrMatrixCoefficients { + matrix_coefficients: u16::from(frame.matrix_coeffs), + colour_primaries: u16::from(frame.colour_primaries), + }, + ); + let matrix = crate::ycbcr_transform_from_matrix(matrix) + .map_err(|_| Error::Unsupported("row YCbCr matrix"))?; + let converter = crate::PreparedYcbcrToRgb::new(8, range, matrix, true); + let width = geometry.width as usize; + let height = height as usize; + let x = geometry.crop[0] as usize; + let y = y as usize; + let len = width * height * 3; + self.rgb + .try_reserve_exact(len.saturating_sub(self.rgb.len())) + .map_err(|_| Error::AllocationFailed)?; + self.rgb.resize(len, 0); + if let Some(halo) = halo { + for row in 0..height { + for column in 0..width { + let cy = (y + row) / 2; + let sy = start as usize + y + row; + let neighbor_y = if sy.is_multiple_of(2) { + (sy / 2) + .saturating_sub(1) + .max(geometry.crop[2] as usize / 2) + } else { + (sy / 2 + 1) + .min(((geometry.crop[2] + geometry.height) as usize).div_ceil(2) - 1) + }; + let cx = (x + column) / 2; + let neighbor_x = if column % 2 == 0 { + cx.saturating_sub(1).max(x / 2) + } else { + (cx + 1).min((x + width).div_ceil(2) - 1) + }; + let chroma = |plane: &[u16], component: usize| { + let stride = frame.c_stride(); + let current = &plane[cy * stride..(cy + 1) * stride]; + let neighbor = if neighbor_y < start as usize / 2 { + halo.before[component] + } else if neighbor_y >= (start as usize + frame.height as usize) / 2 { + halo.after[component] + } else { + let local = neighbor_y - start as usize / 2; + &plane[local * stride..(local + 1) * stride] + }; + (9 * i32::from(current[cx]) + + 3 * i32::from(current[neighbor_x]) + + 3 * i32::from(neighbor[cx]) + + i32::from(neighbor[neighbor_x]) + + 8) + / 16 + }; + let (r, g, b) = converter.convert( + i32::from(frame.y_plane[(y + row) * frame.y_stride() + x + column]), + chroma(&frame.cb_plane, 0), + chroma(&frame.cr_plane, 1), + ); + self.rgb[(row * width + column) * 3..(row * width + column) * 3 + 3] + .copy_from_slice(&[r as u8, g as u8, b as u8]); + } + } + } else if let crate::PreparedYcbcrTransform::MatrixFull { coeffs, .. } = converter.transform + { + crate::heic_decoder::hevc::color_convert::convert_420_8bit_region_to_interleaved( + &frame.y_plane, + &frame.cb_plane, + &frame.cr_plane, + frame.y_stride(), + frame.c_stride(), + x, + y, + width, + height, + coeffs.r_cr_fp8, + coeffs.g_cb_fp8, + coeffs.g_cr_fp8, + coeffs.b_cb_fp8, + 3, + &mut self.rgb, + ); + } else if let Some(params) = crate::prepared_float_matrix_params(converter.transform) { + crate::heic_decoder::hevc::color_convert::convert_float_matrix_8bit_region_to_interleaved( + &frame.y_plane, &frame.cb_plane, &frame.cr_plane, + frame.y_stride(), frame.c_stride(), 2, 2, x, y, width, height, + params, 3, &mut self.rgb, + ); + } else { + for row in 0..height { + for column in 0..width { + let yi = (y + row) * frame.y_stride() + x + column; + let ci = (y + row) / 2 * frame.c_stride() + (x + column) / 2; + let (r, g, b) = converter.convert( + i32::from(frame.y_plane[yi]), + i32::from(frame.cb_plane[ci]), + i32::from(frame.cr_plane[ci]), + ); + let out = (row * width + column) * 3; + self.rgb[out..out + 3].copy_from_slice(&[r as u8, g as u8, b as u8]); + } + } + } + if transform.is_identity() { + return Ok(&self.rgb); + } + self.converted + .try_reserve_exact(len.saturating_sub(self.converted.len())) + .map_err(|_| Error::AllocationFailed)?; + self.converted.resize(len, 0); + transform.apply(&self.rgb, &mut self.converted)?; + Ok(&self.converted) + } +} diff --git a/src/bounded/resample.rs b/src/bounded/resample.rs index 323cd1b..f0ec35a 100644 --- a/src/bounded/resample.rs +++ b/src/bounded/resample.rs @@ -103,6 +103,19 @@ impl Layout { self.width == self.output_width && self.height == self.output_height } + #[cfg(feature = "incremental-experiment")] + pub(super) fn write_row(&self, x: u32, y: u32, pixels: &[u8], output: &mut [u8]) { + if self.matrix[0] == 1 { + let start = self.pixel_index(x, y); + output[start..start + pixels.len()].copy_from_slice(pixels); + } else { + for (offset, pixel) in pixels.chunks_exact(3).enumerate() { + let start = self.pixel_index(x + offset as u32, y); + output[start..start + 3].copy_from_slice(pixel); + } + } + } + pub(super) fn pixel_index(&self, x: u32, y: u32) -> usize { let [a, b, c, d] = self.matrix; let dx = i64::from(a) * i64::from(x) @@ -327,6 +340,52 @@ impl Accumulator { self.left.fill([0; 3]); } + #[cfg(feature = "incremental-experiment")] + pub(super) fn merge_row( + &mut self, + region: Region, + oy: u32, + sums: &[[u64; 3]], + output: &mut [u8], + ) { + let layout = self.layout; + let denominator = u64::from(layout.width) * u64::from(layout.height); + for ox in region.output_left..region.output_right { + let mut sum = sums[(ox - region.output_left) as usize]; + if u64::from(ox) * u64::from(layout.width) + < u64::from(region.left) * u64::from(layout.output_width) + { + for (value, previous) in sum.iter_mut().zip(self.left[oy as usize]) { + *value += previous; + } + } + if u64::from(ox + 1) * u64::from(layout.width) + > u64::from(region.right) * u64::from(layout.output_width) + { + self.left[oy as usize] = sum; + continue; + } + if u64::from(oy) * u64::from(layout.height) + < u64::from(region.top) * u64::from(layout.output_height) + { + for (value, previous) in sum.iter_mut().zip(self.top[ox as usize]) { + *value += previous; + } + } + if u64::from(oy + 1) * u64::from(layout.height) + <= u64::from(region.bottom) * u64::from(layout.output_height) + { + let index = layout.pixel_index(ox, oy); + for channel in 0..3 { + output[index + channel] = + ((sum[channel] + denominator / 2) / denominator) as u8; + } + } else { + self.bottom[ox as usize] = sum; + } + } + } + pub(super) fn merge(&mut self, region: Region, sums: &Contribution, output: &mut [u8]) { let layout = self.layout; let out_width = (region.output_right - region.output_left) as usize; diff --git a/src/bounded/tests.rs b/src/bounded/tests.rs index a581da8..3843a6a 100644 --- a/src/bounded/tests.rs +++ b/src/bounded/tests.rs @@ -558,6 +558,114 @@ fn display_mapping_matches_existing_transform_plan() { } } +#[cfg(all(feature = "incremental-experiment", not(feature = "decoder-tracing")))] +#[test] +fn incremental_accepts_legacy_pixi_without_relaxing_bit_depth_checks() { + let nals = nal_sets(); + for (channels, expected) in [ + (vec![1, 8], true), + (vec![3, 8, 8, 8], true), + (vec![1, 10], false), + (vec![3, 8, 10, 8], false), + (vec![2, 8, 8], false), + (vec![0], false), + (vec![1, 8, 8], false), + ] { + let bytes = fixture( + &[configuration(&nals[1])], + &nals[3], + &[full_box(b"pixi", 0, &channels)], + ); + let result = incremental::decode(BoundedInput::Bytes(&bytes), Default::default()); + assert_eq!(result.is_ok(), expected, "{channels:?}: {result:?}"); + } +} + +#[cfg(all(feature = "incremental-experiment", not(feature = "decoder-tracing")))] +#[test] +fn incremental_rejects_unimplemented_chroma_transform_orders() { + let nals = nal_sets(); + let config = configuration(&nals[1]); + let crop = [13u32, 1, 11, 1, 0, 1, 0, 1] + .into_iter() + .flat_map(u32::to_be_bytes) + .collect::>(); + for extras in [ + vec![box_bytes(b"irot", &[1]), box_bytes(b"clap", &crop)], + vec![box_bytes(b"clap", &crop), box_bytes(b"clap", &crop)], + ] { + let bytes = fixture(std::slice::from_ref(&config), &nals[3], &extras); + assert!(matches!( + incremental::decode(BoundedInput::Bytes(&bytes), Default::default()), + Err(BoundedDecodeError::Unsupported( + "repeated crop or crop after orientation" + )) + )); + } + let bytes = fixture( + &[config.clone(), config], + &nals[3], + &[box_bytes(b"clap", &crop)], + ); + assert!(matches!( + incremental::decode(BoundedInput::Bytes(&bytes), Default::default()), + Err(BoundedDecodeError::Unsupported( + "odd grid crop chroma interpolation" + )) + )); +} + +#[cfg(feature = "incremental-experiment")] +#[test] +fn shared_reference_validation_preserves_auxiliary_and_exif_rules() { + let original = include_bytes!("testdata/apple-semantic-mattes.heic"); + let hair = b"urn:com:apple:photo:2019:aux:semantichairmatte"; + let offset = original + .windows(hair.len()) + .position(|b| b == hair) + .unwrap(); + for (replacement, expected) in [ + (hair.as_slice(), None), + ( + b"urn:unknown:required".as_slice(), + Some("required auxiliary type"), + ), + ( + b"urn:mpeg:hevc:2015:auxid:1".as_slice(), + Some("alpha auxiliary"), + ), + ] { + let mut bytes = original.to_vec(); + bytes[offset..offset + hair.len()].fill(0); + bytes[offset..offset + replacement.len()].copy_from_slice(replacement); + let budget = memory::Budget::new(128 * 1024 * 1024); + let mut source = container::Source::new(BoundedInput::Bytes(&bytes)).unwrap(); + let metadata = source.metadata(&budget).unwrap(); + let index = + container::Index::parse_primary(&metadata, source.len(), &budget, true).unwrap(); + let result = grid::validate_references(&index, &mut source, &[], &budget); + match expected { + None => assert!(result.is_ok()), + Some(expected) => assert!( + matches!(result, Err(BoundedDecodeError::Unsupported(message)) if message == expected) + ), + } + } + let nals = nal_sets(); + for orientation in 1..=8 { + let bytes = fixture_with_exif(&[configuration(&nals[1])], &nals[3], &[], Some(orientation)); + let budget = memory::Budget::new(128 * 1024 * 1024); + let mut source = container::Source::new(BoundedInput::Bytes(&bytes)).unwrap(); + let metadata = source.metadata(&budget).unwrap(); + let index = + container::Index::parse_primary(&metadata, source.len(), &budget, true).unwrap(); + assert_eq!( + grid::validate_references(&index, &mut source, &[], &budget).unwrap(), + Some(orientation) + ); + } +} + #[test] fn exif_orientation_applies_only_without_container_orientation() { let nals = nal_sets(); @@ -702,6 +810,13 @@ fn mirrored_asymmetric_pixels_match_normal_decode() { .any(|p| p != &normal.pixels[..3]) ); let bounded = decode_bounded(BoundedInput::Bytes(&bytes), Default::default()).unwrap(); + #[cfg(feature = "incremental-experiment")] + { + let (incremental, _) = + incremental::decode(BoundedInput::Bytes(&bytes), Default::default()).unwrap(); + assert_eq!(incremental.original_dimensions, bounded.original_dimensions); + assert_eq!(incremental.image.pixels, bounded.image.pixels); + } assert_eq!( (bounded.image.width, bounded.image.height), (normal.width, normal.height) @@ -741,6 +856,13 @@ fn public_decode_applies_associated_exif_after_identity_rotation() { ) .unwrap(); let bounded = decode_bounded(BoundedInput::Bytes(&bytes), Default::default()).unwrap(); + #[cfg(feature = "incremental-experiment")] + { + let (incremental, _) = + incremental::decode(BoundedInput::Bytes(&bytes), Default::default()).unwrap(); + assert_eq!(incremental.original_dimensions, bounded.original_dimensions); + assert_eq!(incremental.image.pixels, bounded.image.pixels); + } assert_eq!( bounded.original_dimensions, (plan.destination_width, plan.destination_height) diff --git a/src/heic-decoder/hevc/bounded.rs b/src/heic-decoder/hevc/bounded.rs index b5c5b80..7b311c1 100644 --- a/src/heic-decoder/hevc/bounded.rs +++ b/src/heic-decoder/hevc/bounded.rs @@ -120,6 +120,58 @@ impl<'a> Prepared<'a> { }) } + pub(crate) fn incremental_workspace(&self) -> Result { + if self.pps.entropy_coding_sync_enabled_flag || self.pps.tiles_enabled_flag { + return Err(HevcError::Unsupported("incremental coding structure")); + } + Ok( + self.geometry.coded_width as usize * self.sps.ctb_size() as usize * 40 + + 2 * 1024 * 1024 + + 512 * 1024, + ) + } + + pub(crate) fn decode_incremental( + self, + mut emit: impl FnMut(u32, &DecodedFrame) -> Result<()>, + ) -> Result { + let parsed = SliceHeader::parse_bounded(&self.slice, &self.sps, &self.pps)?; + super::incremental::decode( + &self.sps, + &self.pps, + &parsed.header, + &mut std::io::Cursor::new(&self.slice.payload[parsed.data_offset..]), + |start, frame| emit(start, frame), + ) + } + + pub(crate) fn output_band(&self) -> Result { + let g = self.geometry; + let mut frame = DecodedFrame::try_with_params(g.coded_width, self.sps.ctb_size(), 8, 1)?; + frame.full_range = g.full_range; + frame.matrix_coeffs = g.matrix; + frame.colour_primaries = g.primaries; + Ok(frame) + } + + pub(crate) fn decode_stream( + self, + reader: &mut dyn std::io::Read, + emit: impl FnMut(u32, &mut DecodedFrame) -> Result<()>, + ) -> Result { + let parsed = SliceHeader::parse_bounded(&self.slice, &self.sps, &self.pps)?; + let mut skip = [0; 512]; + let mut remaining = parsed.data_offset; + while remaining != 0 { + let count = remaining.min(skip.len()); + reader + .read_exact(&mut skip[..count]) + .map_err(|_| HevcError::InvalidBitstream("truncated slice header"))?; + remaining -= count; + } + super::incremental::decode(&self.sps, &self.pps, &parsed.header, reader, emit) + } + pub(crate) fn decode(self) -> Result { let g = self.geometry; let mut frame = DecodedFrame::try_with_params(g.coded_width, g.coded_height, 8, 1)?; diff --git a/src/heic-decoder/hevc/cabac.rs b/src/heic-decoder/hevc/cabac.rs index cd39403..cdc7dc4 100644 --- a/src/heic-decoder/hevc/cabac.rs +++ b/src/heic-decoder/hevc/cabac.rs @@ -178,6 +178,7 @@ impl ContextModel { pub struct CabacDecoder<'a> { /// Input data data: &'a [u8], + stream: Option>, /// Current byte position byte_pos: usize, /// Range register (9 bits, 256-510) @@ -190,6 +191,60 @@ pub struct CabacDecoder<'a> { #[allow(dead_code)] impl<'a> CabacDecoder<'a> { + pub(crate) fn new_stream(reader: &'a mut dyn std::io::Read) -> Result { + let mut decoder = Self::new(&[0, 0])?; + decoder.stream = Some(StreamInput { + reader, + buffer: super::allocation::filled(0, 16384)?, + position: 0, + length: 0, + failed: false, + }); + decoder.byte_pos = 0; + decoder.reinit(); + decoder.check_input()?; + if decoder.byte_pos < 2 { + return Err(HevcError::CabacError("data too short")); + } + Ok(decoder) + } + + pub(crate) fn check_input(&self) -> Result<()> { + if self.stream.as_ref().is_some_and(|input| input.failed) { + Err(HevcError::DecodingError("stream input failed")) + } else { + Ok(()) + } + } + + #[inline] + fn next_byte(&mut self) -> Option { + let byte = if let Some(input) = &mut self.stream { + if input.position == input.length { + match input.reader.read(&mut input.buffer) { + Ok(length) => { + input.position = 0; + input.length = length; + } + Err(_) => { + input.failed = true; + return None; + } + } + } + if input.position == input.length { + return None; + } + let byte = input.buffer[input.position]; + input.position += 1; + byte + } else { + *self.data.get(self.byte_pos)? + }; + self.byte_pos += 1; + Some(byte) + } + /// Get current CABAC state (range, offset) for debugging /// Note: returns (range, value >> 7) for compatibility with old debugging pub fn get_state(&self) -> (u16, u16) { @@ -214,6 +269,7 @@ impl<'a> CabacDecoder<'a> { let mut decoder = Self { data, + stream: None, byte_pos: 0, range: 510, value: 0, @@ -222,15 +278,13 @@ impl<'a> CabacDecoder<'a> { // Initialize value (matching libde265 exactly) decoder.bits_needed = -8; - if decoder.byte_pos < decoder.data.len() { - decoder.value = decoder.data[decoder.byte_pos] as u32; - decoder.byte_pos += 1; + if let Some(byte) = decoder.next_byte() { + decoder.value = u32::from(byte); } decoder.value <<= 8; decoder.bits_needed = 0; - if decoder.byte_pos < decoder.data.len() { - decoder.value |= decoder.data[decoder.byte_pos] as u32; - decoder.byte_pos += 1; + if let Some(byte) = decoder.next_byte() { + decoder.value |= u32::from(byte); decoder.bits_needed = -8; } @@ -250,20 +304,20 @@ impl<'a> CabacDecoder<'a> { self.bits_needed = -9; self.value = 0; - let remaining = self.data.len() - self.byte_pos; - if remaining > 0 { - self.value = (self.data[self.byte_pos] as u32) << 8; - self.byte_pos += 1; + if let Some(byte) = self.next_byte() { + self.value = u32::from(byte) << 8; } - if remaining > 1 { - self.value |= self.data[self.byte_pos] as u32; - self.byte_pos += 1; + if let Some(byte) = self.next_byte() { + self.value |= u32::from(byte); self.bits_needed = -8; } } /// Reinitialize CABAC at an absolute byte offset within this slice data. pub fn seek_to(&mut self, byte_pos: usize) -> Result<()> { + if self.stream.is_some() { + return Err(HevcError::Unsupported("streaming CABAC seek")); + } if byte_pos > self.data.len() { return Err(HevcError::CabacError( "entry point offset beyond slice data", @@ -314,10 +368,9 @@ impl<'a> CabacDecoder<'a> { self.bits_needed += 1; if self.bits_needed >= 0 { - if self.byte_pos < self.data.len() { + if let Some(byte) = self.next_byte() { self.bits_needed = -8; - self.value |= self.data[self.byte_pos] as u32; - self.byte_pos += 1; + self.value |= u32::from(byte); } else { self.bits_needed = -8; } @@ -389,9 +442,8 @@ impl<'a> CabacDecoder<'a> { self.value <<= shift; self.bits_needed += shift as i32; if self.bits_needed >= 0 { - if self.byte_pos < self.data.len() { - self.value |= (self.data[self.byte_pos] as u32) << self.bits_needed; - self.byte_pos += 1; + if let Some(byte) = self.next_byte() { + self.value |= u32::from(byte) << self.bits_needed; } self.bits_needed -= 8; } @@ -533,6 +585,14 @@ pub static INIT_VALUES: [u8; context::NUM_CONTEXTS] = [ 154, 154, ]; +struct StreamInput<'a> { + reader: &'a mut dyn std::io::Read, + buffer: alloc::vec::Vec, + position: usize, + length: usize, + failed: bool, +} + #[cfg(test)] mod tests { use super::CabacDecoder; diff --git a/src/heic-decoder/hevc/ctu.rs b/src/heic-decoder/hevc/ctu.rs index 40c8d26..ee969f2 100644 --- a/src/heic-decoder/hevc/ctu.rs +++ b/src/heic-decoder/hevc/ctu.rs @@ -159,6 +159,7 @@ pub struct SliceContext<'a> { touched_coeffs: [u16; 1024], /// Reusable scaling matrix buffer scaling_buf: [u8; 1024], + map_origin: u32, } impl<'a> SliceContext<'a> { @@ -169,11 +170,40 @@ impl<'a> SliceContext<'a> { header: &'a SliceHeader, slice_data: &'a [u8], ) -> Result { - // DEBUG: Print first few bytes of slice data debug_trace!( "DEBUG: Slice data first 16 bytes: {:02x?}", &slice_data[..16.min(slice_data.len())] ); + Self::new_with_height(sps, pps, header, slice_data, sps.pic_height_in_luma_samples) + } + + pub(crate) fn new_with_height( + sps: &'a Sps, + pps: &'a Pps, + header: &'a SliceHeader, + slice_data: &'a [u8], + height: u32, + ) -> Result { + Self::with_cabac(sps, pps, header, CabacDecoder::new(slice_data)?, height) + } + + pub(crate) fn new_stream( + sps: &'a Sps, + pps: &'a Pps, + header: &'a SliceHeader, + input: &'a mut dyn std::io::Read, + height: u32, + ) -> Result { + Self::with_cabac(sps, pps, header, CabacDecoder::new_stream(input)?, height) + } + + fn with_cabac( + sps: &'a Sps, + pps: &'a Pps, + header: &'a SliceHeader, + cabac: CabacDecoder<'a>, + height: u32, + ) -> Result { debug_trace!( "DEBUG: SPS: {}x{}, ctb_size={}, min_cb_size={}, scaling_list={}", sps.pic_width_in_luma_samples, @@ -192,7 +222,6 @@ impl<'a> SliceContext<'a> { sps.log2_max_tb_size() ); - let cabac = CabacDecoder::new(slice_data)?; let (range, offset) = cabac.get_state(); debug_trace!( "DEBUG: CABAC init state: range={}, offset={}", @@ -244,7 +273,7 @@ impl<'a> SliceContext<'a> { // Map is in units of min_cb_size (typically 8x8) let min_cb_size = 1u32 << sps.log2_min_cb_size(); let ct_depth_map_stride = sps.pic_width_in_luma_samples.div_ceil(min_cb_size); - let ct_depth_map_height = sps.pic_height_in_luma_samples.div_ceil(min_cb_size); + let ct_depth_map_height = height.div_ceil(min_cb_size); let ct_map_size = (ct_depth_map_stride * ct_depth_map_height) as usize; let ct_depth_map = super::allocation::filled(0xFF, ct_map_size)?; @@ -252,7 +281,7 @@ impl<'a> SliceContext<'a> { // This supports NxN partition PU-level resolution let min_pu_size = (min_cb_size / 2).max(1); let intra_mode_map_stride = sps.pic_width_in_luma_samples.div_ceil(min_pu_size); - let intra_mode_map_height = sps.pic_height_in_luma_samples.div_ceil(min_pu_size); + let intra_mode_map_height = height.div_ceil(min_pu_size); let pu_map_size = (intra_mode_map_stride * intra_mode_map_height) as usize; let intra_mode_map = super::allocation::filled(IntraPredMode::Dc.as_u8(), pu_map_size)?; let intra_chroma_mode_map = @@ -261,7 +290,7 @@ impl<'a> SliceContext<'a> { // QP map at min_tb_size granularity let min_tb_size = 1u32 << sps.log2_min_tb_size(); let qp_map_stride = sps.pic_width_in_luma_samples.div_ceil(min_tb_size); - let qp_map_height = sps.pic_height_in_luma_samples.div_ceil(min_tb_size); + let qp_map_height = height.div_ceil(min_tb_size); let qp_map = super::allocation::filled(slice_qp as i8, (qp_map_stride * qp_map_height) as usize)?; @@ -295,11 +324,19 @@ impl<'a> SliceContext<'a> { last_qpy_in_prev_qg: slice_qp, current_qg_x: -1, current_qg_y: -1, - sao_map: SaoMap::new(sps.pic_width_in_ctbs(), sps.pic_height_in_ctbs())?, + sao_map: SaoMap::new( + sps.pic_width_in_ctbs(), + if height < sps.pic_height_in_luma_samples { + 3 + } else { + sps.pic_height_in_ctbs() + }, + )?, residual_buf: [0i16; 1024], coeff_buf: [0i16; 1024], touched_coeffs: [0u16; 1024], scaling_buf: [16u8; 1024], + map_origin: 0, }) } @@ -422,6 +459,51 @@ impl<'a> SliceContext<'a> { Ok(()) } + pub(crate) fn decode_row(&mut self, row: u32, frame: &mut DecodedFrame) -> Result<()> { + let size = self.sps.ctb_size(); + let origin = row.saturating_sub(1) * size; + let delta = origin - self.map_origin; + let cb = 1 << self.sps.log2_min_cb_size(); + let pu = cb / 2; + let tb = 1 << self.sps.log2_min_tb_size(); + super::picture::shift_rows( + &mut self.ct_depth_map, + (delta / cb * self.ct_depth_map_stride) as usize, + 0xFF, + ); + super::picture::shift_rows( + &mut self.intra_mode_map, + (delta / pu * self.intra_mode_map_stride) as usize, + IntraPredMode::Dc.as_u8(), + ); + super::picture::shift_rows( + &mut self.intra_chroma_mode_map, + (delta / pu * self.intra_mode_map_stride) as usize, + IntraPredMode::Dc.as_u8(), + ); + super::picture::shift_rows( + &mut self.qp_map, + (delta / tb * self.qp_map_stride) as usize, + self.header.slice_qp_y as i8, + ); + self.map_origin = origin; + frame.advance_rows(origin); + self.ctb_y = row; + for x in 0..self.sps.pic_width_in_ctbs() { + self.ctb_x = x; + *self.sao_map.get_mut(x, row) = super::sao::SaoInfo::default(); + self.decode_ctu(x * size, row * size, frame)?; + let end = self.cabac.decode_terminate() != 0; + self.cabac.check_input()?; + let last = + row + 1 == self.sps.pic_height_in_ctbs() && x + 1 == self.sps.pic_width_in_ctbs(); + if end != last { + return Err(HevcError::InvalidBitstream("incremental slice termination")); + } + } + Ok(()) + } + /// Decode a single CTU (Coding Tree Unit) fn decode_ctu(&mut self, x_ctb: u32, y_ctb: u32, frame: &mut DecodedFrame) -> Result<()> { let log2_ctb_size = self.sps.log2_ctb_size(); @@ -705,6 +787,9 @@ impl<'a> SliceContext<'a> { fn get_ct_depth(&self, x: u32, y: u32) -> u8 { let min_cb_size = 1u32 << self.sps.log2_min_cb_size(); let map_x = x / min_cb_size; + let Some(y) = y.checked_sub(self.map_origin) else { + return 0xFF; + }; let map_y = y / min_cb_size; if map_x >= self.ct_depth_map_stride @@ -723,7 +808,7 @@ impl<'a> SliceContext<'a> { // Fill the ct_depth_map for this CU region let start_x = x0 / min_cb_size; - let start_y = y0 / min_cb_size; + let start_y = (y0 - self.map_origin) / min_cb_size; let num_blocks = cb_size / min_cb_size; for dy in 0..num_blocks { @@ -1628,6 +1713,7 @@ impl<'a> SliceContext<'a> { let size = 1usize << log2_size; let residual = &self.residual_buf; let max_val = (1i32 << bit_depth) - 1; + let y0 = y0 - frame.plane_origin(c_idx); let (plane, stride) = frame.plane_mut(c_idx); let last_row_end = (y0 as usize + size - 1) * stride + x0 as usize + size; if last_row_end <= plane.len() { @@ -1838,7 +1924,7 @@ impl<'a> SliceContext<'a> { let stride = self.intra_mode_map_stride; let count = ((1u32 << log2_size) / min_pu).max(1); let start_x = x0 / min_pu; - let start_y = y0 / min_pu; + let start_y = (y0 - self.map_origin) / min_pu; for dy in 0..count { for dx in 0..count { let idx = ((start_y + dy) * stride + (start_x + dx)) as usize; @@ -1855,7 +1941,7 @@ impl<'a> SliceContext<'a> { let stride = self.intra_mode_map_stride; let count = ((1u32 << log2_size) / min_pu).max(1); let start_x = x0 / min_pu; - let start_y = y0 / min_pu; + let start_y = (y0 - self.map_origin) / min_pu; for dy in 0..count { for dx in 0..count { let idx = ((start_y + dy) * stride + (start_x + dx)) as usize; @@ -1870,6 +1956,9 @@ impl<'a> SliceContext<'a> { fn get_intra_mode_at(&self, x: u32, y: u32) -> IntraPredMode { let min_pu = self.min_pu_size(); let stride = self.intra_mode_map_stride; + let Some(y) = y.checked_sub(self.map_origin) else { + return IntraPredMode::Dc; + }; let idx = ((y / min_pu) * stride + (x / min_pu)) as usize; if idx < self.intra_mode_map.len() { IntraPredMode::from_u8(self.intra_mode_map[idx]).unwrap_or(IntraPredMode::Dc) @@ -1883,6 +1972,9 @@ impl<'a> SliceContext<'a> { fn get_intra_chroma_mode_at(&self, x: u32, y: u32) -> IntraPredMode { let min_pu = self.min_pu_size(); let stride = self.intra_mode_map_stride; + let Some(y) = y.checked_sub(self.map_origin) else { + return IntraPredMode::Dc; + }; let idx = ((y / min_pu) * stride + (x / min_pu)) as usize; if idx < self.intra_chroma_mode_map.len() { IntraPredMode::from_u8(self.intra_chroma_mode_map[idx]).unwrap_or(IntraPredMode::Dc) @@ -1976,6 +2068,9 @@ impl<'a> SliceContext<'a> { /// Get QPY at a sample position from the QP map fn get_qpy_at(&self, x: u32, y: u32) -> i32 { let min_tb = 1u32 << self.sps.log2_min_tb_size(); + let Some(y) = y.checked_sub(self.map_origin) else { + return self.header.slice_qp_y; + }; let idx = ((y / min_tb) * self.qp_map_stride + (x / min_tb)) as usize; if idx < self.qp_map.len() { self.qp_map[idx] as i32 @@ -1989,7 +2084,7 @@ impl<'a> SliceContext<'a> { let min_tb = 1u32 << self.sps.log2_min_tb_size(); let count = ((1u32 << log2_cb_size) / min_tb).max(1); let start_x = x0 / min_tb; - let start_y = y0 / min_tb; + let start_y = (y0 - self.map_origin) / min_tb; for dy in 0..count { for dx in 0..count { let idx = ((start_y + dy) * self.qp_map_stride + (start_x + dx)) as usize; diff --git a/src/heic-decoder/hevc/deblock.rs b/src/heic-decoder/hevc/deblock.rs index 54c0464..38da6ca 100644 --- a/src/heic-decoder/hevc/deblock.rs +++ b/src/heic-decoder/hevc/deblock.rs @@ -53,18 +53,38 @@ pub fn apply_deblocking_filter( tc_offset: i32, cb_qp_offset: i32, cr_qp_offset: i32, +) { + apply_deblocking_rows( + frame, + beta_offset, + tc_offset, + cb_qp_offset, + cr_qp_offset, + 0, + frame.height, + ); +} + +pub(crate) fn apply_deblocking_rows( + frame: &mut DecodedFrame, + beta_offset: i32, + tc_offset: i32, + cb_qp_offset: i32, + cr_qp_offset: i32, + start: u32, + end: u32, ) { let width = frame.width; - let height = frame.height; + let height = end; // Pass 1: Vertical edges // Process at 8-sample intervals in x, 4-sample intervals in y let mut x = 8u32; while x < width { - let mut y = 0u32; + let mut y = start; while y < height { let bx = x / 4; - let by = y / 4; + let by = (y - frame.row_origin) / 4; let idx = (by * frame.deblock_stride + bx) as usize; if idx < frame.deblock_flags.len() && (frame.deblock_flags[idx] & DEBLOCK_FLAG_VERT) != 0 @@ -86,12 +106,12 @@ pub fn apply_deblocking_filter( // Pass 2: Horizontal edges // Process at 4-sample intervals in x, 8-sample intervals in y - let mut y = 8u32; + let mut y = start.max(8); while y < height { let mut x = 0u32; while x < width { let bx = x / 4; - let by = y / 4; + let by = (y - frame.row_origin) / 4; let idx = (by * frame.deblock_stride + bx) as usize; if idx < frame.deblock_flags.len() && (frame.deblock_flags[idx] & DEBLOCK_FLAG_HORIZ) != 0 @@ -112,7 +132,7 @@ pub fn apply_deblocking_filter( // Chroma deblocking (only for bS=2, which is all edges for I-slices) if frame.chroma_format > 0 { - apply_chroma_deblocking(frame, tc_offset, cb_qp_offset, cr_qp_offset); + apply_chroma_deblocking(frame, tc_offset, cb_qp_offset, cr_qp_offset, start, end); } } @@ -168,6 +188,7 @@ fn filter_edge_luma( } let stride = frame.y_stride(); + let y = y - frame.row_origin; let plane = &mut frame.y_plane; // Compute stride-based addressing: @@ -330,9 +351,11 @@ fn apply_chroma_deblocking( tc_offset: i32, cb_qp_offset: i32, cr_qp_offset: i32, + start: u32, + end: u32, ) { let width = frame.width; - let height = frame.height; + let height = end; let bit_depth_c = frame.bit_depth as i32; // Same as luma for typical HEIC let max_val = (1i32 << bit_depth_c) - 1; @@ -345,7 +368,7 @@ fn apply_chroma_deblocking( }; let c_stride = frame.c_stride(); - let c_height = height / sub_y; + let c_height = (height - frame.row_origin) / sub_y; let c_width = width / sub_x; // For 4:2:0: chroma edges are at 8-chroma-pixel intervals (16 luma pixels). @@ -362,10 +385,10 @@ fn apply_chroma_deblocking( // Pass 1: Vertical edges let mut x = x_step_vert; while x < width { - let mut y = 0u32; + let mut y = start; while y < height { let bx = x / 4; - let by = y / 4; + let by = (y - frame.row_origin) / 4; let idx = (by * frame.deblock_stride + bx) as usize; if idx < frame.deblock_flags.len() && (frame.deblock_flags[idx] & DEBLOCK_FLAG_VERT) != 0 @@ -378,13 +401,13 @@ fn apply_chroma_deblocking( }; let cx = x / sub_x; - let cy = y / sub_y; + let cy = (y - frame.row_origin) / sub_y; // Per-sample transquant-bypass exemption (H.265 8.7.2.5.7) let mut p_bypass = [false; 4]; let mut q_bypass = [false; 4]; for k in 0..4u32 { - let ly = (cy + k) * sub_y; + let ly = (cy + k) * sub_y + frame.row_origin; p_bypass[k as usize] = frame.is_block_bypass(x.wrapping_sub(1), ly); q_bypass[k as usize] = frame.is_block_bypass(x, ly); } @@ -449,12 +472,12 @@ fn apply_chroma_deblocking( } // Pass 2: Horizontal edges - let mut y = y_step_horiz; + let mut y = start.max(y_step_horiz); while y < height { let mut x = 0u32; while x < width { let bx = x / 4; - let by = y / 4; + let by = (y - frame.row_origin) / 4; let idx = (by * frame.deblock_stride + bx) as usize; if idx < frame.deblock_flags.len() && (frame.deblock_flags[idx] & DEBLOCK_FLAG_HORIZ) != 0 @@ -467,7 +490,7 @@ fn apply_chroma_deblocking( }; let cx = x / sub_x; - let cy = y / sub_y; + let cy = (y - frame.row_origin) / sub_y; // Per-sample transquant-bypass exemption (H.265 8.7.2.5.7) let mut p_bypass = [false; 4]; diff --git a/src/heic-decoder/hevc/incremental.rs b/src/heic-decoder/hevc/incremental.rs new file mode 100644 index 0000000..272d380 --- /dev/null +++ b/src/heic-decoder/hevc/incremental.rs @@ -0,0 +1,103 @@ +use super::ctu::SliceContext; +use super::params::{Pps, Sps}; +use super::picture::DecodedFrame; +use super::sao::{SaoMap, write_filtered_rows}; +use super::slice::SliceHeader; +use super::{HevcError, Result, deblock}; + +#[derive(Debug, Default)] +pub(crate) struct Statistics { + pub(crate) rows: u32, + pub(crate) sao_edge_components: usize, + pub(crate) sao_band_components: usize, +} + +pub(crate) fn decode( + sps: &Sps, + pps: &Pps, + header: &SliceHeader, + data: &mut dyn std::io::Read, + mut emit: impl FnMut(u32, &mut DecodedFrame) -> Result<()>, +) -> Result { + if pps.entropy_coding_sync_enabled_flag + || pps.tiles_enabled_flag + || !header.first_slice_segment_in_pic_flag + { + return Err(HevcError::Unsupported("incremental coding structure")); + } + let size = sps.ctb_size(); + let width = sps.pic_width_in_luma_samples; + let height = sps.pic_height_in_luma_samples; + let mut raw = DecodedFrame::try_with_params(width, 2 * size, 8, 1)?; + let mut filtered = DecodedFrame::try_with_params(width, 3 * size, 8, 1)?; + let mut output = DecodedFrame::try_with_params(width, size, 8, 1)?; + raw.height = height; + filtered.height = height; + output.full_range = sps.video_full_range_flag; + output.matrix_coeffs = sps.matrix_coeffs; + output.colour_primaries = sps.colour_primaries; + let mut context = SliceContext::new_stream(sps, pps, header, data, 2 * size)?; + let mut statistics = Statistics::default(); + for row in 0..sps.pic_height_in_ctbs() { + let start = row * size; + let rows = size.min(height - start); + context.decode_row(row, &mut raw)?; + filtered.advance_rows(row.saturating_sub(2) * size); + filtered.copy_rows_from(&raw, start, rows); + if !header.slice_deblocking_filter_disabled_flag { + deblock::apply_deblocking_rows( + &mut filtered, + i32::from(header.slice_beta_offset_div2) * 2, + i32::from(header.slice_tc_offset_div2) * 2, + i32::from(pps.pps_cb_qp_offset), + i32::from(pps.pps_cr_qp_offset), + start, + start + rows, + ); + } + for x in 0..sps.pic_width_in_ctbs() { + let info = context.sao_map.get(x, row); + statistics.sao_edge_components += info.sao_type_idx.iter().filter(|&&n| n == 2).count(); + statistics.sao_band_components += info.sao_type_idx.iter().filter(|&&n| n == 1).count(); + } + if row > 0 { + emit_rows( + &filtered, + &context.sao_map, + size, + start - size, + size, + &mut output, + &mut emit, + )?; + statistics.rows += size; + } + if start + rows == height { + emit_rows( + &filtered, + &context.sao_map, + size, + start, + rows, + &mut output, + &mut emit, + )?; + statistics.rows += rows; + } + } + Ok(statistics) +} + +fn emit_rows( + frame: &DecodedFrame, + map: &SaoMap, + size: u32, + start: u32, + rows: u32, + output: &mut DecodedFrame, + emit: &mut impl FnMut(u32, &mut DecodedFrame) -> Result<()>, +) -> Result<()> { + output.height = rows; + write_filtered_rows(frame, map, size, start, rows, output); + emit(start, output) +} diff --git a/src/heic-decoder/hevc/intra.rs b/src/heic-decoder/hevc/intra.rs index 8e66c29..af5dc6b 100644 --- a/src/heic-decoder/hevc/intra.rs +++ b/src/heic-decoder/hevc/intra.rs @@ -77,6 +77,7 @@ pub fn predict_intra( } // Resolve plane once to avoid per-pixel match on c_idx + let y = y - frame.plane_origin(c_idx); let (plane, stride) = frame.plane_mut(c_idx); let max_val = (1i32 << bit_depth) - 1; @@ -259,6 +260,9 @@ fn fill_border_samples( } }; + let origin = frame.plane_origin(c_idx); + let y = y - origin; + let frame_h = frame_h - origin; let avail_left = x > 0; let avail_top = y > 0; let avail_top_left = avail_left && avail_top; diff --git a/src/heic-decoder/hevc/mod.rs b/src/heic-decoder/hevc/mod.rs index 062d0d6..196d413 100644 --- a/src/heic-decoder/hevc/mod.rs +++ b/src/heic-decoder/hevc/mod.rs @@ -11,6 +11,7 @@ pub(crate) mod color_convert; mod ctu; mod deblock; pub(crate) mod debug; +pub(crate) mod incremental; mod intra; pub(crate) mod params; mod picture; diff --git a/src/heic-decoder/hevc/picture.rs b/src/heic-decoder/hevc/picture.rs index 93ac0cc..b8bf7d8 100644 --- a/src/heic-decoder/hevc/picture.rs +++ b/src/heic-decoder/hevc/picture.rs @@ -64,9 +64,71 @@ pub struct DecodedFrame { /// QP map at 4x4 block granularity (for deblocking) #[doc(hidden)] pub qp_map: Vec, + pub(crate) row_origin: u32, } impl DecodedFrame { + pub(crate) fn plane_origin(&self, c_idx: u8) -> u32 { + if c_idx != 0 && self.chroma_format == 1 { + self.row_origin / 2 + } else { + self.row_origin + } + } + + pub(crate) fn advance_rows(&mut self, origin: u32) { + let rows = (origin - self.row_origin) as usize; + let width = self.width as usize; + let chroma_width = self.c_stride(); + let chroma_rows = if self.chroma_format == 1 { + rows / 2 + } else { + rows + }; + shift_rows(&mut self.y_plane, rows * width, UNINIT_SAMPLE); + shift_rows( + &mut self.cb_plane, + chroma_rows * chroma_width, + UNINIT_SAMPLE, + ); + shift_rows( + &mut self.cr_plane, + chroma_rows * chroma_width, + UNINIT_SAMPLE, + ); + shift_rows( + &mut self.deblock_flags, + rows / 4 * self.deblock_stride as usize, + 0, + ); + shift_rows(&mut self.qp_map, rows / 4 * self.deblock_stride as usize, 0); + self.row_origin = origin; + } + + pub(crate) fn copy_rows_from(&mut self, source: &Self, start: u32, rows: u32) { + for component in 0..3 { + let subsample = if component != 0 && self.chroma_format == 1 { + 2 + } else { + 1 + }; + let target_y = (start - self.row_origin) / subsample; + let source_y = (start - source.row_origin) / subsample; + let (input, stride) = source.plane(component); + let (output, _) = self.plane_mut(component); + let len = (rows / subsample) as usize * stride; + output[target_y as usize * stride..][..len] + .copy_from_slice(&input[source_y as usize * stride..][..len]); + } + let stride = self.deblock_stride as usize; + let target = ((start - self.row_origin) / 4) as usize * stride; + let source_start = ((start - source.row_origin) / 4) as usize * stride; + let len = (rows / 4) as usize * stride; + self.deblock_flags[target..][..len] + .copy_from_slice(&source.deblock_flags[source_start..][..len]); + self.qp_map[target..][..len].copy_from_slice(&source.qp_map[source_start..][..len]); + } + /// Create a frame with specific parameters /// /// # Panics @@ -102,6 +164,7 @@ impl DecodedFrame { Ok(Self { width, + row_origin: 0, height, y_plane: super::allocation::filled(UNINIT_SAMPLE, luma_size)?, cb_plane: super::allocation::filled(UNINIT_SAMPLE, chroma_size)?, @@ -125,7 +188,7 @@ impl DecodedFrame { /// Mark a vertical TU/CU boundary at luma position (x, y) with given size pub(crate) fn mark_tu_boundary(&mut self, x: u32, y: u32, size: u32) { let bx = x / 4; - let by = y / 4; + let by = (y - self.row_origin) / 4; let bs = size / 4; // Mark vertical edge at x (left edge of TU) @@ -152,7 +215,7 @@ impl DecodedFrame { /// Store QP for a block region at 4x4 granularity pub(crate) fn store_block_qp(&mut self, x: u32, y: u32, size: u32, qp: i8) { let bx = x / 4; - let by = y / 4; + let by = (y - self.row_origin) / 4; let bs = size / 4; for j in 0..bs { for i in 0..bs { @@ -167,7 +230,7 @@ impl DecodedFrame { /// Mark a CU region as cu_transquant_bypass at 4x4 granularity pub(crate) fn store_block_bypass(&mut self, x: u32, y: u32, size: u32) { let bx = x / 4; - let by = y / 4; + let by = (y - self.row_origin) / 4; let bs = size / 4; for j in 0..bs { for i in 0..bs { @@ -183,7 +246,7 @@ impl DecodedFrame { /// cu_transquant_bypass CU. #[inline] pub(crate) fn is_block_bypass(&self, x: u32, y: u32) -> bool { - let idx = ((y / 4) * self.deblock_stride + x / 4) as usize; + let idx = (((y - self.row_origin) / 4) * self.deblock_stride + x / 4) as usize; self.deblock_flags .get(idx) .is_some_and(|f| f & DEBLOCK_FLAG_BYPASS != 0) @@ -794,3 +857,10 @@ impl DecodedFrame { } } } + +pub(crate) fn shift_rows(data: &mut [T], count: usize, value: T) { + let count = count.min(data.len()); + data.copy_within(count.., 0); + let keep = data.len() - count; + data[keep..].fill(value); +} diff --git a/src/heic-decoder/hevc/sao.rs b/src/heic-decoder/hevc/sao.rs index 9e7e93f..dfe64db 100644 --- a/src/heic-decoder/hevc/sao.rs +++ b/src/heic-decoder/hevc/sao.rs @@ -67,12 +67,12 @@ impl SaoMap { #[inline] pub fn get(&self, ctb_x: u32, ctb_y: u32) -> &SaoInfo { - &self.data[(ctb_y * self.width_ctbs + ctb_x) as usize] + &self.data[((ctb_y % self.height_ctbs) * self.width_ctbs + ctb_x) as usize] } #[inline] pub fn get_mut(&mut self, ctb_x: u32, ctb_y: u32) -> &mut SaoInfo { - &mut self.data[(ctb_y * self.width_ctbs + ctb_x) as usize] + &mut self.data[((ctb_y % self.height_ctbs) * self.width_ctbs + ctb_x) as usize] } } @@ -519,3 +519,248 @@ fn apply_sao_edge( } } } + +pub(crate) fn write_filtered_rows( + frame: &DecodedFrame, + map: &SaoMap, + ctb_size: u32, + start: u32, + rows: u32, + output: &mut DecodedFrame, +) { + let bypass = frame.has_bypass_blocks(); + let maximum = (1i32 << frame.bit_depth) - 1; + for component in 0..3 { + let sub = if component == 0 { 1 } else { 2 }; + let origin = frame.plane_origin(component); + let (source, stride) = frame.plane(component); + let (destination, _) = output.plane_mut(component); + let width = frame.width / sub; + let height = frame.height / sub; + let first_y = start / sub; + let last_y = (start + rows) / sub; + for ctb_x in 0..map.width_ctbs { + let first_x = ctb_x * ctb_size / sub; + let last_x = ((ctb_x + 1) * ctb_size / sub).min(width); + let info = map.get(ctb_x, start / ctb_size); + let c = component as usize; + let offsets = info.sao_offset_val[c]; + for y in first_y..last_y { + let input = (y - origin) as usize * stride + first_x as usize; + let target = (y - first_y) as usize * stride + first_x as usize; + let count = (last_x - first_x) as usize; + destination[target..target + count].copy_from_slice(&source[input..input + count]); + } + match info.sao_type_idx[c] { + 1 => { + let mut table = [0i32; 32]; + for (i, &offset) in offsets.iter().enumerate() { + table[(usize::from(info.sao_band_position[c]) + i) & 31] = + i32::from(offset); + } + for y in first_y..last_y { + let input = (y - origin) as usize * stride; + let target = (y - first_y) as usize * stride; + for x in first_x..last_x { + if bypass && frame.is_block_bypass(x * sub, y * sub) { + continue; + } + let sample = i32::from(source[input + x as usize]); + let offset = table[(sample >> (frame.bit_depth - 5)) as usize]; + destination[target + x as usize] = + (sample + offset).clamp(0, maximum) as u16; + } + } + } + 2 => { + let (dx0, dy0, dx1, dy1) = EO_OFFSETS[info.sao_eo_class[c] as usize & 3]; + let x0 = first_x.max((-dx0).max(-dx1).max(0) as u32); + let x1 = last_x.min(width - dx0.max(dx1).max(0) as u32); + let y0 = first_y.max((-dy0).max(-dy1).max(0) as u32); + let y1 = last_y.min(height - dy0.max(dy1).max(0) as u32); + let table = [ + i32::from(offsets[0]), + i32::from(offsets[1]), + 0, + -i32::from(offsets[2]), + -i32::from(offsets[3]), + ]; + for y in y0..y1 { + let input = (y - origin) as usize * stride; + let target = (y - first_y) as usize * stride; + let a = ((y as i32 + dy0 - origin as i32) as usize * stride) as isize + + dx0 as isize; + let b = ((y as i32 + dy1 - origin as i32) as usize * stride) as isize + + dx1 as isize; + if !bypass { + let first = x0 as usize; + let last = x1 as usize; + edge_row( + &source[input + first..input + last], + &source + [(a + first as isize) as usize..(a + last as isize) as usize], + &source + [(b + first as isize) as usize..(b + last as isize) as usize], + &mut destination[target + first..target + last], + offsets, + maximum, + ); + continue; + } + for x in x0..x1 { + if bypass && frame.is_block_bypass(x * sub, y * sub) { + continue; + } + let sample = i32::from(source[input + x as usize]); + let left = i32::from(source[(a + x as isize) as usize]); + let right = i32::from(source[(b + x as isize) as usize]); + let category = + (2 + (sample - left).signum() + (sample - right).signum()) as usize; + destination[target + x as usize] = + (sample + table[category]).clamp(0, maximum) as u16; + } + } + } + _ => {} + } + } + } +} + +fn edge_row( + input: &[u16], + a: &[u16], + b: &[u16], + output: &mut [u16], + offsets: [i8; 4], + maximum: i32, +) { + #[cfg(target_arch = "aarch64")] + let done = { + use archmage::SimdToken; + archmage::NeonToken::summon() + .filter(|_| maximum <= 255) + .map_or(0, |token| { + edge_row_neon(token, input, a, b, output, offsets, maximum as i16) + }) + }; + #[cfg(not(target_arch = "aarch64"))] + let done = 0; + let table = [ + i32::from(offsets[0]), + i32::from(offsets[1]), + 0, + -i32::from(offsets[2]), + -i32::from(offsets[3]), + ]; + for i in done..output.len() { + let sample = i32::from(input[i]); + let category = + 2 + (sample - i32::from(a[i])).signum() + (sample - i32::from(b[i])).signum(); + output[i] = (sample + table[category as usize]).clamp(0, maximum) as u16; + } +} + +#[cfg(target_arch = "aarch64")] +#[archmage::arcane] +fn edge_row_neon( + _token: archmage::NeonToken, + input: &[u16], + a: &[u16], + b: &[u16], + output: &mut [u16], + offsets: [i8; 4], + maximum: i16, +) -> usize { + use core::arch::aarch64::*; + use safe_unaligned_simd::aarch64::{vld1q_u16, vst1q_u16}; + let zero = vdupq_n_s16(0); + let maximum = vdupq_n_s16(maximum); + let values = [ + vdupq_n_s16(i16::from(offsets[0])), + vdupq_n_s16(i16::from(offsets[1])), + vdupq_n_s16(-i16::from(offsets[2])), + vdupq_n_s16(-i16::from(offsets[3])), + ]; + let mut done = 0; + for chunk in output.chunks_exact_mut(8) { + let sample = vld1q_u16(input[done..done + 8].try_into().unwrap()); + let left = vld1q_u16(a[done..done + 8].try_into().unwrap()); + let right = vld1q_u16(b[done..done + 8].try_into().unwrap()); + let sign0 = vsubq_u16(vcltq_u16(sample, left), vcgtq_u16(sample, left)); + let sign1 = vsubq_u16(vcltq_u16(sample, right), vcgtq_u16(sample, right)); + let category = vreinterpretq_s16_u16(vaddq_u16(sign0, sign1)); + let mut offset = zero; + for (value, index) in values.iter().zip([-2, -1, 1, 2]) { + offset = vbslq_s16(vceqq_s16(category, vdupq_n_s16(index)), *value, offset); + } + let result = vminq_s16( + maximum, + vmaxq_s16(zero, vaddq_s16(vreinterpretq_s16_u16(sample), offset)), + ); + vst1q_u16(chunk.try_into().unwrap(), vreinterpretq_u16_s16(result)); + done += 8; + } + done +} + +#[cfg(test)] +mod incremental_tests { + use super::edge_row; + + #[test] + fn edge_rows_match_scalar_categories_clipping_and_tails() { + for maximum in [255, 1023] { + for length in 0..=67 { + let input: Vec = (0..length) + .map(|i| match i % 7 { + 0 => 0, + 1 => maximum as u16, + _ => ((i * 71) % (maximum as usize + 1)) as u16, + }) + .collect(); + let a: Vec = input + .iter() + .enumerate() + .map(|(i, &v)| match i % 3 { + 0 => v.saturating_sub(1), + 1 => v, + _ => v.saturating_add(1).min(maximum as u16), + }) + .collect(); + let b: Vec = input + .iter() + .enumerate() + .map(|(i, &v)| match i / 3 % 3 { + 0 => v.saturating_sub(2), + 1 => v, + _ => v.saturating_add(2).min(maximum as u16), + }) + .collect(); + for offsets in [[0, 0, 0, 0], [1, 2, 3, 4], [7, 6, 5, 4], [-7, 7, -7, 7]] { + let table = [ + i32::from(offsets[0]), + i32::from(offsets[1]), + 0, + -i32::from(offsets[2]), + -i32::from(offsets[3]), + ]; + let expected: Vec = input + .iter() + .zip(&a) + .zip(&b) + .map(|((&v, &a), &b)| { + let category = 2 + + (i32::from(v) - i32::from(a)).signum() + + (i32::from(v) - i32::from(b)).signum(); + (i32::from(v) + table[category as usize]).clamp(0, maximum) as u16 + }) + .collect(); + let mut actual = vec![0; length]; + edge_row(&input, &a, &b, &mut actual, offsets, maximum); + assert_eq!(actual, expected); + } + } + } + } +} diff --git a/src/heic-decoder/hevc/transforms.rs b/src/heic-decoder/hevc/transforms.rs index 10a8fd7..23a988a 100644 --- a/src/heic-decoder/hevc/transforms.rs +++ b/src/heic-decoder/hevc/transforms.rs @@ -59,6 +59,7 @@ impl DecodedFrame { }; Self { + row_origin: 0, width: nw, height: nh, y_plane, @@ -130,6 +131,7 @@ impl DecodedFrame { }; Self { + row_origin: 0, width: w, height: h, y_plane, @@ -204,6 +206,7 @@ impl DecodedFrame { }; Self { + row_origin: 0, width: nw, height: nh, y_plane, @@ -270,6 +273,7 @@ impl DecodedFrame { }; Self { + row_origin: 0, width: w, height: h, y_plane, @@ -336,6 +340,7 @@ impl DecodedFrame { }; Self { + row_origin: 0, width: w, height: h, y_plane, diff --git a/src/lib.rs b/src/lib.rs index 3bc451e..b8da88c 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -47,6 +47,8 @@ use std::path::{Path, PathBuf}; use std::ptr::{self, NonNull}; mod bounded; +#[cfg(feature = "incremental-experiment")] +pub use bounded::incremental::decode as decode_incremental_experiment; #[path = "heic-decoder/mod.rs"] mod heic_decoder; pub use bounded::{ diff --git a/tests/incremental-memory.rs b/tests/incremental-memory.rs new file mode 100644 index 0000000..aae9cec --- /dev/null +++ b/tests/incremental-memory.rs @@ -0,0 +1,238 @@ +#[path = "../src/bin/support/allocation.rs"] +mod allocation; +use allocation::*; +use heic_decoder::{BoundedDecodeOptions, BoundedInput, decode_incremental_experiment}; +use std::path::Path; +use std::sync::atomic::Ordering; + +fn measure( + input: BoundedInput<'_>, + side: u32, + budget: usize, + request_limit: usize, + expect_sao_edges: bool, +) -> (heic_decoder::BoundedRgbImage, usize) { + let baseline = LIVE.load(Ordering::SeqCst); + PEAK.store(baseline, Ordering::SeqCst); + LARGEST.store(0, Ordering::SeqCst); + DENIED.store(0, Ordering::SeqCst); + MAX_REQUEST.store(request_limit, Ordering::SeqCst); + MAX_LIVE.store(baseline + budget, Ordering::SeqCst); + ACTIVE.store(true, Ordering::SeqCst); + let result = decode_incremental_experiment( + input, + BoundedDecodeOptions { + max_side: side, + max_memory_bytes: budget, + }, + ); + ACTIVE.store(false, Ordering::SeqCst); + let peak = PEAK.load(Ordering::SeqCst) - baseline; + assert_eq!(DENIED.load(Ordering::SeqCst), 0); + assert!(peak <= budget); + let (image, stats) = result.unwrap(); + assert!(stats[0] > 0); + assert_eq!(stats[1] > 0, expect_sao_edges); + (image, peak) +} + +fn reference(path: &Path, width: u32, height: u32) -> Vec { + let golden = path.with_extension("png"); + let image = { + let decoder = png::Decoder::new(std::io::BufReader::new( + std::fs::File::open(golden).unwrap(), + )); + let mut reader = decoder.read_info().unwrap(); + let mut pixels = vec![0; reader.output_buffer_size().unwrap()]; + let info = reader.next_frame(&mut pixels).unwrap(); + assert_eq!(info.color_type, png::ColorType::Rgba); + assert_eq!(info.bit_depth, png::BitDepth::Eight); + pixels.truncate(info.buffer_size()); + assert!(pixels.chunks_exact(4).all(|p| p[3] == 255)); + heic_decoder::DecodedRgbImage { + width: info.width, + height: info.height, + pixels: pixels + .chunks_exact(4) + .flat_map(|p| p[..3].iter().copied()) + .collect(), + source_bit_depth: 8, + icc_profile: None, + } + }; + let mut output = vec![0; (width * height * 3) as usize]; + for oy in 0..height { + let y0 = f64::from(oy) * f64::from(image.height) / f64::from(height); + let y1 = f64::from(oy + 1) * f64::from(image.height) / f64::from(height); + for ox in 0..width { + let x0 = f64::from(ox) * f64::from(image.width) / f64::from(width); + let x1 = f64::from(ox + 1) * f64::from(image.width) / f64::from(width); + let mut sum = [0.0; 3]; + for sy in y0.floor() as u32..y1.ceil() as u32 { + for sx in x0.floor() as u32..x1.ceil() as u32 { + let weight = (x1.min(f64::from(sx + 1)) - x0.max(f64::from(sx))) + * (y1.min(f64::from(sy + 1)) - y0.max(f64::from(sy))); + let i = ((sy * image.width + sx) * 3) as usize; + for (c, value) in sum.iter_mut().enumerate() { + *value += f64::from(image.pixels[i + c]) * weight; + } + } + } + let i = ((oy * width + ox) * 3) as usize; + for (c, value) in sum.iter().enumerate() { + output[i + c] = (value / ((x1 - x0) * (y1 - y0))).round() as u8; + } + } + } + output +} + +fn main() { + if cfg!(feature = "decoder-tracing") { + return; + } + let directory = std::env::var_os("INCREMENTAL_FIXTURES_DIR").map_or_else( + || Path::new(env!("CARGO_MANIFEST_DIR")).join(".heic-test-assets/incremental-corpus"), + std::path::PathBuf::from, + ); + assert!( + directory.join("manifest.json").is_file(), + "Generate the external corpus with scripts/incremental/generate.rb; see TESTING.md" + ); + let mut scratch = Vec::new(); + let mut odd_scratch = Vec::new(); + for name in [ + "direct", + "tall", + "grid", + "grid-pipeline", + "undefined-grid-nclx", + "crop", + "oriented", + "pipeline", + "odd-short", + "odd-tall", + "odd-pipeline", + "odd-width-grid", + "odd-height-grid", + ] { + let path = directory.join(format!("{name}.heic")); + let bytes = std::fs::read(&path).unwrap(); + let budget = if name.ends_with("pipeline") { + 8 * 1024 * 1024 + } else { + 4 * 1024 * 1024 + }; + let request_limit = if name.ends_with("pipeline") { + 512 * 1024 + } else { + 128 * 1024 + }; + let expect_sao_edges = name != "odd-height-grid"; + let before = THREADS.load(Ordering::SeqCst); + let (file_image, path_peak) = measure( + BoundedInput::Path(&path), + 65, + budget, + request_limit, + expect_sao_edges, + ); + let path_workers = THREADS.load(Ordering::SeqCst) - before; + let before = THREADS.load(Ordering::SeqCst); + let (byte_image, byte_peak) = measure( + BoundedInput::Bytes(&bytes), + 65, + budget, + request_limit, + expect_sao_edges, + ); + let byte_workers = THREADS.load(Ordering::SeqCst) - before; + assert_eq!(file_image.image.pixels, byte_image.image.pixels); + assert!(path_peak.abs_diff(byte_peak) < 64 * 1024); + let image = &file_image.image; + let expected = reference(&path, image.width, image.height); + assert!( + image + .pixels + .iter() + .zip(expected) + .all(|(&a, b)| a.abs_diff(b) <= 1), + "{name}: area-filter parity" + ); + if name.ends_with("pipeline") { + assert!(path_workers > 0); + assert!(byte_workers > 0); + } + if name == "direct" || name == "tall" { + scratch.push(path_peak - image.pixels.len()); + } + if name == "odd-short" || name == "odd-tall" { + odd_scratch.push(path_peak - image.pixels.len()); + } + println!( + "{name}: {}x{}, heap_peak={path_peak}, largest_request={}", + image.width, + image.height, + LARGEST.load(Ordering::SeqCst) + ); + } + assert!(scratch[0].abs_diff(scratch[1]) < 16 * 1024); + assert!(odd_scratch[0].abs_diff(odd_scratch[1]) < 16 * 1024); + let path = directory.join("invalid-grid-references.bin"); + let bytes = std::fs::read(&path).unwrap(); + for input in [BoundedInput::Path(&path), BoundedInput::Bytes(&bytes)] { + let budget = 256 * 1024; + let baseline = LIVE.load(Ordering::SeqCst); + PEAK.store(baseline, Ordering::SeqCst); + LARGEST.store(0, Ordering::SeqCst); + DENIED.store(0, Ordering::SeqCst); + MAX_REQUEST.store(budget, Ordering::SeqCst); + MAX_LIVE.store(baseline + budget, Ordering::SeqCst); + ACTIVE.store(true, Ordering::SeqCst); + let result = decode_incremental_experiment( + input, + BoundedDecodeOptions { + max_side: 65, + max_memory_bytes: budget, + }, + ); + ACTIVE.store(false, Ordering::SeqCst); + assert!(matches!( + result, + Err(heic_decoder::BoundedDecodeError::Malformed("tile count")) + )); + assert_eq!(DENIED.load(Ordering::SeqCst), 0); + assert!(PEAK.load(Ordering::SeqCst) - baseline <= budget); + println!( + "invalid grid count: peak={}, largest_request={}", + PEAK.load(Ordering::SeqCst) - baseline, + LARGEST.load(Ordering::SeqCst) + ); + } + let path = directory.join("tall.heic"); + let baseline = LIVE.load(Ordering::SeqCst); + PEAK.store(baseline, Ordering::SeqCst); + LARGEST.store(0, Ordering::SeqCst); + MAX_REQUEST.store(64 * 1024, Ordering::SeqCst); + MAX_LIVE.store(baseline + 3 * 1024 * 1024, Ordering::SeqCst); + ACTIVE.store(true, Ordering::SeqCst); + let result = decode_incremental_experiment( + BoundedInput::Path(&path), + BoundedDecodeOptions { + max_side: 6000, + max_memory_bytes: 3 * 1024 * 1024, + }, + ); + ACTIVE.store(false, Ordering::SeqCst); + assert!(matches!( + result, + Err(heic_decoder::BoundedDecodeError::MemoryBudgetExceeded { .. }) + )); + assert!(LARGEST.load(Ordering::SeqCst) < 65536); + assert_eq!(DENIED.load(Ordering::SeqCst), 0); + println!( + "under-budget rejection: peak={}, largest_request={}", + PEAK.load(Ordering::SeqCst) - baseline, + LARGEST.load(Ordering::SeqCst) + ); +}