diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 8713c08..4e63809 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -130,6 +130,11 @@ jobs: run: | set -euo pipefail perl -0pi -e 's/(\[workspace\.package\][\s\S]*?\nversion = ")[^"]+(")/$1$ENV{NEXT_VERSION}$2/' Cargo.toml + # The library and its TinyBus contract ship from the same source tag. + # Raise the package dependency floor with that tag so packaged + # consumers cannot resolve an older contract without the new API. + perl -0pi -e 's/(tinydocs-bus = \{ version = ")[^"]+("[,}])/$1$ENV{NEXT_VERSION}$2/' Cargo.toml + grep -Fq "tinydocs-bus = { version = \"$NEXT_VERSION\", path = \"crates/tinydocs-bus\" }" Cargo.toml cargo update -p tinydocs-bus --precise "$NEXT_VERSION" cargo update -p "$CRATE_NAME" --precise "$NEXT_VERSION" diff --git a/Cargo.lock b/Cargo.lock index b6ccc12..9e7a72e 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -37,6 +37,12 @@ dependencies = [ "memchr", ] +[[package]] +name = "allocator-api2" +version = "0.2.21" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "683d7910e743518b0e34f1186f92494becacb047c7b6bf616c96772180fef923" + [[package]] name = "anyhow" version = "1.0.104" @@ -117,6 +123,12 @@ version = "0.8.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "5e764a1d40d510daf35e07be9eb06e75770908c27d411ee6c92109c9840eaaf7" +[[package]] +name = "bitflags" +version = "1.3.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bef38d45163c2f1dde094a7dfd33ccf595c92905c8f8f4fdc18d06fb1037718a" + [[package]] name = "bitflags" version = "2.13.2" @@ -268,6 +280,15 @@ dependencies = [ "inout", ] +[[package]] +name = "color" +version = "0.3.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2ec7c5eb7a16992b1904d76c517d170ab353b0e0b3d5a0c81a8a0cd1037893cf" +dependencies = [ + "bytemuck", +] + [[package]] name = "color_quant" version = "1.1.0" @@ -400,6 +421,15 @@ dependencies = [ "syn 3.0.6", ] +[[package]] +name = "document-features" +version = "0.2.12" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d4b8a88685455ed29a21542a33abd9cb6510b6b129abadabdcef0f4c55bc8f61" +dependencies = [ + "litrs", +] + [[package]] name = "docx-rs" version = "0.4.22" @@ -485,6 +515,18 @@ dependencies = [ "regex-syntax", ] +[[package]] +name = "fast_image_resize" +version = "5.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "fbc7fe45cf92b43817ff62a3723e862b85bd1d06288f63007f7645d1d2f7a060" +dependencies = [ + "cfg-if", + "document-features", + "num-traits", + "thiserror", +] + [[package]] name = "fastrand" version = "2.5.0" @@ -506,6 +548,15 @@ dependencies = [ "simd-adler32", ] +[[package]] +name = "fearless_simd" +version = "0.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8fb2907d1f08b2b316b9223ced5b0e89d87028ba8deae9764741dba8ff7f3903" +dependencies = [ + "bytemuck", +] + [[package]] name = "filetime" version = "0.2.29" @@ -539,6 +590,21 @@ version = "1.0.7" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "3f9eec918d3f24069decb9af1554cad7c880e2da24a9afd88aca000531ab82c1" +[[package]] +name = "foldhash" +version = "0.1.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d9c4f5dac5e15c24eb999c26181a6ca40b39fe946cbe4c263c7209467bc83af2" + +[[package]] +name = "font-types" +version = "0.10.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "39a654f404bbcbd48ea58c617c2993ee91d1cb63727a37bf2323a4edeed1b8c5" +dependencies = [ + "bytemuck", +] + [[package]] name = "font-types" version = "0.11.3" @@ -640,12 +706,108 @@ dependencies = [ "zerocopy", ] +[[package]] +name = "hashbrown" +version = "0.15.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9229cfe53dfd69f0609a49f65461bd93001ea1ef889cd5529dd176593f5338a1" +dependencies = [ + "allocator-api2", + "equivalent", + "foldhash", +] + [[package]] name = "hashbrown" version = "0.17.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "ed5909b6e89a2db4456e54cd5f673791d7eca6732202bbf2a9cc504fe2f9b84a" +[[package]] +name = "hayro" +version = "0.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6ec16ea18d2ab01c94ab300f9415f8900cd056085c6926f70a7c3e7391192769" +dependencies = [ + "bytemuck", + "fast_image_resize", + "hayro-interpret", + "image", + "kurbo 0.12.0", + "vello_cpu", +] + +[[package]] +name = "hayro-ccitt" +version = "0.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2db6494c3070f0e3cd9de52ee1a562ba3b2f832cd17b5537180c28294ca088eb" + +[[package]] +name = "hayro-font" +version = "0.4.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d688da07ebfa1f0508697bdcc739a80a840706e2cfba39360820003a9bc49ed2" +dependencies = [ + "log", + "phf", +] + +[[package]] +name = "hayro-interpret" +version = "0.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "298f36b94e75fa4055b4a71db9c641085d6197b214a4e13f3c4aedebb5cea234" +dependencies = [ + "bitflags 2.13.2", + "hayro-font", + "hayro-syntax", + "kurbo 0.12.0", + "log", + "moxcms 0.7.11", + "phf", + "rustc-hash", + "siphasher", + "skrifa 0.40.0", + "smallvec", + "yoke", +] + +[[package]] +name = "hayro-jbig2" +version = "0.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7d7d4358462d3e5aeeb4dcfdea5467821d6551f6cb4607cc037ac99131bc1e32" +dependencies = [ + "hayro-ccitt", +] + +[[package]] +name = "hayro-jpeg2000" +version = "0.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "07d8a5e080cc7429956acf933caf1ef8d4880ae70cbb26eeaec3ae228c7c5f61" +dependencies = [ + "fearless_simd", + "log", +] + +[[package]] +name = "hayro-syntax" +version = "0.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b08efad1aa85ea1e19ae3bb46832779222c32887fae7b30d53a07b1d1c0ba553" +dependencies = [ + "flate2", + "hayro-ccitt", + "hayro-jbig2", + "hayro-jpeg2000", + "log", + "rustc-hash", + "smallvec", + "zune-jpeg", +] + [[package]] name = "hmac" version = "0.12.1" @@ -690,9 +852,9 @@ dependencies = [ "byteorder-lite", "color_quant", "gif", - "moxcms", + "moxcms 0.8.1", "num-traits", - "png", + "png 0.18.1", "tiff", "zune-core", "zune-jpeg", @@ -705,7 +867,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "cc4e190f5d26ca7051642629da2c52fc03bde85a03197c99408dcd291734c855" dependencies = [ "equivalent", - "hashbrown", + "hashbrown 0.17.1", ] [[package]] @@ -745,6 +907,17 @@ dependencies = [ "wasm-bindgen", ] +[[package]] +name = "kurbo" +version = "0.12.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ce9729cc38c18d86123ab736fd2e7151763ba226ac2490ec092d1dd148825e32" +dependencies = [ + "arrayvec", + "euclid 0.22.14", + "smallvec", +] + [[package]] name = "kurbo" version = "0.13.1" @@ -763,6 +936,12 @@ version = "0.2.189" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "3eaf3ede3fee6db1a4c2ee091bf8a8b4dccdc6d17f656fb07896ee72867612f2" +[[package]] +name = "linebender_resource_handle" +version = "0.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d4a5ff6bcca6c4867b1c4fd4ef63e4db7436ef363e0ad7531d1558856bae64f4" + [[package]] name = "linked-hash-map" version = "0.5.6" @@ -775,6 +954,12 @@ version = "0.12.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "32a66949e030da00e8c7d4434b251670a91556f4144941d37452769c25d58a53" +[[package]] +name = "litrs" +version = "1.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "11d3d7f243d5c5a8b9bb5d6dd2b1602c0cb0b9db1621bafc7ed66e35ff9fe092" + [[package]] name = "log" version = "0.4.34" @@ -788,7 +973,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "25aab26d99567469098e64a02f42679f8965c6401263eefa31d8f2dcc37a221c" dependencies = [ "aes", - "bitflags", + "bitflags 2.13.2", "cbc", "ecb", "encoding_rs", @@ -844,6 +1029,16 @@ dependencies = [ "simd-adler32", ] +[[package]] +name = "moxcms" +version = "0.7.11" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ac9557c559cd6fc9867e122e20d2cbefc9ca29d80d027a8e39310920ed2f0a97" +dependencies = [ + "num-traits", + "pxfm", +] + [[package]] name = "moxcms" version = "0.8.1" @@ -950,12 +1145,68 @@ dependencies = [ "ttf-parser", ] +[[package]] +name = "peniko" +version = "0.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b3c76095c9a636173600478e0373218c7b955335048c2bcd12dc6a79657649d8" +dependencies = [ + "bytemuck", + "color", + "kurbo 0.12.0", + "linebender_resource_handle", + "smallvec", +] + [[package]] name = "percent-encoding" version = "2.3.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "9b4f627cb1b25917193a259e49bdad08f671f8d9708acfd5fe0a8c1455d87220" +[[package]] +name = "phf" +version = "0.13.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c1562dc717473dbaa4c1f85a36410e03c047b2e7df7f45ee938fbef64ae7fadf" +dependencies = [ + "phf_macros", + "phf_shared", + "serde", +] + +[[package]] +name = "phf_generator" +version = "0.13.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "135ace3a761e564ec88c03a77317a7c6b80bb7f7135ef2544dbe054243b89737" +dependencies = [ + "fastrand", + "phf_shared", +] + +[[package]] +name = "phf_macros" +version = "0.13.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "812f032b54b1e759ccd5f8b6677695d5268c588701effba24601f6932f8269ef" +dependencies = [ + "phf_generator", + "phf_shared", + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "phf_shared" +version = "0.13.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e57fef6bc5981e38c2ce2d63bfa546861309f875b8a75f092d1d54ae2d64f266" +dependencies = [ + "siphasher", +] + [[package]] name = "pin-project-lite" version = "0.2.17" @@ -981,13 +1232,26 @@ dependencies = [ "time", ] +[[package]] +name = "png" +version = "0.17.16" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "82151a2fc869e011c153adc57cf2789ccb8d9906ce52c0b39a6b5697749d7526" +dependencies = [ + "bitflags 1.3.2", + "crc32fast", + "fdeflate", + "flate2", + "miniz_oxide 0.8.9", +] + [[package]] name = "png" version = "0.18.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "60769b8b31b2a9f263dae2776c37b1b28ae246943cf719eb6946a1db05128a61" dependencies = [ - "bitflags", + "bitflags 2.13.2", "crc32fast", "fdeflate", "flate2", @@ -1124,6 +1388,26 @@ version = "1.8.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "a611d15b50743feb4c76b7d03edcb0e64f399c26961e4efe6975bc398be6aa3d" +[[package]] +name = "read-fonts" +version = "0.35.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6717cf23b488adf64b9d711329542ba34de147df262370221940dfabc2c91358" +dependencies = [ + "bytemuck", + "font-types 0.10.1", +] + +[[package]] +name = "read-fonts" +version = "0.37.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7b634fabf032fab15307ffd272149b622260f55974d9fad689292a5d33df02e5" +dependencies = [ + "bytemuck", + "font-types 0.11.3", +] + [[package]] name = "read-fonts" version = "0.39.2" @@ -1131,7 +1415,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "c4ed38b89c2c77ff968c524145ad65fb010f38af5c7a224b53b81d47ac2daa81" dependencies = [ "bytemuck", - "font-types", + "font-types 0.11.3", ] [[package]] @@ -1189,7 +1473,7 @@ version = "1.1.5" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "891efababe418670775f199f0d233d84843c227a0949a883ce15b37c78d6629d" dependencies = [ - "bitflags", + "bitflags 2.13.2", "errno", "libc", "linux-raw-sys", @@ -1355,6 +1639,32 @@ version = "0.1.5" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "e3a9fe34e3e7a50316060351f37187a3f546bce95496156754b601a5fa71b76e" +[[package]] +name = "siphasher" +version = "1.0.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "33f4fe9184a62d842c9ef383018f3306d8ba224fd9d836f56d7288308847c256" + +[[package]] +name = "skrifa" +version = "0.37.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8c31071dedf532758ecf3fed987cdb4bd9509f900e026ab684b4ecb81ea49841" +dependencies = [ + "bytemuck", + "read-fonts 0.35.0", +] + +[[package]] +name = "skrifa" +version = "0.40.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7fbdfe3d2475fbd7ddd1f3e5cf8288a30eb3e5f95832829570cd88115a7434ac" +dependencies = [ + "bytemuck", + "read-fonts 0.37.0", +] + [[package]] name = "skrifa" version = "0.42.1" @@ -1362,7 +1672,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "0c34617370ae968efb7161bb2beb517d9084659aae19e24b89e3db25b46e4564" dependencies = [ "bytemuck", - "read-fonts", + "read-fonts 0.39.2", ] [[package]] @@ -1377,6 +1687,12 @@ version = "1.16.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "f9395f0f0eee849a9b707b2f06bb92a6a422090e2123bb2ef8e87a0e61892a8e" +[[package]] +name = "stable_deref_trait" +version = "1.2.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6ce2be8dc25455e1f91df71bfa12ad37d7af1092ae736f3a6cd0e37bc7810596" + [[package]] name = "stringprep" version = "0.1.5" @@ -1394,9 +1710,9 @@ version = "0.2.6" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "38803281d1c23166c5ebcb455439a5d2afe711cc909cf88af72448c297756ad6" dependencies = [ - "kurbo", + "kurbo 0.13.1", "rustc-hash", - "skrifa", + "skrifa 0.42.1", "write-fonts", ] @@ -1428,6 +1744,17 @@ dependencies = [ "unicode-ident", ] +[[package]] +name = "synstructure" +version = "0.14.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "901704edd0dfe137f1987838ee4f259e4e063c31371bdb423f7ae38ec6f77f02" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.6", +] + [[package]] name = "syntect" version = "5.3.0" @@ -1583,8 +1910,11 @@ name = "tinydocs" version = "0.1.20" dependencies = [ "docx-rs", + "hayro", "pdf-extract", + "png 0.17.16", "ppt-rs", + "quick-xml 0.41.0", "serde_json", "tinydocs-bus", "zip 8.6.0", @@ -1614,6 +1944,7 @@ dependencies = [ "tinydocs", "tinydocs-bus", "tokio", + "zip 8.6.0", ] [[package]] @@ -1820,6 +2151,33 @@ dependencies = [ "wasm-bindgen", ] +[[package]] +name = "vello_common" +version = "0.0.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2dc01cc9e1e2511e5e77024e8d4c2287b9272cafc05ba669ed1056579219dd73" +dependencies = [ + "bytemuck", + "fearless_simd", + "hashbrown 0.15.5", + "log", + "peniko", + "png 0.17.16", + "skrifa 0.37.0", + "smallvec", +] + +[[package]] +name = "vello_cpu" +version = "0.0.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9228d8ed0f1f030de9a09b66b4750012194c0f171030ec74214ba1ca3b9eed23" +dependencies = [ + "bytemuck", + "hashbrown 0.15.5", + "vello_common", +] + [[package]] name = "version_check" version = "0.9.5" @@ -2026,11 +2384,11 @@ version = "0.48.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "cb731d4c4d93eacc69a1ad2f270f905788a98e4a3438267bcafbe08d3431c8d8" dependencies = [ - "font-types", + "font-types 0.11.3", "indexmap", - "kurbo", + "kurbo 0.13.1", "log", - "read-fonts", + "read-fonts 0.39.2", ] [[package]] @@ -2058,6 +2416,29 @@ dependencies = [ "linked-hash-map", ] +[[package]] +name = "yoke" +version = "0.8.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "709fe23a0424b6a435d82152b1bd3fdfb0833487d5fa90d05d42762a9891fef5" +dependencies = [ + "stable_deref_trait", + "yoke-derive", + "zerofrom", +] + +[[package]] +name = "yoke-derive" +version = "0.8.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ec8ebde2db3681e8c9980cc27822030e68752690ddfa9473e739aeb4dbde6d71" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.6", + "synstructure", +] + [[package]] name = "zerocopy" version = "0.8.59" @@ -2078,6 +2459,27 @@ dependencies = [ "syn 2.0.119", ] +[[package]] +name = "zerofrom" +version = "0.1.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0ec05a11813ea801ff6d75110ad09cd0824ddba17dfe17128ea0d5f68e6c5272" +dependencies = [ + "zerofrom-derive", +] + +[[package]] +name = "zerofrom-derive" +version = "0.1.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f75b4683f6c7f45248d4d64056a24298c6281e0993356d7d1b4a1a962ef10d4a" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.6", + "synstructure", +] + [[package]] name = "zeroize" version = "1.9.0" diff --git a/Cargo.toml b/Cargo.toml index 3a3233c..b030823 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -59,7 +59,16 @@ ppt-rs = { version = "0.2", optional = true } # it. pdf-extract = { version = "0.12", optional = true } +# Bounded OOXML intake reads only selected XML parts, never extracts files. +zip = { version = "8", default-features = false, features = ["deflate"], optional = true } +# Streaming XML events avoid building attacker-controlled DOM trees. +quick-xml = { version = "0.41", optional = true } +# Pure Rust PDF rasterization, isolated from the default extraction build. +hayro = { version = "0.5", optional = true } + [dev-dependencies] +# Decode raster fixtures to assert scanned-page pixels survive rendering. +png = "0.17" # `.docx` output is a zip container; the tests re-open the produced bytes and # assert on the OOXML parts inside. zip = { version = "8", default-features = false, features = ["deflate"] } @@ -82,6 +91,10 @@ docx = ["dep:docx-rs"] pptx = ["dep:ppt-rs"] # `.pdf` text extraction via `pdf-extract`. pdf = ["dep:pdf-extract"] +# Bounded DOCX/PPTX/XLSX and per-page PDF text intake. +intake = ["pdf", "dep:zip", "dep:quick-xml"] +# Selected PDF page rendering to PNG. +pdf-render = ["pdf", "dep:hayro"] # Lints apply to the whole crate and to every target. CI runs clippy with # `-D warnings`, so anything set to "warn" here fails the build in CI. diff --git a/README.md b/README.md index 26b3dec..4ae6bd2 100644 --- a/README.md +++ b/README.md @@ -136,6 +136,8 @@ exposes: GenerateDocx(DocumentSpec) -> OutputRef GeneratePptx(deck, Option) -> OutputRef ExtractText(StreamRef) -> OutputRef +ExtractDocument(ExtractDocumentSpec, StreamRef) -> ExtractedDocument +RenderPdf(RenderPdfSpec, StreamRef) -> RenderedPdf ReadOutput(output_id, offset, len) -> base64 ReleaseOutput(output_id) -> () ``` @@ -241,3 +243,19 @@ cargo clippy --all-targets --no-default-features -- -D warnings ## License GPL-3.0-only. See [LICENSE](LICENSE). + +## Document intake and vision inputs + +Optional `intake` extracts bounded PDF page text and DOCX/PPTX/XLSX visible +text with source provenance. Optional `pdf-render` rasterizes explicitly +selected PDF pages to PNG for host-owned vision. The shipped module enables +both features. See [intake bounds](src/intake/README.md) and +[rendering bounds](src/pdf_render/README.md) for limits and parser resource +limitations. Existing generation and `ExtractText` payloads are unchanged. + +The new methods extend contract version 2 additively. A version-2 module from +an older release can still serve the original five methods but will not have +`ExtractDocument` or `RenderPdf`; hosts must detect unavailable members and +require a published module release for intake. Do not derive release checksums +from a local build. `OutputRef` is now defined in `tinydocs-bus` and remains +re-exported by `tinydocs-module` with the same wire fields. diff --git a/crates/tinydocs-bus/src/intake/mod.rs b/crates/tinydocs-bus/src/intake/mod.rs new file mode 100644 index 0000000..2f4df50 --- /dev/null +++ b/crates/tinydocs-bus/src/intake/mod.rs @@ -0,0 +1,8 @@ +//! Bounded document intake and selected PDF rasterization contracts. + +mod types; +pub use types::*; + +#[cfg(test)] +#[path = "mod_tests.rs"] +mod tests; diff --git a/crates/tinydocs-bus/src/intake/mod_tests.rs b/crates/tinydocs-bus/src/intake/mod_tests.rs new file mode 100644 index 0000000..a0bcde2 --- /dev/null +++ b/crates/tinydocs-bus/src/intake/mod_tests.rs @@ -0,0 +1,71 @@ +//! Intake wire round trips and strict field handling. +#![allow(clippy::unwrap_used, clippy::expect_used, clippy::panic)] +use super::*; +#[test] +fn request_and_result_payloads_round_trip() { + let extraction = ExtractDocumentSpec::new(DocumentFormat::Xlsx); + assert_eq!( + serde_json::from_str::(&serde_json::to_string(&extraction).unwrap()) + .unwrap(), + extraction + ); + let result = ExtractedDocument { + format: DocumentFormat::Pdf, + section_count: 2, + sections: vec![DocumentSectionText { + source: "page:2".into(), + index: 2, + text: String::new(), + scanned_candidate: true, + }], + truncated: true, + }; + assert_eq!( + serde_json::from_str::(&serde_json::to_string(&result).unwrap()) + .unwrap(), + result + ); + let render = RenderPdfSpec { + pages: vec![2], + max_dimension: 1024, + max_total_pixels: 4_000_000, + max_output_bytes: 8_000_000, + }; + assert_eq!( + serde_json::from_str::(&serde_json::to_string(&render).unwrap()).unwrap(), + render + ); + let output = RenderedPdf { + page_count: 2, + pages: vec![RenderedPdfPage { + page: 2, + width: 10, + height: 20, + output: OutputRef { + output_id: "capability".into(), + total_bytes: 42, + sha256: "abc".into(), + }, + }], + }; + assert_eq!( + serde_json::from_str::(&serde_json::to_string(&output).unwrap()).unwrap(), + output + ); +} +#[test] +fn rejects_unknown_wire_fields_and_formats() { + assert!( + serde_json::from_str::( + r#"{"format":"docx","max_text_bytes":100,"max_sections":1,"extra":true}"# + ) + .is_err() + ); + assert!(serde_json::from_str::(r#""doc""#).is_err()); + assert!( + serde_json::from_str::( + r#"{"output_id":"x","total_bytes":1,"sha256":"y","bytes":[]}"# + ) + .is_err() + ); +} diff --git a/crates/tinydocs-bus/src/intake/types.rs b/crates/tinydocs-bus/src/intake/types.rs new file mode 100644 index 0000000..7608029 --- /dev/null +++ b/crates/tinydocs-bus/src/intake/types.rs @@ -0,0 +1,110 @@ +//! Supplied-byte extraction and selected PDF rendering wire types. +use serde::{Deserialize, Serialize}; + +/// Supported document formats; callers identify the format before sending bytes. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum DocumentFormat { + /// PDF text layers, preserving page boundaries. + Pdf, + /// Word document body. + Docx, + /// `PowerPoint` slide text. + Pptx, + /// Excel worksheet cells. + Xlsx, +} +/// Extraction budgets, subject to library hard ceilings. +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct ExtractDocumentSpec { + /// Input document format. + pub format: DocumentFormat, + /// Total UTF-8 text bytes, at most 200,000. + pub max_text_bytes: u32, + /// Maximum returned sections, at most 256. + pub max_sections: u32, +} +impl ExtractDocumentSpec { + /// Default bounded extraction for a format. + #[must_use] + pub const fn new(format: DocumentFormat) -> Self { + Self { + format, + max_text_bytes: 100_000, + max_sections: 128, + } + } +} +/// One document body, slide, worksheet or PDF page. +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct DocumentSectionText { + /// Stable document part path or `page:N` label. + pub source: String, + /// One-based page/slide/sheet index. + pub index: u32, + /// Visible extracted text; empty text is preserved. + pub text: String, + /// A PDF page with no text layer; a host may offer vision/OCR. + pub scanned_candidate: bool, +} +/// Extraction result, including explicit text/section truncation. +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct ExtractedDocument { + /// Input format. + pub format: DocumentFormat, + /// Total number of document parts/pages before section truncation. + pub section_count: u32, + /// Extracted sections in document order. + pub sections: Vec, + /// True when text or sections were omitted by a budget. + pub truncated: bool, +} +/// PDF page rendering budgets; pages are explicitly selected and one-based. +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct RenderPdfSpec { + /// Selected page numbers, at most eight, without duplicates. + pub pages: Vec, + /// Longest raster edge, at most 2048 pixels. + pub max_dimension: u32, + /// Aggregate pixel budget, at most 16 million. + pub max_total_pixels: u64, + /// Aggregate encoded PNG budget, at most 32 MiB. + pub max_output_bytes: u64, +} +/// Handle for an output retained until released or expired. +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct OutputRef { + /// Opaque output capability. + pub output_id: String, + /// Byte length of the output. + pub total_bytes: u64, + /// Lowercase hex SHA-256 of the output. + pub sha256: String, +} +/// One PNG raster held through the existing output lifecycle. +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct RenderedPdfPage { + /// One-based source page. + pub page: u32, + /// Raster width in pixels. + pub width: u32, + /// Raster height in pixels. + pub height: u32, + /// PNG bytes read with `ReadOutput` and freed with `ReleaseOutput`. + pub output: OutputRef, +} +/// Selected PDF rasters and source page count. +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct RenderedPdf { + /// Total source pages. + pub page_count: u32, + /// Selected pages in request order. + pub pages: Vec, +} diff --git a/crates/tinydocs-bus/src/lib.rs b/crates/tinydocs-bus/src/lib.rs index c7f776e..b0f1e9a 100644 --- a/crates/tinydocs-bus/src/lib.rs +++ b/crates/tinydocs-bus/src/lib.rs @@ -21,3 +21,10 @@ pub use spec::{ WirePresentationSpec, WireSlideImage, WireSlideSpec, }; pub use version::{CONTRACT_VERSION, is_compatible}; + +/// Document intake and raster output vocabulary. +pub mod intake; +pub use intake::{ + DocumentFormat, DocumentSectionText, ExtractDocumentSpec, ExtractedDocument, OutputRef, + RenderPdfSpec, RenderedPdf, RenderedPdfPage, +}; diff --git a/crates/tinydocs-bus/src/names.rs b/crates/tinydocs-bus/src/names.rs index af0e444..6596f62 100644 --- a/crates/tinydocs-bus/src/names.rs +++ b/crates/tinydocs-bus/src/names.rs @@ -14,6 +14,10 @@ pub mod methods { pub const GENERATE_PPTX: &str = "GeneratePptx"; /// `ExtractText` — extract text from a streamed PDF. pub const EXTRACT_TEXT: &str = "ExtractText"; + /// `ExtractDocument` — bounded document text and provenance from a stream. + pub const EXTRACT_DOCUMENT: &str = "ExtractDocument"; + /// `RenderPdf` — selected PDF pages as held PNG outputs. + pub const RENDER_PDF: &str = "RenderPdf"; /// `ReadOutput` — read a bounded base64-encoded output chunk. pub const READ_OUTPUT: &str = "ReadOutput"; /// `ReleaseOutput` — release a held output. @@ -21,10 +25,12 @@ pub mod methods { } /// All method names in the declaration order used by the module interface. -pub const METHODS: [&str; 5] = [ +pub const METHODS: [&str; 7] = [ methods::GENERATE_DOCX, methods::GENERATE_PPTX, methods::EXTRACT_TEXT, + methods::EXTRACT_DOCUMENT, + methods::RENDER_PDF, methods::READ_OUTPUT, methods::RELEASE_OUTPUT, ]; diff --git a/crates/tinydocs-bus/src/version.rs b/crates/tinydocs-bus/src/version.rs index 8472dbd..bb2ae7e 100644 --- a/crates/tinydocs-bus/src/version.rs +++ b/crates/tinydocs-bus/src/version.rs @@ -1,6 +1,9 @@ //! Wire-contract versioning for `TinyDocs` hosts. /// Current `TinyDocs` wire-contract version. +/// +/// Intake methods are additive to version 2. Older version-2 modules may not +/// serve them; hosts must check method availability or require a newer release. pub const CONTRACT_VERSION: u32 = 2; /// Returns whether a host requiring `required` can bind to this contract. diff --git a/crates/tinydocs-module/Cargo.toml b/crates/tinydocs-module/Cargo.toml index eb5a8d3..85a5a9b 100644 --- a/crates/tinydocs-module/Cargo.toml +++ b/crates/tinydocs-module/Cargo.toml @@ -17,7 +17,7 @@ crate-type = ["rlib", "cdylib"] [dependencies] # The pure document library remains independently publishable and bus-agnostic. -tinydocs = { path = "../..", default-features = false, features = ["docx", "pptx", "pdf"] } +tinydocs = { path = "../..", default-features = false, features = ["docx", "pptx", "pdf", "intake", "pdf-render"] } # The transport-free vocabulary shared with every TinyBus host. tinydocs-bus = { path = "../tinydocs-bus" } # TinyBus provides the typed service interface and dynamic module host ABI. @@ -52,3 +52,4 @@ base64 = "0.23" # document directly rather than depending on the module's own wire types, so a # rename here would be caught as a contract change. serde_json = "1" +zip = { version = "8", default-features = false, features = ["deflate"] } diff --git a/crates/tinydocs-module/src/outputs/mod.rs b/crates/tinydocs-module/src/outputs/mod.rs index b3b7cd2..c170fbe 100644 --- a/crates/tinydocs-module/src/outputs/mod.rs +++ b/crates/tinydocs-module/src/outputs/mod.rs @@ -32,7 +32,6 @@ use std::collections::HashMap; use std::sync::Mutex; use std::time::{Duration, Instant}; -use serde::{Deserialize, Serialize}; use sha2::{Digest, Sha256}; /// Largest chunk a caller may read in one `ReadOutput`. @@ -56,17 +55,7 @@ pub const MAX_LIVE_OUTPUTS: usize = 32; /// How long an output may go unread before it is dropped. pub const IDLE_TTL: Duration = Duration::from_secs(300); -/// A handle to a produced document, and what a caller needs to read it back. -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct OutputRef { - /// Opaque identifier, valid until read and released or expired. - pub output_id: String, - /// Total size in bytes, so a caller knows when it is done. - pub total_bytes: u64, - /// Lowercase hex SHA-256, so a caller can verify what it assembled. - pub sha256: String, -} +pub use tinydocs_bus::OutputRef; /// Why an output operation was refused. #[derive(Debug, Clone, PartialEq, Eq, thiserror::Error)] diff --git a/crates/tinydocs-module/src/service/mod.rs b/crates/tinydocs-module/src/service/mod.rs index 14325b8..76a2135 100644 --- a/crates/tinydocs-module/src/service/mod.rs +++ b/crates/tinydocs-module/src/service/mod.rs @@ -1,11 +1,13 @@ //! `TinyBus` service boundary for the document surface. //! -//! One object, `/ai/tinyhumans/tinydocs/Documents`, exporting five methods: +//! One object, `/ai/tinyhumans/tinydocs/Documents`, exporting seven methods: //! //! ```text //! GenerateDocx(DocumentSpec) -> OutputRef //! GeneratePptx(WirePresentationSpec, Option) -> OutputRef //! ExtractText(StreamRef) -> OutputRef +//! ExtractDocument(spec, StreamRef) -> ExtractedDocument +//! RenderPdf(spec, StreamRef) -> RenderedPdf //! ReadOutput(output_id, offset, len) -> base64 //! ReleaseOutput(output_id) -> () //! ``` @@ -55,7 +57,10 @@ use tinybus::{Connection, Error as BusError, Result as BusResult}; use tinydocs::spec::presentation::MAX_IMAGE_BYTES; use tinydocs::spec::{DocumentSpec, PresentationSpec, SlideImage, SlideSpec}; use tinydocs::{Error, pdf, pptx}; -use tinydocs_bus::{BUS_NAME, OBJECT_PATH}; +use tinydocs_bus::{ + BUS_NAME, ExtractDocumentSpec, ExtractedDocument, OBJECT_PATH, RenderPdfSpec, RenderedPdf, + RenderedPdfPage, +}; use crate::outputs::{OutputError, OutputRef, OutputStore}; @@ -118,6 +123,23 @@ impl Documents { self.hold(text.into_bytes()) } + /// Extract bounded document text and section provenance from a stream. + async fn extract_document( + &self, + spec: ExtractDocumentSpec, + document: StreamRef, + ) -> BusResult { + let bytes = self.read_stream(&document).await?; + blocking(move || tinydocs::intake::extract(&bytes, &spec)).await + } + + /// Render explicitly selected PDF pages and hold each PNG output. + async fn render_pdf(&self, spec: RenderPdfSpec, document: StreamRef) -> BusResult { + let bytes = self.read_stream(&document).await?; + let images = blocking(move || tinydocs::pdf_render::render(&bytes, &spec)).await?; + self.hold_images(images) + } + /// Read up to `len` bytes of a held document at `offset`, base64-encoded. async fn read_output(&self, output_id: String, offset: u64, len: u64) -> BusResult { let bytes = self @@ -136,6 +158,31 @@ impl Documents { } impl Documents { + /// Roll back every allocated output when a batch cannot be retained. + fn hold_images(&self, images: tinydocs::pdf_render::PdfImages) -> BusResult { + let mut result = RenderedPdf { + page_count: images.page_count, + pages: Vec::new(), + }; + for image in images.pages { + match self.hold(image.bytes) { + Ok(output) => result.pages.push(RenderedPdfPage { + page: image.page, + width: image.width, + height: image.height, + output, + }), + Err(error) => { + for page in result.pages { + let _ = self.outputs.release(&page.output.output_id, Instant::now()); + } + return Err(error); + } + } + } + Ok(result) + } + /// Hold a produced document and return its handle. fn hold(&self, bytes: Vec) -> BusResult { self.outputs @@ -340,6 +387,8 @@ pub(crate) mod exports { "GenerateDocx", "GeneratePptx", "ExtractText", + "ExtractDocument", + "RenderPdf", "ReadOutput", "ReleaseOutput", ], diff --git a/crates/tinydocs-module/src/service/mod_tests.rs b/crates/tinydocs-module/src/service/mod_tests.rs index 112a4a0..7c92091 100644 --- a/crates/tinydocs-module/src/service/mod_tests.rs +++ b/crates/tinydocs-module/src/service/mod_tests.rs @@ -22,6 +22,8 @@ const DECLARED_METHODS: &[&str] = &[ "GenerateDocx", "GeneratePptx", "ExtractText", + "ExtractDocument", + "RenderPdf", "ReadOutput", "ReleaseOutput", ]; @@ -272,3 +274,58 @@ async fn a_malformed_read_is_refused_by_name() { .expect_err("read past the end"); assert_eq!(err.wire_name(), TRANSFER_FAILED_ERROR); } + +#[tokio::test] +async fn rendered_pages_use_the_existing_output_lifecycle() { + let documents = service().await; + let result = documents + .hold_images(tinydocs::pdf_render::PdfImages { + page_count: 3, + pages: vec![tinydocs::pdf_render::PdfPageImage { + page: 2, + width: 8, + height: 9, + bytes: b"png bytes".to_vec(), + }], + }) + .unwrap(); + assert_eq!(result.pages[0].page, 2); + let output = &result.pages[0].output; + assert_eq!( + documents + .outputs + .read_chunk(&output.output_id, 0, output.total_bytes, Instant::now()) + .unwrap(), + b"png bytes" + ); + documents + .release_output(output.output_id.clone()) + .await + .unwrap(); + assert_eq!(documents.outputs.live_count(), 0); +} +#[tokio::test] +async fn a_refused_raster_batch_releases_every_partial_output() { + let documents = service().await; + for _ in 0..crate::outputs::MAX_LIVE_OUTPUTS - 1 { + documents.hold(vec![1]).unwrap(); + } + let image = || tinydocs::pdf_render::PdfPageImage { + page: 1, + width: 1, + height: 1, + bytes: vec![1], + }; + assert!( + documents + .hold_images(tinydocs::pdf_render::PdfImages { + page_count: 2, + pages: vec![image(), image()] + }) + .is_err() + ); + assert_eq!( + documents.outputs.live_count(), + crate::outputs::MAX_LIVE_OUTPUTS - 1 + ); +} diff --git a/crates/tinydocs-module/tests/module_e2e.rs b/crates/tinydocs-module/tests/module_e2e.rs index 0bbd834..0cf4daa 100644 --- a/crates/tinydocs-module/tests/module_e2e.rs +++ b/crates/tinydocs-module/tests/module_e2e.rs @@ -12,7 +12,7 @@ #![allow(clippy::unwrap_used, clippy::expect_used, clippy::panic)] -use std::time::Duration; +use std::{io::Write, time::Duration}; use base64::Engine as _; use tinybus::Connection; @@ -48,6 +48,7 @@ async fn the_built_module_serves_every_format_over_a_real_broker() { generates_a_docx(&proxy).await; generates_a_pptx_from_a_streamed_image_pair(&client, &target, &proxy).await; extracts_text_from_a_streamed_pdf(&client, &target, &proxy).await; + extracts_and_renders_document_intake(&client, &target, &proxy).await; refuses_a_stream_that_contradicts_the_spec(&client, &target).await; assert!(matches!(modules.list()[0].state, ModuleState::Ready)); @@ -226,6 +227,95 @@ async fn extracts_text_from_a_streamed_pdf( ); } +/// New intake methods preserve streamed input and held output lifecycles. +async fn extracts_and_renders_document_intake( + client: &Connection, + target: &Target, + proxy: &tinybus::Proxy, +) { + let docx = docx_with_text("Office intake through TinyBus"); + let spec = tinydocs_bus::ExtractDocumentSpec::new(tinydocs_bus::DocumentFormat::Docx); + let extracted_docx: tinydocs_bus::ExtractedDocument = client + .call_with_stream( + target.destination.clone(), + target.path.clone(), + target.interface.clone(), + tinybus::MemberName::new(methods::EXTRACT_DOCUMENT).unwrap(), + |stream| serde_json::json!([spec, stream]), + &docx, + ) + .await + .expect("DOCX intake should succeed through the module"); + assert_eq!(extracted_docx.format, tinydocs_bus::DocumentFormat::Docx); + assert_eq!(extracted_docx.section_count, 1); + assert_eq!(extracted_docx.sections[0].source, "word/document.xml"); + assert!( + extracted_docx.sections[0] + .text + .contains("Office intake through TinyBus") + ); + + let pdf = pdf_with_text("Intake provenance"); + let spec = tinydocs_bus::ExtractDocumentSpec::new(tinydocs_bus::DocumentFormat::Pdf); + let extracted: tinydocs_bus::ExtractedDocument = client + .call_with_stream( + target.destination.clone(), + target.path.clone(), + target.interface.clone(), + tinybus::MemberName::new(methods::EXTRACT_DOCUMENT).unwrap(), + |stream| serde_json::json!([spec, stream]), + &pdf, + ) + .await + .unwrap(); + assert_eq!(extracted.section_count, 1); + assert_eq!(extracted.sections[0].source, "page:1"); + assert!(extracted.sections[0].text.contains("Intake provenance")); + let spec = tinydocs_bus::RenderPdfSpec { + pages: vec![1], + max_dimension: 128, + max_total_pixels: 100_000, + max_output_bytes: 1_000_000, + }; + let rendered: tinydocs_bus::RenderedPdf = client + .call_with_stream( + target.destination.clone(), + target.path.clone(), + target.interface.clone(), + tinybus::MemberName::new(methods::RENDER_PDF).unwrap(), + |stream| serde_json::json!([spec, stream]), + &pdf, + ) + .await + .unwrap(); + assert_eq!(rendered.page_count, 1); + let png = download(proxy, &rendered.pages[0].output).await; + assert!(png.starts_with(b"\x89PNG\r\n\x1a\n")); + proxy + .call::<()>( + methods::RELEASE_OUTPUT, + (rendered.pages[0].output.output_id.clone(),), + ) + .await + .unwrap(); +} + +fn docx_with_text(text: &str) -> Vec { + let mut archive = zip::ZipWriter::new(std::io::Cursor::new(Vec::new())); + archive + .start_file( + "word/document.xml", + zip::write::SimpleFileOptions::default(), + ) + .unwrap(); + write!( + archive, + "{text}" + ) + .unwrap(); + archive.finish().unwrap().into_inner() +} + /// The lengths in the spec are the authority. /// /// A short stream must fail rather than produce a deck with a picture assembled diff --git a/docs/plans/document-intake-render.md b/docs/plans/document-intake-render.md new file mode 100644 index 0000000..7d0ba34 --- /dev/null +++ b/docs/plans/document-intake-render.md @@ -0,0 +1,51 @@ +# Document intake and selected PDF rendering implementation + +Status: Implemented. +Specification: [Document intake and selected PDF rendering](../specs/document-intake-render.md). + +This records the implementation sequence for the accepted document-intake work. +Hosts own upload storage, OCR/inference policy, cancellation, and stronger parser +isolation. TinyDocs owns supplied-byte parsing and its module contract. + +1. Define validated extraction/render specs, provenance results, and shared + `OutputRef` in `crates/tinydocs-bus/src/intake/`; declare additive member names + in `src/names.rs`. Add serialization and rejection tests before wiring I/O. +2. Add optional `intake` and `pdf-render` features in `Cargo.toml` and deliberate + exports in `src/lib.rs`. Preserve existing writer/extraction features. +3. Implement bounded text sinks, selected OOXML member reads, shared-string + resolution, and relationship ordering in `src/intake/mod.rs` and + `office_order.rs`. Add DOCX/PPTX/XLSX fixtures and malformed/budget tests in + their sibling test files. Test adversarial empty shared strings and output + capacity retention before adding allocation checks. +4. Preflight ZIP metadata in `src/intake/zip_admission.rs` before eager reader + allocation. Verify oversized counts, names, ZIP64 metadata, embedded footer + confusion, and unusual valid preambles with synthetic archives. +5. Extract PDF sections with bounded text sinks and scanned candidates. Build + deterministic text/raster fixtures in `src/pdf/fixtures.rs`; test mixed pages + and truncation without external files or network services. +6. Raster selected pages in `src/pdf_render/mod.rs`; validate page, edge, pixel, + and PNG output budgets. Test actual raster pixels rather than only headers. +7. Add streamed service handlers and output-store rollback in + `crates/tinydocs-module/src/service/` and `outputs/`. Extend + `tests/module_e2e.rs` to exercise the compiled module through TinyBus and + read/release generated PNGs. +8. Document APIs and parser limitations in module READMEs, the root README, + and the linked specification. Keep dependency advisories enabled; use patched + `quick-xml` and adapt attribute decoding through the reader's decoder. + +## Verification + +Run from the TinyDocs repository root: + +```sh +cargo fmt --all -- --check +cargo test --workspace --all-features +cargo clippy --workspace --all-targets --all-features -- -D warnings +cargo +1.88.0 check --workspace --all-features +cargo deny --all-features check all +``` + +The module E2E workflow builds and loads the cdylib before executing the bus +fixture suite. Each completed step has corresponding sibling regression tests; +publishing module artifacts and updating downstream release pins remain outside +this plan. diff --git a/docs/specs/document-intake-render.md b/docs/specs/document-intake-render.md new file mode 100644 index 0000000..fee3a2a --- /dev/null +++ b/docs/specs/document-intake-render.md @@ -0,0 +1,63 @@ +# Document intake and selected PDF rendering + +Status: Implemented. Owner: TinyDocs. + +## Problem + +Agent hosts need document text with page, slide, and worksheet provenance and +selected PDF page images when text extraction finds scanned pages. Intake must +operate on supplied bytes without extracting files or contacting a network. + +## Behavior + +`intake::extract(bytes, &ExtractDocumentSpec)` accepts PDF, DOCX, PPTX, and XLSX. +The result reports total section count, ordered source labels, bounded text, +truncation, and PDF scanned-page candidates. Blank PDF text is a candidate for +host-side analysis, not a determination that OCR will succeed. Office ordering +follows presentation/workbook relationships. XLSX supports shared strings, +inline strings, and cached values; phonetic annotations are excluded from cell +values. Explicit Office breaks and tabs preserve word boundaries. It does not +calculate formulas. + +`pdf_render::render(bytes, &RenderPdfSpec)` renders explicitly selected 1-based +PDF pages to PNG with width and height metadata. It does not perform OCR or +choose pages. Original bytes remain the host's responsibility. + +The TinyBus `ExtractDocument` member accepts the extraction spec and input stream +and returns bounded inline section text. `RenderPdf` returns page metadata and +held `OutputRef` values, read and released through existing output-store members. +Invalid arguments, malformed or encrypted documents, unsupported structures, +and exceeded hard budgets return typed domain errors. Earlier module members +remain compatible; the additive contract remains version 2. + +## Constraints + +- Inputs are nonempty and at most 64 MiB. +- Extraction allows at most 200,000 text bytes and 256 returned sections. +- OOXML allows 2,048 members, 256-byte member names, 8 MiB per XML part, + 32 MiB total expansion, and XML nesting depth 128. ZIP metadata admission + runs before constructing the eager ZIP reader. DTDs are rejected. +- Shared strings allow at most 65,536 records and 4 MiB decoded text. Record + count and text checks precede retained allocation. Returned short sections + release unused reserved text capacity. +- PDFs allow at most 4,096 pages; rendering selects at most 8 pages, bounds the + maximum edge to 2,048 pixels, total pixels to 16 million, and encoded output + to 32 MiB. +- The native module runs in-process. These input and output budgets do not + bound all PDF parser internals. A host requiring parser isolation must use + a separate process; a task timeout alone cannot terminate blocking parsing. +- Library features `intake` and `pdf-render` are opt-in. The module enables + them. No archive extraction, networking, OCR, or provider inference occurs. + +## Acceptance criteria + +Fixture tests verify ordered Office sections, shared/inline/cached worksheet +values, mixed text/scanned PDFs, raster pixels and PNG dimensions, truncation, +malformed XML, hostile ZIP metadata, and all documented budget rejections. +Bus tests verify schema validation, streamed input, held output reads/releases, +and rollback when a rendering output cannot be retained. MSRV is Rust 1.88. + +## Open questions + +None blocking the library contract. Publishing the updated module is a separate +release operation; hosts must not call new members on an older pinned artifact. diff --git a/src/intake/README.md b/src/intake/README.md new file mode 100644 index 0000000..8ab78be --- /dev/null +++ b/src/intake/README.md @@ -0,0 +1,45 @@ +# Document intake + +`extract(bytes, &ExtractDocumentSpec)` accepts PDF, DOCX, PPTX and XLSX bytes. +Each section retains its source page or OOXML part path. Empty PDF text layers +are retained and marked `scanned_candidate`; that flag is a suggestion for host +vision, not a determination that the page contains an image or needs OCR. + +The library accepts at most 64 MiB of input, 2,048 ZIP members, 32 MiB of declared +and read ZIP expansion, 8 MiB per member, 256-byte member names, and 128 nested +XML elements. A bounded metadata admission pass checks member counts, decoded +names and at most 1 MiB each of central-directory and ZIP64 footer metadata +before constructing the eager ZIP index. Raw duplicate names are rejected in +a set capped at 2,048 borrowed names, before the ZIP library can deduplicate +them. Alternative footer signatures in +metadata are rejected conservatively; signatures in member payloads are hidden +only while indexing, so the parser cannot fall back to an unadmitted archive. +DTDs, unresolved entities, encrypted ZIP/PDF documents, duplicate member names +and malformed structures fail explicitly. ZIP contents are never +written to disk. Only document body/slide/worksheet parts are returned; archive +listing is outside this module. + +The result has at most 256 sections and 200,000 UTF-8 text bytes, with explicit +`truncated` and total `section_count`. Source labels are separately bounded. +Caller budgets may narrow these ceilings. Unicode is never cut in the middle +of a code point. Office text is bounded while XML is parsed, including every +resolved shared-string append; repeating a large XLSX string cannot amplify the +output allocation beyond the remaining caller budget. Parsing continues after +truncation to validate the selected XML. + +PPTX presentation and XLSX workbook manifests determine section order through +their relationships. Unreferenced slide/sheet parts are excluded. Missing, +external, duplicate or unsupported references fail explicitly, as do missing +manifests; there is no filename-order fallback. Relationship paths are resolved +inside the package and retain the actual source part path. XLSX resolves shared +and inline strings and uses cached cell values rather than evaluating formulas; +paths identify sheets without inventing workbook display names. + +PDF text is extracted page by page with a bounded output writer. At most 4,096 +source pages are accepted. The PDF parser may allocate for objects, fonts and +expanded streams internally: the input/page/output limits are **not a total +parser memory or CPU bound**. Calls are synchronous. A native TinyBus module +runs in the host process; a Tokio timeout does not stop a blocking parser. +Hosts own deadline/cancellation policy and any stronger isolation they need. + +No network OCR, office application, shell process or filesystem read is used. diff --git a/src/intake/mod.rs b/src/intake/mod.rs new file mode 100644 index 0000000..5d1620b --- /dev/null +++ b/src/intake/mod.rs @@ -0,0 +1,462 @@ +//! Bounded document text intake without filesystem or network access. +//! +//! OOXML reads selected ZIP members and parses XML events. PDF parsing is +//! CPU-bound; hosts own execution deadlines and any stronger isolation. +use crate::{Error, Result}; +use quick_xml::{Reader, events::Event}; +use std::borrow::Cow; +use std::io::Read; +mod office_order; +mod zip_admission; +pub use tinydocs_bus::intake::{ + DocumentFormat, DocumentSectionText, ExtractDocumentSpec, ExtractedDocument, +}; + +const MAX_INPUT: usize = 64 * 1024 * 1024; +const MAX_EXPANDED: u64 = 32 * 1024 * 1024; +const MAX_PART: u64 = 8 * 1024 * 1024; +const MAX_ENTRIES: usize = 2048; +const MAX_SHARED_STRINGS: usize = 65_536; +const MAX_SHARED_TEXT_BYTES: usize = 4 * 1024 * 1024; + +/// Extract bounded visible text with document/page/slide/worksheet provenance. +/// +/// # Errors +/// Rejects invalid budgets, malformed/encrypted documents, XML entities/DTDs, +/// excessive ZIP expansion, excessive XML nesting and unsupported structures. +pub fn extract(bytes: &[u8], spec: &ExtractDocumentSpec) -> Result { + if bytes.is_empty() || bytes.len() > MAX_INPUT { + return Err(Error::invalid_input( + "bytes", + "document is empty or exceeds 64 MiB", + )); + } + if spec.max_text_bytes == 0 + || spec.max_text_bytes > 200_000 + || spec.max_sections == 0 + || spec.max_sections > 256 + { + return Err(Error::invalid_input( + "spec", + "text/section budget outside hard bounds", + )); + } + if spec.format == DocumentFormat::Pdf { + return extract_pdf(bytes, spec); + } + extract_office(bytes, spec) +} +fn extract_office(bytes: &[u8], spec: &ExtractDocumentSpec) -> Result { + zip_admission::zip_preflight(bytes)?; + let mut archive = zip_admission::open_admitted_zip(bytes)?; + let mut expanded_read = 0u64; + let parts = office_parts(&mut archive, spec.format, &mut expanded_read)?; + let shared = if spec.format == DocumentFormat::Xlsx { + match archive.by_name("xl/sharedStrings.xml") { + Ok(part) => xml_strings(&read_part(part, &mut expanded_read)?)?, + Err(zip::result::ZipError::FileNotFound) => Vec::new(), + Err(error) => return Err(failed(error)), + } + } else { + Vec::new() + }; + let mut result = ExtractedDocument { + format: spec.format, + section_count: u32::try_from(parts.len()).map_err(failed)?, + sections: Vec::new(), + truncated: parts.len() > spec.max_sections as usize, + }; + let mut remaining = spec.max_text_bytes as usize; + for (index, (name, part_index)) in parts + .into_iter() + .take(spec.max_sections as usize) + .enumerate() + { + let xml = read_part( + archive.by_index(part_index).map_err(failed)?, + &mut expanded_read, + )?; + let text = xml_text(&xml, &shared, remaining)?; + remaining -= text.text.len(); + result.truncated |= text.truncated; + result.sections.push(DocumentSectionText { + source: name, + index: u32::try_from(index + 1).map_err(failed)?, + text: text.text, + scanned_candidate: false, + }); + } + Ok(result) +} +fn read_part(part: impl Read, expanded: &mut u64) -> Result> { + let mut xml = Vec::new(); + part.take(MAX_PART + 1) + .read_to_end(&mut xml) + .map_err(failed)?; + *expanded = expanded.saturating_add(xml.len() as u64); + if xml.len() as u64 > MAX_PART || *expanded > MAX_EXPANDED { + return Err(Error::extraction_failed("XML expansion limit exceeded")); + } + Ok(xml) +} +fn office_parts( + archive: &mut zip::ZipArchive>, + format: DocumentFormat, + expanded_read: &mut u64, +) -> Result> { + if archive.len() > MAX_ENTRIES { + return Err(Error::extraction_failed("too many ZIP members")); + } + let mut expanded = 0u64; + let mut parts = Vec::new(); + for i in 0..archive.len() { + let part = archive.by_index(i).map_err(failed)?; + expanded = expanded + .checked_add(part.size()) + .ok_or_else(|| Error::extraction_failed("ZIP size overflow"))?; + if expanded > MAX_EXPANDED || part.size() > MAX_PART || part.encrypted() { + return Err(Error::extraction_failed( + "encrypted ZIP or expansion limit exceeded", + )); + } + let name = part.name(); + if name.len() > 256 { + return Err(Error::extraction_failed("ZIP member name limit exceeded")); + } + let selected = match format { + DocumentFormat::Docx => name == "word/document.xml", + DocumentFormat::Pptx | DocumentFormat::Xlsx => true, + DocumentFormat::Pdf => false, + }; + if selected { + if parts.iter().any(|(existing, _)| existing == name) { + return Err(Error::extraction_failed("duplicate document part")); + } + parts.push((name.to_owned(), i)); + } + } + if matches!(format, DocumentFormat::Pptx | DocumentFormat::Xlsx) { + parts = office_order::ordered_parts(archive, format, &parts, expanded_read)?; + } + if parts.is_empty() { + return Err(Error::extraction_failed("document has no supported parts")); + } + Ok(parts) +} +fn failed(error: impl std::fmt::Display) -> Error { + Error::extraction_failed(&error.to_string()) +} +pub(super) fn normalize_xml(xml: &[u8]) -> Result> { + let (encoding, content) = if xml.starts_with(&[0xFF, 0xFE]) { + (Some(false), &xml[2..]) + } else if xml.starts_with(&[0xFE, 0xFF]) { + (Some(true), &xml[2..]) + } else if xml.starts_with(&[b'<', 0, b'?', 0]) { + (Some(false), xml) + } else if xml.starts_with(&[0, b'<', 0, b'?']) { + (Some(true), xml) + } else { + (None, xml) + }; + let Some(big_endian) = encoding else { + return Ok(Cow::Borrowed( + xml.strip_prefix(&[0xEF, 0xBB, 0xBF]).unwrap_or(xml), + )); + }; + if content.len() % 2 != 0 { + return Err(Error::extraction_failed("invalid UTF-16 XML length")); + } + let (pairs, _) = content.as_chunks::<2>(); + let units = pairs.iter().map(|pair| { + if big_endian { + u16::from_be_bytes([pair[0], pair[1]]) + } else { + u16::from_le_bytes([pair[0], pair[1]]) + } + }); + let mut utf8 = String::new(); + for character in char::decode_utf16(units) { + utf8.push(character.map_err(failed)?); + } + if utf8.starts_with("") + .ok_or_else(|| Error::extraction_failed("invalid XML declaration"))?; + utf8.drain(..declaration_end + 2); + } + Ok(Cow::Owned(utf8.into_bytes())) +} +fn xml_strings(xml: &[u8]) -> Result> { + let xml = normalize_xml(xml)?; + let mut result = Vec::new(); + let mut current = String::new(); + let mut text_bytes = 0usize; + xml_events(xml.as_ref(), |event, text| { + if let Some(value) = text { + text_bytes = text_bytes + .checked_add(value.len()) + .ok_or_else(|| failed("shared string size overflow"))?; + if text_bytes > MAX_SHARED_TEXT_BYTES { + return Err(failed("shared string text budget exceeded")); + } + current.push_str(value); + } + if event == "si" { + if result.len() >= MAX_SHARED_STRINGS { + return Err(failed("shared string count budget exceeded")); + } + current.shrink_to_fit(); + result.push(std::mem::take(&mut current)); + } + Ok(()) + })?; + Ok(result) +} +fn xml_text(xml: &[u8], shared: &[String], limit: usize) -> Result { + let xml = normalize_xml(xml)?; + let mut output = TextSink { + text: String::with_capacity(limit), + limit, + truncated: false, + }; + let mut shared_cell = false; + let mut value = false; + let mut current = String::new(); + let mut reader = Reader::from_reader(xml.as_ref()); + let mut depth = 0usize; + let mut in_text = false; + let mut phonetic_depth = None; + loop { + match reader.read_event().map_err(failed)? { + Event::Start(e) => { + depth += 1; + if depth > 128 { + return Err(Error::extraction_failed("XML nesting limit exceeded")); + } + let local = e.local_name(); + if local.as_ref() == b"rPh" && phonetic_depth.is_none() { + phonetic_depth = Some(depth); + } + in_text = local.as_ref() == b"t" && phonetic_depth.is_none(); + if phonetic_depth.is_none() { + match local.as_ref() { + b"br" | b"cr" => output.append("\n"), + b"tab" => output.append("\t"), + _ => {} + } + } + value = local.as_ref() == b"v"; + if local.as_ref() == b"c" { + shared_cell = false; + for attr in e.attributes() { + let attr = attr.map_err(failed)?; + if attr.key.as_ref() == b"t" && attr.value.as_ref() == b"s" { + shared_cell = true; + } + } + } + } + Event::Text(e) if in_text || value => { + let decoded = e.decode().map_err(failed)?; + output.append_value(&mut current, value && shared_cell, &decoded)?; + } + Event::GeneralRef(e) if in_text || value => { + let name = e.decode().map_err(failed)?; + let decoded = decode_reference(&name)?; + output.append_value(&mut current, value && shared_cell, &decoded)?; + } + Event::CData(e) if in_text || value => { + let decoded = e.decode().map_err(failed)?; + output.append_value(&mut current, value && shared_cell, &decoded)?; + } + Event::End(e) => { + if phonetic_depth == Some(depth) { + phonetic_depth = None; + } + depth = depth + .checked_sub(1) + .ok_or_else(|| Error::extraction_failed("invalid XML nesting"))?; + let local = e.local_name(); + if local.as_ref() == b"t" || local.as_ref() == b"v" { + if value && shared_cell { + let index: usize = current.parse().map_err(failed)?; + output.append(shared.get(index).ok_or_else(|| { + Error::extraction_failed("invalid shared string index") + })?); + } + current.clear(); + in_text = false; + value = false; + } + if [b"p".as_slice(), b"row", b"c"].contains(&local.as_ref()) { + output.append("\n"); + } + } + Event::Empty(e) if phonetic_depth.is_none() => match e.local_name().as_ref() { + b"br" | b"cr" => output.append("\n"), + b"tab" => output.append("\t"), + _ => {} + }, + Event::DocType(_) => { + return Err(Error::extraction_failed("XML DTDs are not supported")); + } + Event::Eof => { + if depth != 0 { + return Err(Error::extraction_failed("unclosed XML elements")); + } + break; + } + _ => {} + } + } + output.text.shrink_to_fit(); + Ok(output) +} +fn append_index(current: &mut String, value: &str) -> Result<()> { + if current.len().saturating_add(value.len()) > 20 { + return Err(Error::extraction_failed("shared string index is too long")); + } + current.push_str(value); + Ok(()) +} +fn decode_reference(name: &str) -> Result { + quick_xml::escape::unescape(&format!("&{name};")) + .map(std::borrow::Cow::into_owned) + .map_err(failed) +} +fn xml_events(xml: &[u8], mut consume: impl FnMut(&str, Option<&str>) -> Result<()>) -> Result<()> { + let xml = normalize_xml(xml)?; + let mut reader = Reader::from_reader(xml.as_ref()); + let mut depth = 0usize; + let mut in_text = false; + let mut phonetic_depth = None; + loop { + match reader.read_event().map_err(failed)? { + Event::Start(e) => { + depth += 1; + if depth > 128 { + return Err(Error::extraction_failed("XML nesting limit exceeded")); + } + let local = e.local_name(); + if local.as_ref() == b"rPh" && phonetic_depth.is_none() { + phonetic_depth = Some(depth); + } + in_text = local.as_ref() == b"t" && phonetic_depth.is_none(); + } + Event::End(e) => { + if phonetic_depth == Some(depth) { + phonetic_depth = None; + } + depth = depth + .checked_sub(1) + .ok_or_else(|| Error::extraction_failed("invalid XML nesting"))?; + consume( + std::str::from_utf8(e.local_name().as_ref()).map_err(failed)?, + None, + )?; + in_text = false; + } + Event::Text(e) if in_text => consume("", Some(&e.decode().map_err(failed)?))?, + Event::CData(e) if in_text => consume("", Some(&e.decode().map_err(failed)?))?, + Event::Empty(e) if e.local_name().as_ref() == b"si" => consume("si", None)?, + Event::GeneralRef(e) if in_text => { + consume("", Some(&decode_reference(&e.decode().map_err(failed)?)?))?; + } + Event::DocType(_) => { + return Err(Error::extraction_failed("XML DTDs are not supported")); + } + Event::Eof => { + if depth != 0 { + return Err(Error::extraction_failed("unclosed XML elements")); + } + break; + } + _ => {} + } + } + Ok(()) +} +fn extract_pdf(bytes: &[u8], spec: &ExtractDocumentSpec) -> Result { + if !bytes.starts_with(b"%PDF-") { + return Err(Error::invalid_input("bytes", "missing PDF signature")); + } + let document = pdf_extract::Document::load_mem(bytes).map_err(failed)?; + if document.is_encrypted() || document.trailer.has(b"Encrypt") { + return Err(Error::extraction_failed("encrypted PDF is not supported")); + } + let pages = document.get_pages(); + if pages.len() > 4096 { + return Err(Error::extraction_failed("PDF page limit exceeded")); + } + let mut result = ExtractedDocument { + format: DocumentFormat::Pdf, + section_count: u32::try_from(pages.len()).map_err(failed)?, + sections: Vec::new(), + truncated: pages.len() > spec.max_sections as usize, + }; + let mut remaining = spec.max_text_bytes as usize; + for page in pages.keys().take(spec.max_sections as usize) { + let mut sink = TextSink { + text: String::new(), + limit: remaining, + truncated: false, + }; + let writer: &mut dyn std::io::Write = &mut sink; + pdf_extract::output_doc_page( + &document, + &mut pdf_extract::PlainTextOutput::new(writer), + *page, + ) + .map_err(failed)?; + remaining -= sink.text.len(); + result.truncated |= sink.truncated; + let scanned_candidate = sink.text.trim().is_empty() && !sink.truncated; + result.sections.push(DocumentSectionText { + source: format!("page:{page}"), + index: *page, + text: sink.text, + scanned_candidate, + }); + } + Ok(result) +} +struct TextSink { + text: String, + limit: usize, + truncated: bool, +} +impl TextSink { + fn append_value(&mut self, index: &mut String, shared: bool, text: &str) -> Result<()> { + if shared { + append_index(index, text) + } else { + self.append(text); + Ok(()) + } + } + fn append(&mut self, text: &str) { + if self.truncated { + return; + } + let remaining = self.limit.saturating_sub(self.text.len()); + let mut end = text.len().min(remaining); + while !text.is_char_boundary(end) { + end -= 1; + } + self.truncated |= end < text.len(); + self.text.push_str(&text[..end]); + } +} +impl std::io::Write for TextSink { + fn write(&mut self, bytes: &[u8]) -> std::io::Result { + let text = std::str::from_utf8(bytes).map_err(std::io::Error::other)?; + self.append(text); + Ok(bytes.len()) + } + fn flush(&mut self) -> std::io::Result<()> { + Ok(()) + } +} +#[cfg(test)] +#[path = "mod_tests.rs"] +mod tests; diff --git a/src/intake/mod_tests.rs b/src/intake/mod_tests.rs new file mode 100644 index 0000000..d4fbdf6 --- /dev/null +++ b/src/intake/mod_tests.rs @@ -0,0 +1,490 @@ +//! Bounded OOXML intake behavior. +#![allow(clippy::unwrap_used, clippy::expect_used, clippy::panic)] +use super::*; +use std::io::Write; +fn archive(parts: &[(&str, &str)]) -> Vec { + let mut zip = zip::ZipWriter::new(std::io::Cursor::new(Vec::new())); + for (name, xml) in parts { + zip.start_file(*name, zip::write::SimpleFileOptions::default()) + .unwrap(); + zip.write_all(xml.as_bytes()).unwrap(); + } + for (prefix, manifest, rels, root, container, element, kind) in [ + ( + "ppt/slides/slide", + "ppt/presentation.xml", + "ppt/_rels/presentation.xml.rels", + "presentation", + "sldIdLst", + "sldId", + "slide", + ), + ( + "xl/worksheets/sheet", + "xl/workbook.xml", + "xl/_rels/workbook.xml.rels", + "workbook", + "sheets", + "sheet", + "worksheet", + ), + ] { + let mut numbered: Vec<_> = parts + .iter() + .filter_map(|(name, _)| { + name.strip_prefix(prefix) + .and_then(|value| value.strip_suffix(".xml")) + .and_then(|value| value.parse::().ok()) + .map(|n| (n, *name)) + }) + .collect(); + numbered.sort_unstable(); + if numbered.is_empty() || parts.iter().any(|(name, _)| *name == manifest) { + continue; + } + let mut items = String::new(); + for (n, _) in &numbered { + use std::fmt::Write as _; + write!(&mut items, "<{element} r:id='r{n}'/>").unwrap(); + } + let xml = format!("<{root} xmlns:r='r'><{container}>{items}"); + zip.start_file(manifest, zip::write::SimpleFileOptions::default()) + .unwrap(); + zip.write_all(xml.as_bytes()).unwrap(); + let mut relationships = String::new(); + for (n, name) in &numbered { + use std::fmt::Write as _; + write!(&mut relationships, "", name.split_once('/').unwrap().1).unwrap(); + } + zip.start_file(rels, zip::write::SimpleFileOptions::default()) + .unwrap(); + zip.write_all(format!("{relationships}").as_bytes()) + .unwrap(); + } + zip.finish().unwrap().into_inner() +} +#[test] +fn extracts_docx_visible_text_and_provenance() { + let bytes = archive(&[( + "word/document.xml", + "Hello & world", + )]); + let result = extract(&bytes, &ExtractDocumentSpec::new(DocumentFormat::Docx)).unwrap(); + assert_eq!(result.sections[0].source, "word/document.xml"); + assert!(result.sections[0].text.contains("Hello & world")); +} +#[test] +fn truncates_text_without_splitting_unicode() { + let bytes = archive(&[("word/document.xml", "ééé")]); + let mut spec = ExtractDocumentSpec::new(DocumentFormat::Docx); + spec.max_text_bytes = 3; + let result = extract(&bytes, &spec).unwrap(); + assert!(result.truncated); + assert_eq!(result.sections[0].text, "é"); +} + +#[test] +fn extracts_slides_in_manifest_order_and_excludes_metadata() { + let bytes = archive(&[ + ("ppt/slides/slide10.xml", "ten"), + ("ppt/slides/slide2.xml", "two"), + ( + "docProps/core.xml", + "private metadata", + ), + ]); + let result = extract(&bytes, &ExtractDocumentSpec::new(DocumentFormat::Pptx)).unwrap(); + assert_eq!(result.section_count, 2); + assert_eq!(result.sections[0].text, "two"); + assert_eq!(result.sections[1].text, "ten"); +} +#[test] +fn resolves_xlsx_shared_inline_and_numeric_cell_values() { + let bytes = archive(&[ + ( + "xl/sharedStrings.xml", + "Hello & world", + ), + ( + "xl/worksheets/sheet1.xml", + "0Inline42", + ), + ]); + let result = extract(&bytes, &ExtractDocumentSpec::new(DocumentFormat::Xlsx)).unwrap(); + assert!(result.sections[0].text.contains("Hello & world")); + assert!(result.sections[0].text.contains("Inline")); + assert!(result.sections[0].text.contains("42")); +} +#[test] +fn preserves_text_scanned_and_mixed_pdf_page_provenance() { + let bytes = + crate::pdf::fixtures::document(&[("text", false), ("", true), ("mixed", true)], false); + let result = extract(&bytes, &ExtractDocumentSpec::new(DocumentFormat::Pdf)).unwrap(); + assert_eq!(result.section_count, 3); + assert!(result.sections[0].text.contains("text")); + assert!(result.sections[1].scanned_candidate); + assert!(!result.sections[2].scanned_candidate); + assert_eq!(result.sections[2].source, "page:3"); +} +#[test] +fn pdf_text_and_section_budgets_are_explicit() { + let bytes = crate::pdf::fixtures::document(&[("abcdef", false), ("second", false)], false); + let mut spec = ExtractDocumentSpec::new(DocumentFormat::Pdf); + spec.max_sections = 1; + spec.max_text_bytes = 3; + let result = extract(&bytes, &spec).unwrap(); + assert!(result.truncated); + assert_eq!(result.section_count, 2); + assert_eq!(result.sections.len(), 1); + assert!(result.sections[0].text.len() <= 3); +} +#[test] +fn rejects_bad_documents_budgets_and_xml_entities() { + let mut spec = ExtractDocumentSpec::new(DocumentFormat::Docx); + assert!(extract(b"not zip", &spec).is_err()); + for xml in [ + "]>&a;", + "unclosed", + "&unknown;", + ] { + assert!(extract(&archive(&[("word/document.xml", xml)]), &spec).is_err()); + } + let deep = "

".repeat(129) + &"

".repeat(129); + assert!(extract(&archive(&[("word/document.xml", &deep)]), &spec).is_err()); + spec.max_text_bytes = 200_001; + assert!(extract(b"zip", &spec).is_err()); + assert!( + extract( + &archive(&[("other.xml", "ignore")]), + &ExtractDocumentSpec::new(DocumentFormat::Docx) + ) + .is_err() + ); +} +#[test] +fn rejects_zip_expansion_and_member_count_limits() { + let expanded = "a".repeat(usize::try_from(MAX_PART).unwrap() + 1); + assert!( + extract( + &archive(&[("word/document.xml", &expanded)]), + &ExtractDocumentSpec::new(DocumentFormat::Docx) + ) + .is_err() + ); + let names: Vec = (0..=MAX_ENTRIES) + .map(|index| format!("part{index}")) + .collect(); + let parts: Vec<(&str, &str)> = names.iter().map(|name| (name.as_str(), "")).collect(); + assert!( + extract( + &archive(&parts), + &ExtractDocumentSpec::new(DocumentFormat::Docx) + ) + .is_err() + ); +} +#[test] +fn rejects_encrypted_and_malformed_pdf() { + let spec = ExtractDocumentSpec::new(DocumentFormat::Pdf); + assert!( + extract( + &crate::pdf::fixtures::document(&[("secret", false)], true), + &spec + ) + .is_err() + ); + assert!(extract(b"%PDF-broken", &spec).is_err()); + assert!(extract(b"not pdf", &spec).is_err()); +} + +#[test] +fn preserves_cdata_empty_shared_strings_and_missing_shared_table() { + let spec = ExtractDocumentSpec::new(DocumentFormat::Xlsx); + let bytes = archive(&[ + ( + "xl/sharedStrings.xml", + "", + ), + ( + "xl/worksheets/sheet1.xml", + "01", + ), + ]); + let result = extract(&bytes, &spec).unwrap(); + assert!(result.sections[0].text.contains("a < b")); + let inline = archive(&[( + "xl/worksheets/sheet1.xml", + "", + )]); + assert!( + extract(&inline, &spec).unwrap().sections[0] + .text + .contains("raw < text") + ); + let invalid_index = archive(&[( + "xl/worksheets/sheet1.xml", + "42", + )]); + assert!(extract(&invalid_index, &spec).is_err()); +} +#[test] +fn rejects_long_member_names_empty_inputs_and_actual_expansion() { + let spec = ExtractDocumentSpec::new(DocumentFormat::Docx); + assert!(extract(&[], &spec).is_err()); + assert!(extract(&archive(&[(&"x".repeat(257), "")]), &spec).is_err()); + let mut expanded = MAX_EXPANDED; + assert!(read_part(std::io::Cursor::new(b"x"), &mut expanded).is_err()); + let mut expanded = 0; + assert!(read_part(std::io::repeat(b'x').take(MAX_PART + 1), &mut expanded).is_err()); +} +#[test] +fn rejects_malformed_shared_string_xml() { + let spec = ExtractDocumentSpec::new(DocumentFormat::Xlsx); + for xml in [ + "", + "unclosed", + "&undefined;", + ] { + let bytes = archive(&[ + ("xl/sharedStrings.xml", xml), + ("xl/worksheets/sheet1.xml", ""), + ]); + assert!(extract(&bytes, &spec).is_err()); + } + let deep = "".repeat(129) + &"".repeat(129); + assert!( + extract( + &archive(&[ + ("xl/sharedStrings.xml", &deep), + ("xl/worksheets/sheet1.xml", "") + ]), + &spec + ) + .is_err() + ); +} + +#[test] +fn zip_member_limits_are_rejected_by_admission_before_eager_indexing() { + let names: Vec<_> = (0..=MAX_ENTRIES).map(|i| format!("part{i}")).collect(); + let parts: Vec<_> = names.iter().map(|name| (name.as_str(), "")).collect(); + let error = extract( + &archive(&parts), + &ExtractDocumentSpec::new(DocumentFormat::Docx), + ) + .unwrap_err(); + assert!(error.to_string().contains("ZIP admission count limit")); + let error = extract( + &archive(&[(&"x".repeat(257), "")]), + &ExtractDocumentSpec::new(DocumentFormat::Docx), + ) + .unwrap_err(); + assert!(error.to_string().contains("ZIP admission name limit")); +} + +#[test] +fn worksheet_expansion_is_bounded_while_resolving_repeated_shared_strings() { + let shared = vec!["é".repeat(50_000)]; + let xml = format!( + "{}", + "0".repeat(100) + ); + let text = xml_text(xml.as_bytes(), &shared, 31).unwrap(); + assert!( + text.text.len() <= 31, + "shared strings must be bounded before appending" + ); + assert!(text.truncated); + assert!(text.text.capacity() <= 31); + let strings = format!("{}", shared[0]); + let bytes = archive(&[ + ("xl/sharedStrings.xml", &strings), + ("xl/worksheets/sheet1.xml", &xml), + ]); + let mut spec = ExtractDocumentSpec::new(DocumentFormat::Xlsx); + spec.max_text_bytes = 31; + let result = extract(&bytes, &spec).unwrap(); + assert!(result.truncated); + assert_eq!(result.section_count, 1); + assert_eq!(result.sections[0].text, "é".repeat(15)); +} + +#[test] +fn office_manifest_order_excludes_orphans_and_preserves_part_provenance() { + let bytes = archive(&[ + ( + "ppt/presentation.xml", + "", + ), + ( + "ppt/_rels/presentation.xml.rels", + "", + ), + ("ppt/slides/slide1.xml", "first filename"), + ("ppt/slides/slide2.xml", "first displayed"), + ("ppt/slides/slide3.xml", "orphan"), + ]); + let result = extract(&bytes, &ExtractDocumentSpec::new(DocumentFormat::Pptx)).unwrap(); + assert_eq!(result.section_count, 2); + assert_eq!(result.sections[0].source, "ppt/slides/slide2.xml"); + assert_eq!(result.sections[0].text, "first displayed"); + assert_eq!(result.sections[1].source, "ppt/slides/slide1.xml"); +} + +#[test] +fn workbook_order_uses_relationships_instead_of_names_and_ignores_orphans() { + let bytes = archive(&[ + ( + "xl/workbook.xml", + "", + ), + ( + "xl/_rels/workbook.xml.rels", + "", + ), + ("xl/worksheets/alpha.xml", "A"), + ("xl/worksheets/beta.xml", "B"), + ( + "xl/worksheets/sheet3.xml", + "orphan", + ), + ]); + let mut spec = ExtractDocumentSpec::new(DocumentFormat::Xlsx); + spec.max_sections = 1; + let result = extract(&bytes, &spec).unwrap(); + assert_eq!(result.section_count, 2); + assert!(result.truncated); + assert_eq!(result.sections[0].source, "xl/worksheets/beta.xml"); + assert_eq!(result.sections[0].index, 1); + assert_eq!(result.sections[0].text, "B"); +} + +#[test] +fn invalid_document_relationships_and_manifests_fail_explicitly() { + let manifest = ""; + for relationships in [ + "", + "", + "", + "", + "", + "", + "", + ] { + let bytes = archive(&[ + ("xl/workbook.xml", manifest), + ("xl/_rels/workbook.xml.rels", relationships), + ("xl/worksheets/sheet1.xml", ""), + ]); + assert!(extract(&bytes, &ExtractDocumentSpec::new(DocumentFormat::Xlsx)).is_err()); + } + for manifest in [ + "", + "", + "", + ] { + let bytes = archive(&[ + ("xl/workbook.xml", manifest), + ( + "xl/_rels/workbook.xml.rels", + "", + ), + ("xl/worksheets/sheet1.xml", ""), + ]); + assert!(extract(&bytes, &ExtractDocumentSpec::new(DocumentFormat::Xlsx)).is_err()); + } +} + +#[test] +fn budget_exhaustion_still_validates_xml_and_shared_string_indices() { + let shared = vec!["large output".repeat(1000)]; + for xml in [ + "0&undefined;", + "0999999999999999999999", + "042", + "0", + ] { + assert!(xml_text(xml.as_bytes(), &shared, 1).is_err(), "{xml}"); + } + let output = xml_text(b"abcdef&", &[], 3).unwrap(); + assert_eq!(output.text, "abc"); + assert!(output.truncated); + assert_eq!(output.text.capacity(), 3); +} + +#[test] +fn shared_strings_enforce_count_and_allocation_budgets_before_push() { + let too_many = format!("{}", "".repeat(65_537)); + assert!(xml_strings(too_many.as_bytes()).is_err()); + let too_large = format!( + "{}", + "a".repeat(4 * 1024 * 1024 + 1) + ); + assert!(xml_strings(too_large.as_bytes()).is_err()); + let accepted = xml_strings(b"ok").unwrap(); + assert_eq!(accepted, vec!["", "ok"]); +} + +#[test] +fn short_section_does_not_retain_unused_output_budget() { + for xml in [b"".as_slice(), b"ok".as_slice()] { + let text = xml_text(xml, &[], 200_000).unwrap(); + assert_eq!(text.text.capacity(), text.text.len()); + } +} + +#[test] +fn decodes_utf16_ooxml_parts_before_parsing() { + for (bom, encode) in [ + (&[0xFF, 0xFE][..], u16::to_le_bytes as fn(u16) -> [u8; 2]), + (&[0xFE, 0xFF][..], u16::to_be_bytes as fn(u16) -> [u8; 2]), + ] { + let source = "Résumé"; + let mut xml = bom.to_vec(); + for unit in source.encode_utf16() { + xml.extend_from_slice(&encode(unit)); + } + let shared = xml_strings(&xml).unwrap(); + assert_eq!(shared, vec!["Résumé"]); + assert_eq!( + xml_text(b"0", &shared, 1024) + .unwrap() + .text, + "Résumé\n" + ); + } +} + +#[test] +fn explicit_office_breaks_and_tabs_preserve_word_boundaries() { + let docx = + b"Helloworldend"; + assert_eq!( + xml_text(docx, &[], 1024).unwrap().text, + "Hello\nworld\tend\n" + ); + let pptx = + b"Helloworld"; + assert_eq!(xml_text(pptx, &[], 1024).unwrap().text, "Hello\nworld\n"); +} + +#[test] +fn worksheet_phonetic_annotations_do_not_change_shared_or_inline_values() { + let shared = xml_strings("東京とうきょう駅".as_bytes()).unwrap(); + assert_eq!(shared, vec!["東京駅"]); + let xml = "0東京とうきょう駅"; + assert_eq!( + xml_text(xml.as_bytes(), &shared, 1024).unwrap().text, + "東京駅\n東京駅\n\n" + ); +} + +#[test] +fn word_carriage_return_elements_preserve_line_boundaries() { + for xml in [ + b"Helloworld".as_slice(), + b"Helloworld".as_slice(), + ] { + assert_eq!(xml_text(xml, &[], 1024).unwrap().text, "Hello\nworld\n"); + } +} diff --git a/src/intake/office_order.rs b/src/intake/office_order.rs new file mode 100644 index 0000000..e00937c --- /dev/null +++ b/src/intake/office_order.rs @@ -0,0 +1,199 @@ +//! Resolve displayed slide and worksheet order through OOXML manifests. + +use super::{ + DocumentFormat, MAX_ENTRIES, Result, failed, normalize_xml, read_part, + zip_admission::AdmittedReader, +}; +use crate::Error; +use quick_xml::{Reader, events::Event}; +use std::collections::{BTreeMap, BTreeSet}; + +/// Return only referenced parts, in presentation/workbook order. +pub(super) fn ordered_parts( + archive: &mut zip::ZipArchive>, + format: DocumentFormat, + parts: &[(String, usize)], + expanded: &mut u64, +) -> Result> { + let (manifest, relationships, directory, item, container, kind) = match format { + DocumentFormat::Pptx => ( + "ppt/presentation.xml", + "ppt/_rels/presentation.xml.rels", + "ppt", + "sldId", + "sldIdLst", + "slide", + ), + DocumentFormat::Xlsx => ( + "xl/workbook.xml", + "xl/_rels/workbook.xml.rels", + "xl", + "sheet", + "sheets", + "worksheet", + ), + _ => return Err(invalid("unsupported ordered document format")), + }; + let manifest_xml = read_part(archive.by_name(manifest).map_err(failed)?, expanded)?; + let relationships_xml = read_part(archive.by_name(relationships).map_err(failed)?, expanded)?; + let mut ids = Vec::new(); + elements(&manifest_xml, |name, parent, attrs| { + if name == item && parent == container { + let id = attrs + .iter() + .find(|(key, _)| key.ends_with(":id")) + .ok_or_else(|| invalid("missing document relationship id"))? + .1 + .clone(); + if ids.len() >= MAX_ENTRIES { + return Err(invalid("too many document references")); + } + ids.push(id); + } + Ok(()) + })?; + let mut targets = BTreeMap::new(); + elements(&relationships_xml, |name, parent, attrs| { + if name != "Relationship" || parent != "Relationships" { + return Ok(()); + } + let get = |key: &str| { + attrs + .iter() + .find(|(name, _)| name == key) + .map(|(_, value)| value.as_str()) + }; + let id = get("Id").ok_or_else(|| invalid("missing relationship id"))?; + let relevant = get("Type").is_some_and(|value| value.rsplit('/').next() == Some(kind)); + let target = if relevant && get("TargetMode") != Some("External") { + Some(resolve_target( + directory, + get("Target").ok_or_else(|| invalid("missing relationship target"))?, + )?) + } else { + None + }; + if targets.len() >= MAX_ENTRIES || targets.insert(id.to_owned(), target).is_some() { + return Err(invalid("duplicate or excessive relationships")); + } + Ok(()) + })?; + let available: BTreeMap<_, _> = parts + .iter() + .map(|(name, index)| (name.as_str(), *index)) + .collect(); + let mut seen = BTreeSet::new(); + let mut ordered = Vec::new(); + for id in ids { + let target = targets + .get(&id) + .and_then(Option::as_ref) + .ok_or_else(|| invalid("missing, external or unsupported document relationship"))?; + let index = available + .get(target.as_str()) + .ok_or_else(|| invalid("referenced document part is missing"))?; + if !seen.insert(target.clone()) { + return Err(invalid("duplicate document part reference")); + } + ordered.push((target.clone(), *index)); + } + Ok(ordered) +} + +fn invalid(reason: &str) -> Error { + Error::extraction_failed(reason) +} + +fn resolve_target(directory: &str, target: &str) -> Result { + if target.contains(['\\', ':', '?', '#', '\0']) { + return Err(invalid("invalid document relationship target")); + } + let mut components = if target.starts_with('/') { + Vec::new() + } else { + vec![directory] + }; + for component in target.split('/') { + match component { + "" | "." => {} + ".." => { + components + .pop() + .ok_or_else(|| invalid("relationship escapes package"))?; + } + _ => components.push(component), + } + } + let path = components.join("/"); + if path.is_empty() || path.len() > 256 { + return Err(invalid("invalid document relationship target")); + } + Ok(path) +} + +/// Stream bounded manifest metadata, including self-closing elements. +fn elements( + xml: &[u8], + mut consume: impl FnMut(&str, &str, &[(String, String)]) -> Result<()>, +) -> Result<()> { + let xml = normalize_xml(xml)?; + let mut reader = Reader::from_reader(xml.as_ref()); + let mut parents = Vec::::new(); + loop { + let event = reader.read_event().map_err(failed)?; + let is_empty = matches!(&event, Event::Empty(_)); + match event { + Event::Start(element) | Event::Empty(element) => { + let local = element.local_name(); + let name = std::str::from_utf8(local.as_ref()).map_err(failed)?; + if name.len() > 256 || parents.len() >= 128 { + return Err(invalid("XML nesting or name limit exceeded")); + } + let name = name.to_owned(); + let mut attrs = Vec::new(); + for attribute in element.attributes() { + let attribute = attribute.map_err(failed)?; + if attrs.len() >= 32 + || attribute.key.as_ref().len() > 256 + || attribute.value.len() > 256 + { + return Err(invalid("XML metadata limit exceeded")); + } + attrs.push(( + std::str::from_utf8(attribute.key.as_ref()) + .map_err(failed)? + .to_owned(), + attribute + .decoded_and_normalized_value( + quick_xml::XmlVersion::Implicit1_0, + reader.decoder(), + ) + .map_err(failed)? + .into_owned(), + )); + } + consume(&name, parents.last().map_or("", String::as_str), &attrs)?; + if !is_empty { + parents.push(name); + } + } + Event::End(_) => { + parents + .pop() + .ok_or_else(|| invalid("invalid XML nesting"))?; + } + Event::DocType(_) => return Err(invalid("XML DTDs are not supported")), + Event::Eof => { + if !parents.is_empty() { + return Err(invalid("unclosed XML elements")); + } + return Ok(()); + } + _ => {} + } + } +} + +#[cfg(test)] +#[path = "office_order_tests.rs"] +mod tests; diff --git a/src/intake/office_order_tests.rs b/src/intake/office_order_tests.rs new file mode 100644 index 0000000..67dc3b2 --- /dev/null +++ b/src/intake/office_order_tests.rs @@ -0,0 +1,88 @@ +//! Manifest parsing and relationship paths remain bounded and package-local. +#![allow(clippy::unwrap_used, clippy::expect_used, clippy::panic)] +use super::*; + +#[test] +fn resolves_normalized_internal_targets_and_rejects_external_or_escaping_paths() { + for (target, expected) in [ + ("./worksheets/sheet1.xml", "xl/worksheets/sheet1.xml"), + ("/xl/worksheets/sheet1.xml", "xl/worksheets/sheet1.xml"), + ("../xl/worksheets/sheet1.xml", "xl/worksheets/sheet1.xml"), + ] { + assert_eq!(resolve_target("xl", target).unwrap(), expected); + } + for target in [ + "https://example.com/s.xml", + "../../escape.xml", + "../", + "s.xml?query", + "s.xml#part", + "a\\b", + "a\0b", + ] { + assert!(resolve_target("xl", target).is_err(), "{target}"); + } + assert!(resolve_target("xl", &"x".repeat(257)).is_err()); +} + +#[test] +fn manifest_reader_caps_depth_names_attributes_and_rejects_invalid_structure() { + let read = |xml: &str| elements(xml.as_bytes(), |_, _, _| Ok(())); + assert!(read(&format!("{}{}", "".repeat(129), "".repeat(129))).is_err()); + assert!(read(&format!("<{} />", "a".repeat(257))).is_err()); + assert!(read(&format!("", "v".repeat(257))).is_err()); + assert!(read(&format!("", "k".repeat(257))).is_err()); + let mut attrs = String::new(); + for i in 0..33 { + use std::fmt::Write as _; + write!(&mut attrs, " k{i}='v'").unwrap(); + } + assert!(read(&format!("")).is_err()); + for malformed in [ + "", + "", + "", + "", + "", + "", + ] { + assert!(read(malformed).is_err(), "{malformed}"); + } + let mut records = Vec::new(); + elements( + b"", + |name, parent, attrs| { + if name == "sheet" { + records.push((parent.to_owned(), attrs[0].1.clone())); + } + Ok(()) + }, + ) + .unwrap(); + assert_eq!( + records, + vec![ + ("sheets".into(), "a&b".into()), + ("sheets".into(), "c".into()) + ] + ); +} + +#[test] +fn manifest_reader_accepts_utf16_xml() { + let mut xml = vec![0xFF, 0xFE]; + for unit in + "".encode_utf16() + { + xml.extend_from_slice(&unit.to_le_bytes()); + } + let mut values = Vec::new(); + elements(&xml, |name, parent, attrs| { + if name == "item" { + values.push((parent.to_owned(), attrs[0].1.clone())); + } + Ok(()) + }) + .unwrap(); + assert_eq!(values, vec![("root".into(), "ok".into())]); +} diff --git a/src/intake/zip_admission.rs b/src/intake/zip_admission.rs new file mode 100644 index 0000000..7d1ca4e --- /dev/null +++ b/src/intake/zip_admission.rs @@ -0,0 +1,276 @@ +//! Bounded ZIP central-directory admission before the eager ZIP parser. + +use super::MAX_ENTRIES; +use crate::{Error, Result}; +use std::collections::HashSet; + +fn invalid() -> Error { + Error::extraction_failed("invalid ZIP central directory") +} + +fn number(bytes: &[u8], offset: usize, width: usize) -> Result { + let slice = bytes + .get(offset..offset.checked_add(width).ok_or_else(invalid)?) + .ok_or_else(invalid)?; + Ok(slice + .iter() + .enumerate() + .fold(0, |value, (i, byte)| value | (u64::from(*byte) << (8 * i)))) +} + +fn footer_offset(bytes: &[u8]) -> Result { + let end = (bytes.len().saturating_sub(65_557)..bytes.len().saturating_sub(21)) + .rev() + .find(|&i| { + bytes.get(i..i + 4) == Some(b"PK\x05\x06") + && number(bytes, i + 20, 2) + .is_ok_and(|len| (i + 22) as u64 + len == bytes.len() as u64) + }) + .ok_or_else(invalid)?; + Ok(end) +} + +/// Bound all eagerly allocated ZIP metadata before opening the archive. +/// Counts, directory size, decoded names, extras, comments and ZIP64 metadata +/// are checked before indexing; duplicate-name detection borrows the input +/// bytes and allocates only a set bounded by the admitted member count. +/// Ambiguous alternative metadata footers +/// fail closed; footer signatures in payloads are hidden during indexing. +fn directory(bytes: &[u8]) -> Result { + let end = footer_offset(bytes)?; + let mut count = number(bytes, end + 10, 2)?; + if number(bytes, end + 8, 2)? != count + || number(bytes, end + 4, 2)? != 0 + || number(bytes, end + 6, 2)? != 0 + { + return Err(invalid()); + } + let mut size = number(bytes, end + 12, 4)?; + let mut directory_end = end; + let mut relative = number(bytes, end + 16, 4)?; + if end >= 20 && bytes.get(end - 20..end - 16) == Some(b"PK\x06\x07") { + let offset = usize::try_from(number(bytes, end - 12, 8)?).map_err(|_| invalid())?; + if bytes.get(offset..offset.saturating_add(4)) != Some(b"PK\x06\x06") { + return Err(invalid()); + } + let record_size = number(bytes, offset + 4, 8)?; + if record_size < 44 + || (offset as u64) + .checked_add(12) + .and_then(|v| v.checked_add(record_size)) + != Some((end - 20) as u64) + { + return Err(invalid()); + } + if record_size > 1024 * 1024 { + return Err(Error::extraction_failed( + "ZIP admission metadata limit exceeded", + )); + } + count = number(bytes, offset + 32, 8)?; + if number(bytes, offset + 24, 8)? != count + || number(bytes, offset + 16, 4)? != 0 + || number(bytes, offset + 20, 4)? != 0 + || number(bytes, end - 16, 4)? != 0 + || number(bytes, end - 4, 4)? != 1 + { + return Err(invalid()); + } + size = number(bytes, offset + 40, 8)?; + directory_end = offset; + relative = number(bytes, offset + 48, 8)?; + for (legacy, sentinel, actual) in [ + (number(bytes, end + 10, 2)?, u64::from(u16::MAX), count), + (number(bytes, end + 12, 4)?, u64::from(u32::MAX), size), + (number(bytes, end + 16, 4)?, u64::from(u32::MAX), relative), + ] { + if legacy != sentinel && legacy != actual { + return Err(invalid()); + } + } + } else if count == 65535 || size == u64::from(u32::MAX) { + return Err(invalid()); + } + let size = usize::try_from(size).map_err(|_| invalid())?; + let start = directory_end.checked_sub(size).ok_or_else(invalid)?; + if relative > start as u64 { + return Err(invalid()); + } + Ok(Directory { + start, + end: directory_end, + footer: end, + count, + relative, + }) +} + +struct Directory { + start: usize, + end: usize, + footer: usize, + count: u64, + relative: u64, +} + +pub(super) fn zip_preflight(bytes: &[u8]) -> Result<()> { + let Directory { + start, + end: directory_end, + footer: end, + count, + .. + } = directory(bytes)?; + let size = directory_end - start; + // The eager reader retries earlier footer candidates. Embedded footers in + // admitted metadata are rejected; payload footers are hidden by the reader + // below while it constructs its metadata index. This deliberately excludes + // otherwise-valid names, extras and comments containing those byte sequences: + // document availability is sacrificed rather than allowing an eager fallback. + for (i, signature) in bytes[start..].windows(4).enumerate() { + let position = start + i; + if (signature == b"PK\x05\x06" && position != end) + || (signature == b"PK\x06\x06" && (directory_end == end || position != directory_end)) + { + return Err(invalid()); + } + } + // Each directory entry requires at least 46 bytes. Check before comparing + // the count to a cap: impossible hostile counts are malformed, not truncated. + if count > (size / 46) as u64 { + return Err(invalid()); + } + if count > MAX_ENTRIES as u64 { + return Err(Error::extraction_failed( + "ZIP admission count limit exceeded", + )); + } + // Extra fields and comments are also eagerly allocated by the ZIP reader. + if size > 1024 * 1024 { + return Err(Error::extraction_failed( + "ZIP admission metadata limit exceeded", + )); + } + let mut cursor = start; + let mut names = 0usize; + let mut raw_names = HashSet::with_capacity(usize::try_from(count).map_err(|_| invalid())?); + for _ in 0..count { + if bytes.get(cursor..cursor.saturating_add(4)) != Some(b"PK\x01\x02") { + return Err(invalid()); + } + let name_len = usize::try_from(number(bytes, cursor + 28, 2)?).map_err(|_| invalid())?; + let extra_len = usize::try_from(number(bytes, cursor + 30, 2)?).map_err(|_| invalid())?; + let comment_len = usize::try_from(number(bytes, cursor + 32, 2)?).map_err(|_| invalid())?; + let name_start = cursor + 46; + let next = name_start + name_len + extra_len + comment_len; + if next > directory_end { + return Err(invalid()); + } + let name = &bytes[name_start..name_start + name_len]; + // CP437 characters can expand to three UTF-8 bytes. UTF-8 names use + // the exact lossy-decoding length (without allocating a String). + let mut displayed = if number(bytes, cursor + 8, 2)? & 0x800 != 0 { + std::str::from_utf8(name).map_or(name.len() * 3, str::len) + } else { + name.iter() + .map(|byte| if *byte < 128 { 1 } else { 3 }) + .sum() + }; + let mut field = name_start + name_len; + let extra_end = field + extra_len; + while field < extra_end { + if extra_end - field < 4 { + return Err(invalid()); + } + let tag = number(bytes, field, 2)?; + let len = usize::try_from(number(bytes, field + 2, 2)?).map_err(|_| invalid())?; + field += 4; + if len > extra_end - field { + return Err(invalid()); + } + // The Unicode path extra field can replace the raw name. Bound + // it even when its CRC/version would later cause it to be ignored. + if tag == 0x7075 && len >= 5 { + displayed = displayed.max(len - 5); + } + field += len; + } + names += displayed; + if displayed > 256 || names > MAX_ENTRIES * 256 { + return Err(Error::extraction_failed( + "ZIP admission name limit exceeded", + )); + } + if !raw_names.insert(name) { + return Err(Error::extraction_failed("duplicate ZIP member name")); + } + cursor = next; + } + if cursor != directory_end { + return Err(invalid()); + } + Ok(()) +} + +/// A reader that hides payload footers while the ZIP parser indexes metadata. +/// This prevents fallback to unchecked footers embedded in a member's bytes. +pub(super) struct AdmittedReader<'a> { + cursor: std::io::Cursor<&'a [u8]>, + start: u64, + indexing: std::rc::Rc>, +} +impl std::io::Read for AdmittedReader<'_> { + fn read(&mut self, buffer: &mut [u8]) -> std::io::Result { + let position = self.cursor.position(); + let read = std::io::Read::read(&mut self.cursor, buffer)?; + if self.indexing.get() && position < self.start { + let hidden = usize::try_from((self.start - position).min(read as u64)) + .map_err(std::io::Error::other)?; + // Local header reads are needed during indexing. Hide only footer + // magic, including signatures crossing a read-buffer boundary. + for (index, byte) in buffer[..hidden].iter_mut().enumerate() { + let offset = usize::try_from(position).map_err(std::io::Error::other)? + index; + if self + .cursor + .get_ref() + .get(offset..offset + 4) + .is_some_and(|magic| magic == b"PK\x05\x06" || magic == b"PK\x06\x06") + { + *byte = 0; + } + } + } + Ok(read) + } +} +impl std::io::Seek for AdmittedReader<'_> { + fn seek(&mut self, position: std::io::SeekFrom) -> std::io::Result { + std::io::Seek::seek(&mut self.cursor, position) + } +} + +/// Construct only after successful preflight. Pin the metadata offset and hide +/// all earlier member footer signatures until indexing finishes; payload reads then work +/// normally, including CRC and decompression validation. +pub(super) fn open_admitted_zip(bytes: &[u8]) -> Result>> { + let metadata = directory(bytes)?; + let start = metadata.start as u64; + let archive_offset = start.checked_sub(metadata.relative).ok_or_else(invalid)?; + let indexing = std::rc::Rc::new(std::cell::Cell::new(true)); + let reader = AdmittedReader { + cursor: std::io::Cursor::new(bytes), + start, + indexing: indexing.clone(), + }; + let config = zip::read::Config { + archive_offset: zip::read::ArchiveOffset::Known(archive_offset), + }; + let archive = zip::ZipArchive::with_config(config, reader) + .map_err(|error| Error::extraction_failed(&error.to_string()))?; + indexing.set(false); + Ok(archive) +} + +#[cfg(test)] +#[path = "zip_admission_tests.rs"] +mod tests; diff --git a/src/intake/zip_admission_tests.rs b/src/intake/zip_admission_tests.rs new file mode 100644 index 0000000..90de4c6 --- /dev/null +++ b/src/intake/zip_admission_tests.rs @@ -0,0 +1,246 @@ +//! Hostile metadata is rejected before the eager ZIP index is constructed. +#![allow(clippy::unwrap_used, clippy::expect_used, clippy::panic)] +use super::*; +use crate::intake::{DocumentFormat, ExtractDocumentSpec, extract}; +use std::io::{Cursor, Write}; + +fn fixture(parts: &[(&str, &[u8])]) -> Vec { + let mut writer = zip::ZipWriter::new(Cursor::new(Vec::new())); + for (name, data) in parts { + writer + .start_file( + *name, + zip::write::SimpleFileOptions::default() + .compression_method(zip::CompressionMethod::Stored), + ) + .unwrap(); + writer.write_all(data).unwrap(); + } + writer.finish().unwrap().into_inner() +} +fn docx() -> Vec { + fixture(&[("word/document.xml", b"visible")]) +} +fn extract_docx(bytes: &[u8]) -> crate::Result { + extract(bytes, &ExtractDocumentSpec::new(DocumentFormat::Docx)) +} +fn zip64(bytes: &[u8]) -> Vec { + let end = bytes.len() - 22; + let mut result = bytes[..end].to_vec(); + result.extend_from_slice(b"PK\x06\x06"); + result.extend_from_slice(&44u64.to_le_bytes()); + result.extend_from_slice(&[0; 12]); + result.extend_from_slice(&1u64.to_le_bytes()); + result.extend_from_slice(&1u64.to_le_bytes()); + result.extend_from_slice(&number(bytes, end + 12, 4).unwrap().to_le_bytes()); + result.extend_from_slice(&number(bytes, end + 16, 4).unwrap().to_le_bytes()); + result.extend_from_slice(b"PK\x06\x07"); + result.extend_from_slice(&0u32.to_le_bytes()); + result.extend_from_slice(&(end as u64).to_le_bytes()); + result.extend_from_slice(&1u32.to_le_bytes()); + result.extend_from_slice(&bytes[end..]); + let footer = result.len() - 22; + result[footer + 8..footer + 12].fill(255); + result[footer + 12..footer + 20].fill(255); + result +} + +#[test] +fn admits_valid_zip64_and_rejects_hostile_declared_counts() { + let bytes = docx(); + let valid = zip64(&bytes); + assert_eq!(extract_docx(&valid).unwrap().sections[0].text, "visible"); + let end = bytes.len() - 22; + let mut hostile = bytes.clone(); + hostile[end + 8..end + 12].fill(255); + assert!( + extract_docx(&hostile) + .unwrap_err() + .to_string() + .contains("invalid ZIP central") + ); + let mut hostile64 = valid.clone(); + hostile64[end + 24..end + 40].fill(255); + assert!( + extract_docx(&hostile64) + .unwrap_err() + .to_string() + .contains("invalid ZIP central") + ); + let mut disagreement = valid; + let footer = disagreement.len() - 22; + disagreement[footer + 10..footer + 12].copy_from_slice(&2u16.to_le_bytes()); + assert!(zip_preflight(&disagreement).is_err()); +} + +#[test] +fn rejects_split_archives_impossible_offsets_and_malformed_zip64() { + let bytes = docx(); + let end = bytes.len() - 22; + for (position, value) in [ + (end + 4, 1), + (end + 6, 1), + (end + 8, 2), + (end + 12, 255), + (end + 16, 255), + ] { + let mut malformed = bytes.clone(); + malformed[position] = value; + assert!(zip_preflight(&malformed).is_err()); + } + let valid = zip64(&bytes); + for position in [ + end, + end + 4, + end + 16, + end + 20, + end + 56 + 4, + end + 56 + 8, + end + 56 + 16, + ] { + let mut malformed = valid.clone(); + malformed[position] = 255; + assert!(zip_preflight(&malformed).is_err(), "offset {position}"); + } + for bytes in [b"".as_slice(), b"PK\x05\x06", b"not a zip"] { + assert!(zip_preflight(bytes).is_err()); + } +} + +#[test] +fn bounds_unicode_path_extra_fields_before_allocating_replacement_names() { + let bytes = docx(); + let end = bytes.len() - 22; + let size = usize::try_from(number(&bytes, end + 12, 4).unwrap()).unwrap(); + let central = end - size; + let name_len = usize::try_from(number(&bytes, central + 28, 2).unwrap()).unwrap(); + let mut extra = Vec::new(); + extra.extend_from_slice(&0x7075u16.to_le_bytes()); + extra.extend_from_slice(&262u16.to_le_bytes()); + extra.extend_from_slice(&[1, 0, 0, 0, 0]); + extra.extend_from_slice(&[b'x'; 257]); + let mut hostile = bytes.clone(); + hostile.splice( + central + 46 + name_len..central + 46 + name_len, + extra.iter().copied(), + ); + hostile[central + 30..central + 32] + .copy_from_slice(&u16::try_from(extra.len()).unwrap().to_le_bytes()); + let new_footer = hostile.len() - 22; + hostile[new_footer + 12..new_footer + 16] + .copy_from_slice(&u32::try_from(size + extra.len()).unwrap().to_le_bytes()); + assert!( + extract_docx(&hostile) + .unwrap_err() + .to_string() + .contains("ZIP admission name limit") + ); + let mut malformed = hostile.clone(); + malformed[central + 46 + name_len + 2..central + 46 + name_len + 4].fill(255); + assert!(zip_preflight(&malformed).is_err()); +} + +#[test] +fn central_directory_metadata_and_zip64_extensions_have_fixed_caps() { + let count = 17u16; + let mut bytes = Vec::new(); + for _ in 0..count { + let mut header = [0u8; 46]; + header[..4].copy_from_slice(b"PK\x01\x02"); + header[28..30].copy_from_slice(&1u16.to_le_bytes()); + header[30..32].copy_from_slice(&u16::MAX.to_le_bytes()); + bytes.extend_from_slice(&header); + bytes.push(b'x'); + bytes.extend_from_slice(&vec![0; usize::from(u16::MAX)]); + } + let size = u32::try_from(bytes.len()).unwrap(); + let mut footer = [0u8; 22]; + footer[..4].copy_from_slice(b"PK\x05\x06"); + footer[8..10].copy_from_slice(&count.to_le_bytes()); + footer[10..12].copy_from_slice(&count.to_le_bytes()); + footer[12..16].copy_from_slice(&size.to_le_bytes()); + bytes.extend_from_slice(&footer); + assert!( + extract_docx(&bytes) + .unwrap_err() + .to_string() + .contains("ZIP admission metadata limit") + ); + + let original = docx(); + let offset = original.len() - 22; + let mut huge = zip64(&original); + let extra = 1024 * 1024; + huge.splice(offset + 56..offset + 56, std::iter::repeat_n(0, extra)); + huge[offset + 4..offset + 12].copy_from_slice(&(44 + extra as u64).to_le_bytes()); + assert!( + extract_docx(&huge) + .unwrap_err() + .to_string() + .contains("ZIP admission metadata limit") + ); +} + +#[test] +fn payload_footer_cannot_be_retried_when_primary_directory_is_invalid() { + let nested = fixture(&[("nested.xml", b"unchecked")]); + let bytes = fixture(&[ + ("nested.zip", &nested), + ("word/document.xml", b"visible"), + ]); + assert_eq!(extract_docx(&bytes).unwrap().sections[0].text, "visible"); + let mut prefixed = b"preamble".to_vec(); + prefixed.extend_from_slice(&bytes); + assert_eq!(extract_docx(&prefixed).unwrap().sections[0].text, "visible"); + let end = bytes.len() - 22; + let size = usize::try_from(number(&bytes, end + 12, 4).unwrap()).unwrap(); + let central = end - size; + let mut corrupt = bytes.clone(); + // AES without its required metadata makes the eager parser search earlier + // footers; the embedded ZIP must not become a second admission candidate. + corrupt[central + 10..central + 12].copy_from_slice(&99u16.to_le_bytes()); + zip_preflight(&corrupt).unwrap(); + assert!(extract_docx(&corrupt).is_err()); +} + +#[test] +fn alternate_footer_signatures_in_metadata_fail_closed() { + let mut writer = zip::ZipWriter::new(Cursor::new(Vec::new())); + writer.set_comment("comment PK\u{5}\u{6} footer").unwrap(); + writer + .start_file( + "word/document.xml", + zip::write::SimpleFileOptions::default(), + ) + .unwrap(); + writer.write_all(b"visible").unwrap(); + let bytes = writer.finish().unwrap().into_inner(); + assert!(extract_docx(&bytes).is_err()); +} + +#[test] +fn rejects_duplicate_raw_member_names_before_the_zip_index_can_deduplicate_them() { + let original = "word/document.xml"; + let mut bytes = fixture(&[ + (original, b"first body"), + ("word/otherxxx.xml", b"second body"), + ]); + let metadata = directory(&bytes).unwrap(); + let first = metadata.start; + let next = first + + 46 + + usize::try_from(number(&bytes, first + 28, 2).unwrap()).unwrap() + + usize::try_from(number(&bytes, first + 30, 2).unwrap()).unwrap() + + usize::try_from(number(&bytes, first + 32, 2).unwrap()).unwrap(); + let local = usize::try_from(number(&bytes, next + 42, 4).unwrap()).unwrap(); + assert_eq!( + usize::try_from(number(&bytes, next + 28, 2).unwrap()).unwrap(), + original.len() + ); + // Preserve both valid records, payloads and CRCs while giving the two local + // and central headers the same raw name. ZipWriter disallows this fixture. + bytes[next + 46..next + 46 + original.len()].copy_from_slice(original.as_bytes()); + bytes[local + 30..local + 30 + original.len()].copy_from_slice(original.as_bytes()); + let error = extract_docx(&bytes).unwrap_err(); + assert!(error.to_string().contains("duplicate ZIP member name")); +} diff --git a/src/lib.rs b/src/lib.rs index 70a7d98..b55e699 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -74,7 +74,7 @@ //! //! # Feature flags //! -//! Each format is a separate gate, and every gate is on by default. Turning one +//! Generation and PDF text extraction gates are on by default. Turning one //! off drops its writer and that writer's dependencies; the specs stay, so the //! contract and its validation survive any combination. //! @@ -83,6 +83,8 @@ //! `syntect` and `pulldown-cmark`. //! - `pdf` (default) — `.pdf` text extraction via `pdf-extract`, which also //! drops its font and `PostScript` parsing stack. +//! - `intake` (optional) — bounded PDF/DOCX/PPTX/XLSX text and provenance. +//! - `pdf-render` (optional) — selected PDF page PNGs via Hayro 0.5. pub use tinydocs_bus::spec; @@ -96,3 +98,10 @@ pub mod pptx; pub mod pdf; pub use tinydocs_bus::{Error, Result}; + +/// Bounded document intake with section provenance. +#[cfg(feature = "intake")] +pub mod intake; +/// Selected PDF page rendering for host-owned vision/OCR. +#[cfg(feature = "pdf-render")] +pub mod pdf_render; diff --git a/src/pdf/fixtures.rs b/src/pdf/fixtures.rs new file mode 100644 index 0000000..6d31471 --- /dev/null +++ b/src/pdf/fixtures.rs @@ -0,0 +1,57 @@ +//! Deterministic mixed text/image PDF fixtures. +/// Build pages carrying text and/or a genuine embedded raster image. +pub(crate) fn document(pages: &[(&str, bool)], encrypted: bool) -> Vec { + let mut objects = vec![ + String::new(), + String::new(), + "<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>".to_owned(), + ]; + let mut kids = Vec::new(); + for (text, image) in pages { + let page_id = objects.len() + 1; + kids.push(format!("{page_id} 0 R")); + let content = if *image { + "q 100 0 0 100 0 0 cm /Im1 Do Q\n".to_owned() + } else { + String::new() + } + &format!("BT /F1 24 Tf 10 100 Td ({text}) Tj ET\n"); + objects.push(format!("<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 300] /Resources << /Font << /F1 3 0 R >> /XObject << /Im1 {} 0 R >> >> /Contents {} 0 R >>", page_id+2, page_id+1)); + objects.push(format!( + "<< /Length {} >>\nstream\n{content}endstream", + content.len() + )); + objects.push("<< /Type /XObject /Subtype /Image /Width 1 /Height 1 /ColorSpace /DeviceRGB /BitsPerComponent 8 /Length 7 /Filter /ASCIIHexDecode >>\nstream\n000000>\nendstream".to_owned()); + } + objects[0] = "<< /Type /Catalog /Pages 2 0 R >>".to_owned(); + objects[1] = format!( + "<< /Type /Pages /Kids [{}] /Count {} >>", + kids.join(" "), + pages.len() + ); + let mut bytes = b"%PDF-1.4\n".to_vec(); + let mut offsets = Vec::new(); + for (index, body) in objects.iter().enumerate() { + offsets.push(bytes.len()); + bytes.extend_from_slice(format!("{} 0 obj\n{body}\nendobj\n", index + 1).as_bytes()); + } + let xref = bytes.len(); + bytes.extend_from_slice( + format!("xref\n0 {}\n0000000000 65535 f \n", objects.len() + 1).as_bytes(), + ); + for offset in offsets { + bytes.extend_from_slice(format!("{offset:010} 00000 n \n").as_bytes()); + } + let encrypt = if encrypted { + "/Encrypt << /Filter /Standard /V 1 /R 2 /O (bad) /U (bad) /P -4 >>" + } else { + "" + }; + bytes.extend_from_slice( + format!( + "trailer\n<< /Size {} /Root 1 0 R {encrypt} >>\nstartxref\n{xref}\n%%EOF", + objects.len() + 1 + ) + .as_bytes(), + ); + bytes +} diff --git a/src/pdf/mod.rs b/src/pdf/mod.rs index f0b5dfc..cd049f7 100644 --- a/src/pdf/mod.rs +++ b/src/pdf/mod.rs @@ -87,3 +87,6 @@ pub fn extract_text(bytes: &[u8]) -> Result { #[cfg(test)] #[path = "mod_tests.rs"] mod test; + +#[cfg(test)] +pub(crate) mod fixtures; diff --git a/src/pdf_render/README.md b/src/pdf_render/README.md new file mode 100644 index 0000000..cdcc55e --- /dev/null +++ b/src/pdf_render/README.md @@ -0,0 +1,27 @@ +# PDF rasterization + +The optional `pdf-render` feature uses Hayro 0.5 to render selected source pages +to PNG. `render(bytes, &RenderPdfSpec)` returns `PdfImages`, preserving request +order, one-based page numbers and dimensions. It never eagerly rasterizes all +pages. The TinyBus `RenderPdf` method retains each PNG through the existing +`OutputRef` / `ReadOutput` / `ReleaseOutput` lifecycle. If an output-store +insertion fails partway through a batch, all new handles in that batch are +released; previously held caller outputs remain intact. + +Hard ceilings are 64 MiB input, 4,096 source pages, eight unique selected pages, +2,048 pixels on the longest edge, 16 million aggregate pixels, and 32 MiB +aggregate PNG bytes. Caller budgets can only narrow these limits. All selected +page dimensions and the aggregate pixel budget are checked before rendering. +Encrypted/malformed PDFs, missing pages and exceeded budgets return an error; +no partial successful result is returned. PNG byte limits apply to encoded +outputs; one bounded raster must be encoded before its size is known. + +Hayro runs synchronously in the native module's process. Its parser, image +codecs and display-list allocations do not expose a global memory budget. +Raster/input/output bounds therefore do not guarantee bounded internal parser +or renderer resources. A blocking-pool timeout is not cancellation. The host +owns execution deadlines and any stronger process isolation. No network OCR +or remote inference happens here. + +Hayro 0.5 and the resolved extraction/render dependency graph compile and test +on the declared Rust 1.88 MSRV. diff --git a/src/pdf_render/mod.rs b/src/pdf_render/mod.rs new file mode 100644 index 0000000..4317600 --- /dev/null +++ b/src/pdf_render/mod.rs @@ -0,0 +1,162 @@ +//! Selected PDF page rasterization for host-owned vision and OCR. +//! +//! Rendering is synchronous, has no filesystem/network access, and returns PNG +//! bytes. Hosts own isolation, cancellation and deadlines for parser/renderer +//! work; pixel and output budgets bound returned raster allocations. +use crate::{Error, Result}; +use hayro::hayro_interpret::InterpreterSettings; +use hayro::hayro_syntax::Pdf; +use hayro::vello_cpu::color::palette::css::WHITE; +use std::sync::Arc; +pub use tinydocs_bus::RenderPdfSpec; + +/// One selected PNG raster, prior to placement in a module output store. +#[derive(Debug)] +pub struct PdfPageImage { + /// One-based source page number. + pub page: u32, + /// Raster width in pixels. + pub width: u32, + /// Raster height in pixels. + pub height: u32, + /// Encoded PNG bytes. + pub bytes: Vec, +} +/// Selected PNG rasters and total source page count. +#[derive(Debug)] +pub struct PdfImages { + /// Total pages in the source document. + pub page_count: u32, + /// Selected pages in request order. + pub pages: Vec, +} +/// Rasterize only explicitly selected pages, preserving request order. +/// +/// # Errors +/// Rejects malformed/encrypted PDFs, invalid pages, duplicate selections, +/// dimensions outside 1–2048, batches over eight pages and exceeded pixel/PNG +/// byte budgets. No output is returned if any selected page fails. +pub fn render(bytes: &[u8], spec: &RenderPdfSpec) -> Result { + validate(spec)?; + if bytes.len() > crate::pdf::MAX_DOCUMENT_BYTES || !bytes.starts_with(b"%PDF-") { + return Err(Error::invalid_input( + "bytes", + "PDF signature or size is invalid", + )); + } + let document = pdf_extract::Document::load_mem(bytes).map_err(failed)?; + if document.is_encrypted() || document.trailer.has(b"Encrypt") { + return Err(Error::extraction_failed("encrypted PDF is not supported")); + } + let pdf = Pdf::new(Arc::new(bytes.to_vec())).map_err(|error| failed(format!("{error:?}")))?; + let page_count = u32::try_from(pdf.pages().len()).map_err(failed)?; + if page_count > 4096 { + return Err(Error::extraction_failed("PDF page limit exceeded")); + } + let mut plans = Vec::new(); + let mut pixels = 0u64; + for number in &spec.pages { + let page = pdf.pages().get((*number - 1) as usize).ok_or_else(|| { + Error::invalid_input("pages", "page number exceeds document page count") + })?; + let (source_width, source_height) = page.render_dimensions(); + if !source_width.is_finite() + || !source_height.is_finite() + || source_width <= 0.0 + || source_height <= 0.0 + || !(0.001..=1_000_000.0).contains(&source_width) + || !(0.001..=1_000_000.0).contains(&source_height) + { + return Err(Error::extraction_failed("invalid PDF page dimensions")); + } + let (width, height, scale) = dimensions(source_width, source_height, spec.max_dimension); + pixels = pixels + .checked_add(u64::from(width) * u64::from(height)) + .ok_or_else(|| Error::invalid_input("max_total_pixels", "pixel count overflow"))?; + if pixels > spec.max_total_pixels { + return Err(Error::invalid_input( + "max_total_pixels", + "selected pages exceed pixel budget", + )); + } + plans.push((*number, width, height, scale)); + } + let mut output = PdfImages { + page_count, + pages: Vec::new(), + }; + let mut total_bytes = 0u64; + for (number, width, height, scale) in plans { + let page = &pdf.pages()[(number - 1) as usize]; + let settings = hayro::RenderSettings { + width: Some(width), + height: Some(height), + x_scale: scale, + y_scale: scale, + bg_color: WHITE, + }; + let pixmap = hayro::render(page, &InterpreterSettings::default(), &settings); + let bytes = pixmap.into_png().map_err(failed)?; + total_bytes = total_bytes + .checked_add(bytes.len() as u64) + .ok_or_else(|| Error::extraction_failed("PNG byte count overflow"))?; + if total_bytes > spec.max_output_bytes { + return Err(Error::invalid_input( + "max_output_bytes", + "rendered PNGs exceed byte budget", + )); + } + output.pages.push(PdfPageImage { + page: number, + width: u32::from(width), + height: u32::from(height), + bytes, + }); + } + Ok(output) +} +fn validate(spec: &RenderPdfSpec) -> Result<()> { + if spec.pages.is_empty() + || spec.pages.len() > 8 + || spec.max_dimension == 0 + || spec.max_dimension > 2048 + || spec.max_total_pixels == 0 + || spec.max_total_pixels > 16_000_000 + || spec.max_output_bytes == 0 + || spec.max_output_bytes > 32 * 1024 * 1024 + { + return Err(Error::invalid_input( + "spec", + "render budgets outside hard bounds", + )); + } + for (index, page) in spec.pages.iter().enumerate() { + if *page == 0 || spec.pages[..index].contains(page) { + return Err(Error::invalid_input( + "pages", + "pages must be positive and unique", + )); + } + } + Ok(()) +} +#[allow( + clippy::cast_possible_truncation, + clippy::cast_sign_loss, + clippy::cast_precision_loss, + reason = "validated finite positive dimensions clamp to 1..=2048 before conversion" +)] +fn dimensions(width: f32, height: f32, max: u32) -> (u16, u16, f32) { + let scale = max as f32 / width.max(height); + ( + (width * scale).ceil().clamp(1.0, max as f32) as u16, + (height * scale).ceil().clamp(1.0, max as f32) as u16, + scale, + ) +} +fn failed(error: impl std::fmt::Display) -> Error { + Error::extraction_failed(&error.to_string()) +} +#[cfg(test)] +#[path = "mod_tests.rs"] +mod tests; diff --git a/src/pdf_render/mod_tests.rs b/src/pdf_render/mod_tests.rs new file mode 100644 index 0000000..6956393 --- /dev/null +++ b/src/pdf_render/mod_tests.rs @@ -0,0 +1,77 @@ +//! PDF rendering boundary tests. +#![allow(clippy::unwrap_used, clippy::expect_used, clippy::panic)] +use super::*; +#[test] +fn rejects_unbounded_or_duplicate_page_selection() { + let spec = RenderPdfSpec { + pages: vec![1, 1], + max_dimension: 1024, + max_total_pixels: 4_000_000, + max_output_bytes: 8_000_000, + }; + assert!(render(b"%PDF-invalid", &spec).is_err()); +} + +fn spec() -> RenderPdfSpec { + RenderPdfSpec { + pages: vec![2, 1], + max_dimension: 120, + max_total_pixels: 1_000_000, + max_output_bytes: 1_000_000, + } +} +#[test] +fn renders_only_requested_pages_in_order_with_png_dimensions() { + let images = render( + &crate::pdf::fixtures::document(&[("text", false), ("", true), ("mixed", true)], false), + &spec(), + ) + .unwrap(); + assert_eq!(images.page_count, 3); + assert_eq!(images.pages.len(), 2); + assert_eq!(images.pages[0].page, 2); + assert_eq!(images.pages[1].page, 1); + let decoder = png::Decoder::new(std::io::Cursor::new(&images.pages[0].bytes)); + let mut reader = decoder.read_info().unwrap(); + let mut pixels = vec![0; reader.output_buffer_size()]; + let frame = reader.next_frame(&mut pixels).unwrap(); + assert!( + pixels[..frame.buffer_size()] + .as_chunks::<4>() + .0 + .iter() + .any(|pixel| pixel[0] < 128 && pixel[1] < 128 && pixel[2] < 128), + "scanned image must produce non-white raster pixels" + ); + for page in images.pages { + assert!(page.bytes.starts_with(b"\x89PNG\r\n\x1a\n")); + assert_eq!((page.width, page.height), (80, 120)); + } +} +#[test] +fn refuses_page_pixel_output_dimension_and_encryption_failures() { + let bytes = crate::pdf::fixtures::document(&[("text", false), ("", true)], false); + let mut request = spec(); + request.pages = vec![3]; + assert!(render(&bytes, &request).is_err()); + request = spec(); + request.pages = vec![0]; + assert!(render(&bytes, &request).is_err()); + request = spec(); + request.max_total_pixels = 1; + assert!(render(&bytes, &request).is_err()); + request = spec(); + request.max_output_bytes = 1; + assert!(render(&bytes, &request).is_err()); + request = spec(); + request.max_dimension = 2049; + assert!(render(&bytes, &request).is_err()); + assert!( + render( + &crate::pdf::fixtures::document(&[("secret", false)], true), + &spec() + ) + .is_err() + ); + assert!(render(b"%PDF-broken", &spec()).is_err()); +}