From d00a1ba6ab008fa0c1e831940525f5eb18214e6d Mon Sep 17 00:00:00 2001 From: ksemenenko Date: Thu, 1 Oct 2026 13:56:02 +0200 Subject: [PATCH] Stage cloud PDF sources before synchronous parsing --- CHANGELOG.md | 6 + Directory.Build.props | 2 +- README.md | 6 + docs/Architecture.md | 2 +- docs/Features/file-context.md | 7 + docs/Testing/index.md | 1 + .../FileContextDefaults.cs | 1 + .../FileContextOptions.cs | 21 +++ .../Pdf/FileContextPdfSource.cs | 70 ++++++-- .../Pdf/FileContextPdfSourceStagingMode.cs | 11 ++ .../AsyncOnlySeekablePdfStream.cs | 39 ++++ .../FileContextPdfSourceStagingTests.cs | 170 ++++++++++++++++++ 12 files changed, 315 insertions(+), 21 deletions(-) create mode 100644 src/ManagedCode.FileContext/Pdf/FileContextPdfSourceStagingMode.cs create mode 100644 tests/ManagedCode.FileContext.Tests/AsyncOnlySeekablePdfStream.cs create mode 100644 tests/ManagedCode.FileContext.Tests/FileContextPdfSourceStagingTests.cs diff --git a/CHANGELOG.md b/CHANGELOG.md index 9ddea7b..6436e04 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,12 @@ All notable changes to ManagedCode.FileContext are documented here. +## 1.0.14 + +- Stage seekable cloud PDF streams asynchronously to bounded temporary files before synchronous parser reads and seeks. +- Preserve local file/memory sources by default and expose typed staging mode, buffer size and temporary-directory options. +- Return pooled staging buffers and delete temporary sources on completion, cancellation and failures. + ## 1.0.13 - Parse PDF text, pages and embedded images from bounded seekable streams instead of whole-document arrays. diff --git a/Directory.Build.props b/Directory.Build.props index 375fc58..c622b59 100644 --- a/Directory.Build.props +++ b/Directory.Build.props @@ -12,7 +12,7 @@ Recommended true $(NoWarn);CS1591;MAAI001 - 1.0.13 + 1.0.14 $(Version) diff --git a/README.md b/README.md index 3f85c75..4b54abb 100644 --- a/README.md +++ b/README.md @@ -185,6 +185,12 @@ Results include `StartLine`, `EndLine`, `HasMore`, and `TotalLines` when the end ## Read PDFs and send pages to vision models +Storage-backed PDF tools stage seekable cloud streams asynchronously before parsing. PdfPig's +synchronous reads and seeks then use a temporary file rather than repeated network ranges. Local +seekable files and memory streams remain reusable. `PdfSourceStagingMode` can force `TemporaryFile` +staging, `PdfSourceBufferBytes` bounds the copy buffer, and `PdfTemporaryDirectory` optionally selects +an existing host directory. Disposal, cancellation and staging failures remove temporary sources. + `IFileContextPdf.ReadPdfTextAsync(path)` returns bounded text, `PageCount`, and one-based `PagesWithoutText`. It does not perform OCR. A scanned page can instead be rendered with `RenderPdfPageAsync(path, pageNumber)`, which returns PNG `DataContent`. Use `CountPdfPageImagesAsync` and `ExtractPdfImageAsync` when the original embedded pictures are needed rather than the complete page. The four read-only `file_context_pdf_*` tools expose the same operations from scoped storage. For an authenticated PDF already held as bytes, `FileContextPdfTextExtractor.Extract`, `FileContextPdfImages.RenderPagePng`, and `FileContextPdfImages.ExtractPageImagesPng` work without storing it. PDF source reads default to 100 MiB and accept `FileContextOptions` for a different limit; page rasterization also uses configured pixel and PNG limits. `FileContextImageContent` creates model-visible `DataContent` from PNG bytes or base64 and `UriContent` from an HTTPS URL. A URL reference is not fetched by FileContext, so the model provider must be able to access it. A host must pass image content to its model as image content. A generic OpenAI Chat function result serializes it as text, so hosts must explicitly bridge image tool results into a multimodal model message. diff --git a/docs/Architecture.md b/docs/Architecture.md index 0dbdc3d..425f4c6 100644 --- a/docs/Architecture.md +++ b/docs/Architecture.md @@ -82,7 +82,7 @@ source and retaining it through parsing/rendering. `MaximumConcurrentPdfOperatio `FileContextPdfRenderDocument` exposes page count and sequential page rendering from one parsed PDF; dispose it after the batch. The low-level synchronous image helpers remain caller-scheduled APIs. -PDF text, page rendering and embedded-image APIs accept bounded seekable streams. Storage-backed PDF tools keep seekable provider streams directly and stage non-seekable sources to an automatically deleted temporary file, never a whole-document managed array. Native raster decoding has a separate per-page source-image pixel budget; lowering output scale does not reduce source bitmap allocation. +PDF text, page rendering and embedded-image APIs accept bounded seekable streams. Storage-backed PDF tools retain seekable local `FileStream`/`MemoryStream` inputs but asynchronously stage other streams, including seekable cloud streams, to an automatically deleted temporary file. PdfPig's synchronous byte reads and seeks then remain local, without blocking on repeated network ranges. `PdfSourceStagingMode.TemporaryFile` also stages local inputs. `PdfSourceBufferBytes` bounds every copy read and `PdfTemporaryDirectory` optionally selects an existing host directory. No whole-document managed array is created. Native raster decoding has a separate per-page source-image pixel budget; lowering output scale does not reduce source bitmap allocation. All potentially large operations are controlled by `IOptions`: PDF source/page/image budgets, full-read bytes, range bytes, files scanned, bytes per searched file, matches per file, total search results, graph documents, graph source bytes, and exported graph characters. Non-seekable cloud streams are supported by sequential streaming. diff --git a/docs/Features/file-context.md b/docs/Features/file-context.md index f6c030a..3489adc 100644 --- a/docs/Features/file-context.md +++ b/docs/Features/file-context.md @@ -115,6 +115,13 @@ Verification: DocumentCreationTests and DocumentValidationTests reopen real form ## PDF reads and vision images +PDF source staging keeps synchronous parser I/O local. The default `Automatic` mode reuses +seekable `FileStream` and `MemoryStream` sources and asynchronously stages all other inputs, +including seekable cloud streams. `TemporaryFile` stages every source. The configured +`PdfSourceBufferBytes` bounds each asynchronous copy read; `PdfTemporaryDirectory` may name an +existing host directory. Size rejection, cancellation and copy failures dispose the input and +delete any staged file. Source ownership continues through document rendering and caller disposal. + DOCX reading uses the native `file_context_docx_text` tool. It reads ordinary paragraph and table text from the scoped `.docx` package in bounded windows. Each result includes the next paragraph and character offset when more text remains, so an agent can continue without loading a long diff --git a/docs/Testing/index.md b/docs/Testing/index.md index 79a4501..d1f01c8 100644 --- a/docs/Testing/index.md +++ b/docs/Testing/index.md @@ -10,6 +10,7 @@ The suite is integration-first: - timeout tests cover configured operation expiry, cancellation of every public operation, disabled deadlines, duration validation, and timeout tool results through restored sessions; - concurrent storage tests write and range-read eight independent files through one shared adapter/service; - a sparse 1 GiB filesystem test reads bounded line windows repeatedly, rejects full-file loading, caps allocations, and proves that an oversized line fails before it can be buffered in memory. +- PDF cloud-source tests use real files behind an async-only seekable stream, proving parsing uses the staged local file; they cover large inputs, configured buffers/directories, forced staging, local-file reuse, limits, mid-copy cancellation, failure cleanup and private Unix permissions. Every filesystem test owns a unique temporary root and removes it on disposal. Test execution is serialized so process-wide allocation assertions cannot be distorted by another test. No `IStorage`, Agent Framework, Markdown-LD, or LlmTck mocks are used. diff --git a/src/ManagedCode.FileContext/FileContextDefaults.cs b/src/ManagedCode.FileContext/FileContextDefaults.cs index c487ff4..72968a7 100644 --- a/src/ManagedCode.FileContext/FileContextDefaults.cs +++ b/src/ManagedCode.FileContext/FileContextDefaults.cs @@ -8,6 +8,7 @@ public static class FileContextDefaults public const int FirstLineNumber = 1; public const int MaximumPdfReadBytes = 100 * 1024 * 1024; public const int MaximumConcurrentPdfOperations = 1; + public const int PdfSourceBufferBytes = 81920; public const int MaximumImageBytes = 8 * 1024 * 1024; public const int MaximumDecodedPdfImagePixels = 32_000_000; public const int MaximumRenderedPagePixels = 4_000_000; diff --git a/src/ManagedCode.FileContext/FileContextOptions.cs b/src/ManagedCode.FileContext/FileContextOptions.cs index adc5316..f4f3c3a 100644 --- a/src/ManagedCode.FileContext/FileContextOptions.cs +++ b/src/ManagedCode.FileContext/FileContextOptions.cs @@ -1,3 +1,5 @@ +using ManagedCode.FileContext.Pdf; + namespace ManagedCode.FileContext; /// Controls file access, approval, search, and graph limits for one context provider. @@ -23,6 +25,15 @@ public sealed class FileContextOptions public int MaximumPdfReadBytes { get; set; } = FileContextDefaults.MaximumPdfReadBytes; + /// Stages cloud streams before synchronous parsing; Automatic reuses local files and memory. + public FileContextPdfSourceStagingMode PdfSourceStagingMode { get; set; } = FileContextPdfSourceStagingMode.Automatic; + + /// Maximum bytes requested by one asynchronous PDF source staging read. + public int PdfSourceBufferBytes { get; set; } = FileContextDefaults.PdfSourceBufferBytes; + + /// Existing directory for temporary PDF sources. Null uses the operating system's temp directory. + public string? PdfTemporaryDirectory { get; set; } + /// Maximum simultaneous PDF reads/renders per shared processor, including source buffering. public int MaximumConcurrentPdfOperations { get; set; } = FileContextDefaults.MaximumConcurrentPdfOperations; @@ -81,6 +92,16 @@ internal void Validate() { ValidatePositive(MaximumGeneratedFileBytes, nameof(MaximumGeneratedFileBytes)); ValidatePositive(MaximumPdfReadBytes, nameof(MaximumPdfReadBytes)); + ValidatePositive(PdfSourceBufferBytes, nameof(PdfSourceBufferBytes)); + if (PdfSourceStagingMode is not FileContextPdfSourceStagingMode.Automatic + and not FileContextPdfSourceStagingMode.TemporaryFile) + { + throw new InvalidOperationException("The PDF source staging mode is invalid."); + } + if (PdfTemporaryDirectory is not null && string.IsNullOrWhiteSpace(PdfTemporaryDirectory)) + { + throw new InvalidOperationException("The PDF temporary directory must be a nonempty path or null."); + } ValidatePositive(MaximumConcurrentPdfOperations, nameof(MaximumConcurrentPdfOperations)); ValidatePositive(MaximumImageBytes, nameof(MaximumImageBytes)); ValidatePositive(MaximumDecodedPdfImagePixels, nameof(MaximumDecodedPdfImagePixels)); diff --git a/src/ManagedCode.FileContext/Pdf/FileContextPdfSource.cs b/src/ManagedCode.FileContext/Pdf/FileContextPdfSource.cs index ef1d529..31ea925 100644 --- a/src/ManagedCode.FileContext/Pdf/FileContextPdfSource.cs +++ b/src/ManagedCode.FileContext/Pdf/FileContextPdfSource.cs @@ -1,9 +1,10 @@ +using System.Buffers; + namespace ManagedCode.FileContext.Pdf; -/// Owns a bounded seekable PDF source; non-seekable inputs are staged to disk. +/// Owns a bounded local PDF source; cloud inputs are staged before synchronous random reads. public sealed class FileContextPdfSource : IAsyncDisposable { - private const int CopyBufferBytes = 81920; private const string TemporaryFilePrefix = "filecontext-pdf-"; private FileContextPdfSource(Stream stream) => Stream = stream; @@ -24,28 +25,26 @@ public static async Task OpenAsync(Stream source, FileCont if (source.CanSeek) { Validate(source, options); - source.Position = 0; - return new FileContextPdfSource(source); } - staged = CreateTemporaryFile(); - var buffer = new byte[CopyBufferBytes]; - int read; - while ((read = await source.ReadAsync(buffer, cancellationToken).ConfigureAwait(false)) > 0) + else if (!source.CanRead) { - if (staged.Length + read > options.MaximumPdfReadBytes) - { - throw new IOException("The PDF exceeds the read limit."); - } - await staged.WriteAsync(buffer.AsMemory(0, read), cancellationToken).ConfigureAwait(false); + throw new ArgumentException("The PDF source must be readable.", nameof(source)); } + if (options.PdfSourceStagingMode == FileContextPdfSourceStagingMode.Automatic + && source.CanSeek && source is FileStream or MemoryStream) + { + return new FileContextPdfSource(source); + } + staged = CreateTemporaryFile(options); + await CopyAsync(source, staged, options, cancellationToken).ConfigureAwait(false); staged.Position = 0; await source.DisposeAsync().ConfigureAwait(false); return new FileContextPdfSource(staged); } catch { - if (staged is not null) { await staged.DisposeAsync().ConfigureAwait(false); } - await source.DisposeAsync().ConfigureAwait(false); + try { if (staged is not null) { await staged.DisposeAsync().ConfigureAwait(false); } } + finally { await source.DisposeAsync().ConfigureAwait(false); } throw; } } @@ -64,10 +63,43 @@ internal static void Validate(Stream source, FileContextOptions options) source.Position = 0; } - private static FileStream CreateTemporaryFile() => new( - Path.Combine(Path.GetTempPath(), TemporaryFilePrefix + Guid.NewGuid().ToString("N")), - FileMode.CreateNew, FileAccess.ReadWrite, FileShare.None, CopyBufferBytes, - FileOptions.Asynchronous | FileOptions.DeleteOnClose); + private static async Task CopyAsync(Stream source, Stream staged, FileContextOptions options, + CancellationToken cancellationToken) + { + var buffer = ArrayPool.Shared.Rent(options.PdfSourceBufferBytes); + try + { + int read; + while ((read = await source.ReadAsync(buffer.AsMemory(0, options.PdfSourceBufferBytes), cancellationToken) + .ConfigureAwait(false)) > 0) + { + if (staged.Length + read > options.MaximumPdfReadBytes) + { + throw new IOException("The PDF exceeds the read limit."); + } + await staged.WriteAsync(buffer.AsMemory(0, read), cancellationToken).ConfigureAwait(false); + } + } + finally { ArrayPool.Shared.Return(buffer, clearArray: true); } + } + + private static FileStream CreateTemporaryFile(FileContextOptions options) + { + var fileOptions = new FileStreamOptions + { + Mode = FileMode.CreateNew, + Access = FileAccess.ReadWrite, + Share = FileShare.None, + BufferSize = options.PdfSourceBufferBytes, + Options = FileOptions.Asynchronous | FileOptions.DeleteOnClose + }; + if (!OperatingSystem.IsWindows()) + { + fileOptions.UnixCreateMode = UnixFileMode.UserRead | UnixFileMode.UserWrite; + } + return new FileStream(Path.Combine(options.PdfTemporaryDirectory ?? Path.GetTempPath(), + TemporaryFilePrefix + Guid.NewGuid().ToString("N")), fileOptions); + } public ValueTask DisposeAsync() => Stream.DisposeAsync(); } diff --git a/src/ManagedCode.FileContext/Pdf/FileContextPdfSourceStagingMode.cs b/src/ManagedCode.FileContext/Pdf/FileContextPdfSourceStagingMode.cs new file mode 100644 index 0000000..4e6723b --- /dev/null +++ b/src/ManagedCode.FileContext/Pdf/FileContextPdfSourceStagingMode.cs @@ -0,0 +1,11 @@ +namespace ManagedCode.FileContext.Pdf; + +/// Controls where synchronous PDF parsers perform random reads. +public enum FileContextPdfSourceStagingMode +{ + /// Reuse seekable local files or memory; stage other streams asynchronously to disk. + Automatic, + + /// Stage every input to a temporary file, including seekable local sources. + TemporaryFile +} diff --git a/tests/ManagedCode.FileContext.Tests/AsyncOnlySeekablePdfStream.cs b/tests/ManagedCode.FileContext.Tests/AsyncOnlySeekablePdfStream.cs new file mode 100644 index 0000000..e9f7465 --- /dev/null +++ b/tests/ManagedCode.FileContext.Tests/AsyncOnlySeekablePdfStream.cs @@ -0,0 +1,39 @@ +namespace ManagedCode.FileContext.Tests; + +// Real file contents, with the async-read/synchronous-seek separation of a cloud source. +internal sealed class AsyncOnlySeekablePdfStream(FileStream source) : Stream +{ + public bool WasDisposed { get; private set; } + public int MaximumReadRequestBytes { get; private set; } + public CancellationTokenSource? CancelAfterRead { get; set; } + public bool FailRead { get; set; } + public bool HideSeekability { get; set; } + public override bool CanRead => source.CanRead; + public override bool CanSeek => !HideSeekability; + public override bool CanWrite => false; + public override long Length => source.Length; + public override long Position { get => source.Position; set => source.Position = value; } + + public override async ValueTask ReadAsync(Memory buffer, CancellationToken cancellationToken = default) + { + MaximumReadRequestBytes = Math.Max(MaximumReadRequestBytes, buffer.Length); + if (FailRead) { throw new IOException("Source read failed."); } + var count = await source.ReadAsync(buffer, cancellationToken).ConfigureAwait(false); + if (CancelAfterRead is { } cancellation) { await cancellation.CancelAsync().ConfigureAwait(false); } + return count; + } + + public override int Read(byte[] buffer, int offset, int count) => + throw new InvalidOperationException("The parser must never read this remote source synchronously."); + public override long Seek(long offset, SeekOrigin origin) => source.Seek(offset, origin); + public override void Flush() => throw new NotSupportedException(); + public override void SetLength(long value) => throw new NotSupportedException(); + public override void Write(byte[] buffer, int offset, int count) => throw new NotSupportedException(); + + protected override void Dispose(bool disposing) + { + WasDisposed = true; + if (disposing) { source.Dispose(); } + base.Dispose(disposing); + } +} diff --git a/tests/ManagedCode.FileContext.Tests/FileContextPdfSourceStagingTests.cs b/tests/ManagedCode.FileContext.Tests/FileContextPdfSourceStagingTests.cs new file mode 100644 index 0000000..11a1ba1 --- /dev/null +++ b/tests/ManagedCode.FileContext.Tests/FileContextPdfSourceStagingTests.cs @@ -0,0 +1,170 @@ +using ManagedCode.FileContext.Pdf; + +namespace ManagedCode.FileContext.Tests; + +public sealed class FileContextPdfSourceStagingTests +{ + private const int BufferBytes = 4096; + private const long LargeSourceBytes = (8L * 1024 * 1024) + 1; + + [Fact] + public async Task Seekable_cloud_source_is_staged_before_text_and_page_parsing() + { + await using var scope = await TestStorageScope.CreateAsync(); + var options = Options(scope); + var path = Path.Combine(scope.Directory, "scan.pdf"); + await File.WriteAllBytesAsync(path, FileContextPdfTestPdfs.ScannedPages(43)); + var input = new AsyncOnlySeekablePdfStream(File.OpenRead(path)); + var source = await FileContextPdfSource.OpenAsync(input, options); + var file = source.Stream.ShouldBeOfType(); + try + { + input.WasDisposed.ShouldBeTrue(); + input.MaximumReadRequestBytes.ShouldBeLessThanOrEqualTo(BufferBytes); + file.Name.ShouldStartWith(options.PdfTemporaryDirectory!); + if (!OperatingSystem.IsWindows()) + { + File.GetUnixFileMode(file.Name).ShouldBe(UnixFileMode.UserRead | UnixFileMode.UserWrite); + } + FileContextPdfTextExtractor.Extract(file, 100).PageCount.ShouldBe(43); + FileContextPdfImages.RenderPagePng(file, 16).ShouldNotBeEmpty(); + FileContextPdfImages.ExtractPageImagesPng(file, 9).ShouldHaveSingleItem(); + } + finally { await source.DisposeAsync(); } + Directory.GetFiles(options.PdfTemporaryDirectory!).ShouldBeEmpty(); + } + + [Fact] + public async Task Large_cloud_source_uses_bounded_reads_and_preserves_random_access() + { + await using var scope = await TestStorageScope.CreateAsync(); + var options = Options(scope); + var path = Path.Combine(scope.Directory, "large.pdf"); + await using (var output = File.Create(path)) + { + output.SetLength(LargeSourceBytes); + await output.WriteAsync("%PDF-"u8.ToArray()); + output.Position = LargeSourceBytes - 1; + output.WriteByte(byte.MaxValue); + } + var input = new AsyncOnlySeekablePdfStream(File.OpenRead(path)); + await using (var source = await FileContextPdfSource.OpenAsync(input, options)) + { + source.Stream.Length.ShouldBe(LargeSourceBytes); + source.Stream.Seek(-1, SeekOrigin.End); + source.Stream.ReadByte().ShouldBe(byte.MaxValue); + source.Stream.Position = 0; + source.Stream.ReadByte().ShouldBe('%'); + input.MaximumReadRequestBytes.ShouldBe(BufferBytes); + input.WasDisposed.ShouldBeTrue(); + } + Directory.GetFiles(options.PdfTemporaryDirectory!).ShouldBeEmpty(); + } + + [Fact] + public async Task Temporary_file_mode_stages_local_sources_in_the_configured_directory() + { + await using var scope = await TestStorageScope.CreateAsync(); + var options = Options(scope); + options.PdfSourceStagingMode = FileContextPdfSourceStagingMode.TemporaryFile; + var input = new MemoryStream(FileContextPdfTestPdfs.WithPages(["Readable family history with enough words."])); + await using (var source = await FileContextPdfSource.OpenAsync(input, options)) + { + source.Stream.ShouldBeOfType(); + input.CanRead.ShouldBeFalse(); + FileContextPdfTextExtractor.Extract(source.Stream, 100).Text.ShouldContain("family history"); + } + Directory.GetFiles(options.PdfTemporaryDirectory!).ShouldBeEmpty(); + } + + [Fact] + public async Task Local_file_is_reused_without_creating_a_second_source() + { + await using var scope = await TestStorageScope.CreateAsync(); + var options = Options(scope); + var path = Path.Combine(scope.Directory, "local.pdf"); + await File.WriteAllBytesAsync(path, FileContextPdfTestPdfs.ScannedPages(1)); + var input = File.OpenRead(path); + await using (var source = await FileContextPdfSource.OpenAsync(input, options)) + { + source.Stream.ShouldBeSameAs(input); + FileContextPdfImages.RenderPagePng(source.Stream, 1).ShouldNotBeEmpty(); + Directory.GetFiles(options.PdfTemporaryDirectory!).ShouldBeEmpty(); + } + input.CanRead.ShouldBeFalse(); + } + + [Theory] + [InlineData(true)] + [InlineData(false)] + public async Task Limit_failure_disposes_source_and_removes_staged_file(bool seekable) + { + await using var scope = await TestStorageScope.CreateAsync(); + var options = Options(scope); + options.MaximumPdfReadBytes = BufferBytes; + var input = await InputAsync(scope, BufferBytes + 1); + input.HideSeekability = !seekable; + await Should.ThrowAsync(() => FileContextPdfSource.OpenAsync(input, options)); + input.WasDisposed.ShouldBeTrue(); + Directory.GetFiles(options.PdfTemporaryDirectory!).ShouldBeEmpty(); + } + + [Fact] + public async Task Mid_copy_cancellation_disposes_source_and_removes_staged_file() + { + await using var scope = await TestStorageScope.CreateAsync(); + var options = Options(scope); + using var cancellation = new CancellationTokenSource(); + var input = await InputAsync(scope, BufferBytes + 1); + input.CancelAfterRead = cancellation; + await Should.ThrowAsync(() => + FileContextPdfSource.OpenAsync(input, options, cancellation.Token)); + input.WasDisposed.ShouldBeTrue(); + Directory.GetFiles(options.PdfTemporaryDirectory!).ShouldBeEmpty(); + } + + [Fact] + public async Task Read_failure_disposes_source_and_removes_staged_file() + { + await using var scope = await TestStorageScope.CreateAsync(); + var options = Options(scope); + var input = await InputAsync(scope, BufferBytes); + input.FailRead = true; + await Should.ThrowAsync(() => FileContextPdfSource.OpenAsync(input, options)); + input.WasDisposed.ShouldBeTrue(); + Directory.GetFiles(options.PdfTemporaryDirectory!).ShouldBeEmpty(); + } + + [Fact] + public async Task Invalid_staging_options_are_rejected_before_source_reads() + { + await using var scope = await TestStorageScope.CreateAsync(); + var configurations = new FileContextOptions[] + { + new() { PdfSourceBufferBytes = 0 }, + new() { PdfSourceStagingMode = (FileContextPdfSourceStagingMode)int.MaxValue }, + new() { PdfTemporaryDirectory = " " } + }; + foreach (var options in configurations) + { + var input = await InputAsync(scope, 1); + await Should.ThrowAsync(() => FileContextPdfSource.OpenAsync(input, options)); + input.WasDisposed.ShouldBeTrue(); + input.MaximumReadRequestBytes.ShouldBe(0); + } + } + + private static async Task InputAsync(TestStorageScope scope, int bytes) + { + var path = Path.Combine(scope.Directory, Guid.NewGuid().ToString("N")); + await File.WriteAllBytesAsync(path, new byte[bytes]).ConfigureAwait(false); + return new AsyncOnlySeekablePdfStream(File.OpenRead(path)); + } + + private static FileContextOptions Options(TestStorageScope scope) + { + var directory = Path.Combine(scope.Directory, "staged"); + Directory.CreateDirectory(directory); + return new FileContextOptions { PdfTemporaryDirectory = directory, PdfSourceBufferBytes = BufferBytes }; + } +}