w4c-workflows-api/Services/Nodes/Executors/HttpBodyReader.cs
2026-09-13 11:35:17 +03:00

64 lines
2.3 KiB
C#

using System.Text;
namespace w4c_workflows.Services.Nodes.Executors;
/// <summary>Raw response body plus a "too large" signal, before content interpretation.</summary>
internal sealed record BodyRead(string? Body, byte[]? Bytes, bool TooLarge);
/// <summary>
/// Shared HTTP response-body reader that enforces the per-run response-size quota.
/// A declared Content-Length over the cap is rejected before reading; otherwise
/// the stream is copied in chunks and rejected as soon as the cap is crossed, so
/// an oversized (or unbounded) body never lands in memory. Both the decoded text
/// and the raw bytes are returned, because <c>responseFormat: file</c> needs the
/// exact bytes rather than a lossy string.
/// </summary>
internal static class HttpBodyReader
{
public static async Task<BodyRead> ReadAsync(
HttpResponseMessage response, long maxBytes, CancellationToken ct)
{
if (maxBytes <= 0)
{
var bytes = await response.Content.ReadAsByteArrayAsync(ct);
return new BodyRead(ResponseEncoding(response).GetString(bytes), bytes, false);
}
if (response.Content.Headers.ContentLength is long declared && declared > maxBytes)
return new BodyRead(null, null, true);
await using var stream = await response.Content.ReadAsStreamAsync(ct);
using var buffer = new MemoryStream();
var chunk = new byte[81_920];
int read;
while ((read = await stream.ReadAsync(chunk, ct)) > 0)
{
if (buffer.Length + read > maxBytes)
return new BodyRead(null, null, true);
buffer.Write(chunk, 0, read);
}
var raw = buffer.ToArray();
return new BodyRead(ResponseEncoding(response).GetString(raw), raw, false);
}
/// <summary>Content-Type charset when the server declares one, else UTF-8.</summary>
private static Encoding ResponseEncoding(HttpResponseMessage response)
{
var charset = response.Content.Headers.ContentType?.CharSet;
if (!string.IsNullOrWhiteSpace(charset))
{
try
{
return Encoding.GetEncoding(charset.Trim('"'));
}
catch (ArgumentException)
{
// Unknown charset: fall back to UTF-8 rather than failing the node.
}
}
return Encoding.UTF8;
}
}