Last active
September 9, 2024 15:05
-
-
Save deanebarker/4815459ca3cd7940247da925caa404da to your computer and use it in GitHub Desktop.
Replace the body of a response with HTML extracted from that same body
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| // Doc here: https://deanebarker.net/tech/code/extract-fragment-middleware/ | |
| // Requires AngleSharp (via Nuget) | |
| using AngleSharp.Html.Parser; | |
| using Microsoft.AspNetCore.Http; | |
| using System.IO; | |
| using System.Linq; | |
| using System.Threading.Tasks; | |
| namespace DeaneBarker.Optimizely.Middleware | |
| { | |
| public class ExtractFragmentMiddleware | |
| { | |
| private static readonly string EXTRACTION_PATH_KEY = "extract"; | |
| private static readonly string SCOPE_KEY = "scope"; | |
| private static HtmlParser _parser; | |
| RequestDelegate _next; | |
| static ExtractFragmentMiddleware() | |
| { | |
| _parser = new HtmlParser(); | |
| } | |
| public ExtractFragmentMiddleware(RequestDelegate next) | |
| { | |
| _next = next; | |
| } | |
| public async Task Invoke(HttpContext context) | |
| { | |
| // If we don't have an extraction path, abandon | |
| if(!HasExtractionPath(context)) | |
| { | |
| await _next.Invoke(context); | |
| return; | |
| } | |
| using (var buffer = new MemoryStream()) | |
| { | |
| /// Start capturing the body | |
| var stream = context.Response.Body; | |
| context.Response.Body = buffer; | |
| // Run the rest of the pipeline | |
| await _next.Invoke(context); | |
| // Process the body | |
| buffer.Seek(0, SeekOrigin.Begin); | |
| var reader = new StreamReader(buffer); | |
| using (var bufferReader = new StreamReader(buffer)) | |
| { | |
| string body = await bufferReader.ReadToEndAsync(); | |
| body = ModifyBody(body, GetExtractionPath(context), GetScope(context)); // This is where the work gets done | |
| context.Response.Clear(); | |
| context.Response.ContentType = "text/html"; // A fair assumption? | |
| await context.Response.WriteAsync(body); | |
| context.Response.Body.Seek(0, SeekOrigin.Begin); | |
| await context.Response.Body.CopyToAsync(stream); | |
| context.Response.Body = stream; | |
| } | |
| } | |
| } | |
| private static bool HasExtractionPath(HttpContext context) | |
| { | |
| return context.Request.Query[EXTRACTION_PATH_KEY].Any(); | |
| } | |
| private static string GetExtractionPath(HttpContext context) | |
| { | |
| return context.Request.Query[EXTRACTION_PATH_KEY].First(); | |
| } | |
| private static string GetScope(HttpContext context) | |
| { | |
| return context.Request.Query[SCOPE_KEY].FirstOrDefault()?.ToLower().Trim(); | |
| } | |
| private static string ModifyBody(string html, string path, string scope) | |
| { | |
| if(!html.Contains("<")) return html; // This isn't parsable HTML... | |
| var doc = _parser.ParseDocument(html); | |
| var elements = doc.QuerySelectorAll(path); | |
| if (!elements.Any() || elements == null) return string.Empty; // Sanity check; not sure if needed | |
| return string.Join(string.Empty, elements.Select(e => scope == "inner" ? e.InnerHtml : e.OuterHtml)); | |
| } | |
| } | |
| } |
Sign up for free
to join this conversation on GitHub.
Already have an account?
Sign in to comment