HtmlSource reads one HTML resource as one IngestionDocument. A file, a web response, or a blob is one element. A folder is one element per file. The HTML is converted to markdown first, so headings stay in the document. script and style elements are left out. The non-generic source writes Document and StreamMetaData on an ExpandoObject.

Read one file

The whole file is one dynamic row. Document is the parsed HTML. StreamMetaData carries the path and the resource count. The first element is the heading, not the script above it.

string sourceFile = "guide.html";
File.WriteAllText(sourceFile, """
    <!DOCTYPE html>
    <html>
    <body>
    <script>console.log("ignored");</script>
    <style>.hidden { display: none; }</style>
    <h1>Guide</h1>
    <h2>Install</h2>
    <p>Run dotnet add package ETLBox.AI.</p>
    <h2>Other</h2>
    <p>Unrelated heading body.</p>
    </body>
    </html>
    """);

var source = new HtmlSource(sourceFile);
var dest = new MemoryDestination();

source.LinkTo(dest);
Network.Execute(source);

foreach (dynamic row in dest.Data) {
    StreamMetaData metadata = row.StreamMetaData;
    IngestionDocument parsed = row.Document;
    Console.WriteLine($"RequestUri:{metadata.RequestUri}");
    Console.WriteLine($"Identifier:{parsed.Identifier}");
    Console.WriteLine($"First element:{parsed.Sections[0].Elements[0].Text}");
}

//Outputs
//RequestUri:guide.html
//Identifier:guide.html
//First element:Guide

Build the row yourself

ResultSelector gets the parsed document and the StreamMetaData. Fields such as Source are set here.

public class Article
{
    public string Source { get; set; }
    public IngestionDocument Document { get; set; }
}

string sourceFile = "guide.html";
File.WriteAllText(sourceFile, """
    <!DOCTYPE html>
    <html>
    <body>
    <h1>Guide</h1>
    <h2>Install</h2>
    <p>Run dotnet add package ETLBox.AI.</p>
    <h2>Other</h2>
    <p>Unrelated heading body.</p>
    </body>
    </html>
    """);

var source = new HtmlSource<Article>(sourceFile) {
    ResultSelector = (document, metadata) => new Article {
        Source = metadata.RequestUri,
        Document = document
    }
};
var dest = new MemoryDestination<Article>();

source.LinkTo(dest);
Network.Execute(source);

foreach (Article row in dest.Data)
    Console.WriteLine($"Source:{row.Source} Identifier:{row.Document.Identifier}");

//Outputs
//Source:guide.html Identifier:guide.html

Read the document itself

HtmlSource<IngestionDocument> emits the parsed document. The row has no separate property for the path, because Identifier already is that URI.

string sourceFile = "guide.html";
File.WriteAllText(sourceFile, """
    <!DOCTYPE html>
    <html>
    <body>
    <h1>Guide</h1>
    <h2>Install</h2>
    <p>Run dotnet add package ETLBox.AI.</p>
    <h2>Other</h2>
    <p>Unrelated heading body.</p>
    </body>
    </html>
    """);

var source = new HtmlSource<IngestionDocument>(sourceFile);
var dest = new MemoryDestination<IngestionDocument>();

source.LinkTo(dest);
Network.Execute(source);

foreach (IngestionDocument document in dest.Data)
    Console.WriteLine($"Identifier:{document.Identifier}");

//Outputs
//Identifier:guide.html

Split by heading

HeaderChunker follows the headings that came from the HTML. Context is the heading path. Text is the body under that heading. An IngestionDocument row is the document, so DocumentSelector stays empty. The script in the file is not part of a chunk.

string sourceFile = "guide.html";
File.WriteAllText(sourceFile, """
    <!DOCTYPE html>
    <html>
    <body>
    <script>console.log("ignored");</script>
    <style>.hidden { display: none; }</style>
    <h1>Guide</h1>
    <h2>Install</h2>
    <p>Run dotnet add package ETLBox.AI.</p>
    <h2>Other</h2>
    <p>Unrelated heading body.</p>
    </body>
    </html>
    """);

var source = new HtmlSource<IngestionDocument>(sourceFile);
var chunk = new ChunkTransformation<IngestionDocument> {
    Chunker = new HeaderChunker(new IngestionChunkerOptions(TiktokenTokenizer.CreateForModel("gpt-4o")) {
        MaxTokensPerChunk = 512,
        OverlapTokens = 50
    })
};
var dest = new MemoryDestination<Chunk<IngestionDocument>>();

source.LinkTo(chunk);
chunk.LinkTo(dest);
Network.Execute(source);

foreach (Chunk<IngestionDocument> row in dest.Data)
    Console.WriteLine($"Context:{row.Context} Text:{row.Text}");

//Outputs
//Context:Guide > Install Text:Run dotnet add package ETLBox.AI.
//Context:Guide > Other Text:Unrelated heading body.

Split your own type by heading

[ChunkDocument] names the document property. ResultSelector still builds the row, because the source does not fill a custom type on its own.

public class Guide
{
    public string Source { get; set; }
    [ChunkDocument]
    public IngestionDocument Document { get; set; }
}

string sourceFile = "guide.html";
File.WriteAllText(sourceFile, """
    <!DOCTYPE html>
    <html>
    <body>
    <h1>Guide</h1>
    <h2>Install</h2>
    <p>Run dotnet add package ETLBox.AI.</p>
    <h2>Other</h2>
    <p>Unrelated heading body.</p>
    </body>
    </html>
    """);

var source = new HtmlSource<Guide>(sourceFile) {
    ResultSelector = (document, metadata) => new Guide {
        Source = metadata.RequestUri,
        Document = document
    }
};
//No DocumentSelector needed - the attribute defines the document property
var chunk = new ChunkTransformation<Guide> {
    Chunker = new HeaderChunker(new IngestionChunkerOptions(TiktokenTokenizer.CreateForModel("gpt-4o")) {
        MaxTokensPerChunk = 512,
        OverlapTokens = 50
    })
};
var dest = new MemoryDestination<Chunk<Guide>>();

source.LinkTo(chunk);
chunk.LinkTo(dest);
Network.Execute(source);

foreach (Chunk<Guide> row in dest.Data)
    Console.WriteLine($"Source:{row.Source.Source} Context:{row.Context} Text:{row.Text}");

//Outputs
//Source:guide.html Context:Guide > Install Text:Run dotnet add package ETLBox.AI.
//Source:guide.html Context:Guide > Other Text:Unrelated heading body.