MarkdownSource reads one markdown resource as one IngestionDocument. A file, a web response, or a blob is one element. A folder is one element per file. The non-generic source writes Document and StreamMetaData on an ExpandoObject.

Read one file

The whole file is one dynamic row. Document is the parsed markdown. StreamMetaData carries the path and the resource count. Headings are elements of that document.

string sourceFile = "guide.md";
File.WriteAllText(sourceFile, """
    # Guide

    ## Install

    Run dotnet add package ETLBox.AI.

    ## Other

    Unrelated heading body.
    """);

var source = new MarkdownSource(sourceFile);
var dest = new MemoryDestination();

source.LinkTo(dest);
Network.Execute(source);

foreach (dynamic row in dest.Data) {
    StreamMetaData metadata = row.StreamMetaData;
    IngestionDocument parsed = row.Document;
    Console.WriteLine($"RequestUri:{metadata.RequestUri}");
    Console.WriteLine($"Identifier:{parsed.Identifier}");
    Console.WriteLine($"First element: {parsed.Sections[0].Elements[0].Text}");
    Console.WriteLine($"Second element: {parsed.Sections[0].Elements[1].Text}");
}

//Outputs
//RequestUri:guide.md
//Identifier:guide.md
//First element: Guide
//Second element: Install

Build the row yourself

ResultSelector gets the parsed document and the StreamMetaData. Fields such as Source are set here.

public class Article
{
    public string Source { get; set; }
    public IngestionDocument Document { get; set; }
}

string sourceFile = "guide.md";
File.WriteAllText(sourceFile, """
    # Guide

    ## Install

    Run dotnet add package ETLBox.AI.

    ## Other

    Unrelated heading body.
    """);

var source = new MarkdownSource<Article>(sourceFile) {
    ResultSelector = (document, metadata) => new Article {
        Source = metadata.RequestUri,
        Document = document
    }
};
var dest = new MemoryDestination<Article>();

source.LinkTo(dest);
Network.Execute(source);

foreach (Article row in dest.Data)
    Console.WriteLine($"Source:{row.Source} Identifier:{row.Document.Identifier} Elements:{row.Document.Sections[0].Elements.Count()}");

//Outputs
//Source:guide.md Identifier:guide.md Elements:5

Read the document itself

MarkdownSource<IngestionDocument> emits the parsed document. The row has no separate property for the path, because Identifier already is that URI.

string sourceFile = "guide.md";
File.WriteAllText(sourceFile, """
    # Guide

    ## Install

    Run dotnet add package ETLBox.AI.

    ## Other

    Unrelated heading body.
    """);

var source = new MarkdownSource<IngestionDocument>(sourceFile);
var dest = new MemoryDestination<IngestionDocument>();

source.LinkTo(dest);
Network.Execute(source);

foreach (IngestionDocument document in dest.Data) {
    Console.WriteLine($"Identifier:{document.Identifier}");
    Console.WriteLine($"Elements: {document.Sections[0].Elements.Count()}");
}

//Outputs
//Identifier:guide.md
//Elements: 5

Split by heading

HeaderChunker follows the headings. Context is the heading path. Text is the body under that heading. An IngestionDocument row is the document, so DocumentSelector stays empty.

string sourceFile = "guide.md";
File.WriteAllText(sourceFile, """
    # Guide

    ## Install

    Run dotnet add package ETLBox.AI.

    ## Other

    Unrelated heading body.
    """);

var source = new MarkdownSource<IngestionDocument>(sourceFile);
var chunk = new ChunkTransformation<IngestionDocument> {
    Chunker = new HeaderChunker(new IngestionChunkerOptions(TiktokenTokenizer.CreateForModel("gpt-4o")) {
        MaxTokensPerChunk = 512,
        OverlapTokens = 50
    })
};
var dest = new MemoryDestination<Chunk<IngestionDocument>>();

source.LinkTo(chunk);
chunk.LinkTo(dest);
Network.Execute(source);

foreach (Chunk<IngestionDocument> row in dest.Data)
    Console.WriteLine($"Context:{row.Context} Text:{row.Text}");

//Outputs
//Context:Guide > Install Text:Run dotnet add package ETLBox.AI.
//Context:Guide > Other Text:Unrelated heading body.

Split your own type by heading

[ChunkDocument] names the document property. ResultSelector still builds the row, because the source does not fill a custom type on its own.

public class Guide
{
    public string Source { get; set; }
    [ChunkDocument]
    public IngestionDocument Document { get; set; }
}

string sourceFile = "guide.md";
File.WriteAllText(sourceFile, """
    # Guide

    ## Install

    Run dotnet add package ETLBox.AI.

    ## Other

    Unrelated heading body.
    """);

var source = new MarkdownSource<Guide>(sourceFile) {
    ResultSelector = (document, metadata) => new Guide {
        Source = metadata.RequestUri,
        Document = document
    }
};
//No DocumentSelector needed - the attribute defines the document property
var chunk = new ChunkTransformation<Guide> {
    Chunker = new HeaderChunker(new IngestionChunkerOptions(TiktokenTokenizer.CreateForModel("gpt-4o")) {
        MaxTokensPerChunk = 512,
        OverlapTokens = 50
    })
};
var dest = new MemoryDestination<Chunk<Guide>>();

source.LinkTo(chunk);
chunk.LinkTo(dest);
Network.Execute(source);

foreach (Chunk<Guide> row in dest.Data)
    Console.WriteLine($"Source:{row.Source.Source} Context:{row.Context} Text:{row.Text}");

//Outputs
//Source:guide.md Context:Guide > Install Text:Run dotnet add package ETLBox.AI.
//Source:guide.md Context:Guide > Other Text:Unrelated heading body.