diff options
Diffstat (limited to 'src')
| -rw-r--r-- | src/sisudoc/ocda/io_in/read_source_files.d | 15 |
1 files changed, 14 insertions, 1 deletions
diff --git a/src/sisudoc/ocda/io_in/read_source_files.d b/src/sisudoc/ocda/io_in/read_source_files.d index f99bb9b..4ceb039 100644 --- a/src/sisudoc/ocda/io_in/read_source_files.d +++ b/src/sisudoc/ocda/io_in/read_source_files.d @@ -188,7 +188,20 @@ template spineRawMarkupContent() { @trusted final private char[][] header0Content1(in string src_text) { // cast(char[]) /+ split string on _first_ match of "^:?A~\s" into [header, content] array/tuple +/ char[][] header_and_content; - auto m = (cast(char[]) src_text).matchFirst(rgx.heading_a); + /+ ↓ a utf-8 byte order mark at the start of the file is kept out of + the split, and so out of the yaml header. Left in, it becomes + part of the first key, so "title:" arrives as "title:" and + every value under that key is silently lost: the document keeps + its text and loses its title. It is not stripped at read time, + because source_txt_str is what source.digest is taken over and + that digest names the file as it sits on disk. + +/ + string _src = src_text; + enum string _bom = ""; + if (_src.length >= _bom.length && _src[0.._bom.length] == _bom) { + _src = _src[_bom.length..$]; + } + auto m = (cast(char[]) _src).matchFirst(rgx.heading_a); header_and_content ~= m.pre; header_and_content ~= m.hit ~ m.post; assert(header_and_content.length == 2, |
