///|
// Formatted text: the HTML-like inline markup the converter produces
// (asciidoctor-pdf's FormattedText::Parser tag set) parsed into styled
// fragments.
///|
/// Character formatting of a fragment.
priv struct Style {
family : String
size : Double
bold : Bool
italic : Bool
color : Color
/// where the text links to, as Prawn draws it: `` jumps to a
/// named destination, `` is a URI action even when the URI is
/// only a fragment (`link:#name[]`)
link : @pagelayout.LinkTarget?
background : Color?
/// 1 for superscript, -1 for subscript (Prawn raises a superscript by 0.85
/// of its ascender and drops a subscript by its descender)
script : Int
/// padding around the text inside its background (mark, kbd): widens the
/// fragment by twice this, as asciidoctor-pdf's `border_offset` does
border_offset : Double
underline : Bool
strike : Bool
/// a word joiner (`class="wj"`, a footnote's label): the word before it
/// must fit on its line together with it
wj : Bool
/// an inline image (the index of its `InlineImage` in the session), set
/// as a placeholder as wide as the image; -1 for text
image : Int
/// the style of the document font the text is set in (Prawn's current
/// font, as `theme_font` sets it); `bold` and `italic` are the styles of
/// the markup (and those inherited explicitly, as headings do)
doc_bold : Bool
doc_italic : Bool
/// the markup names a font family (``, ``, ``)
font_set : Bool
/// the theme's `text_transform` (`uppercase`, `lowercase`, `capitalize`)
text_transform : String?
} derive(Eq)
///|
fn Style::new(
family? : String = base_font_family(),
size? : Double = base_font_size(),
bold? : Bool = false,
italic? : Bool = false,
doc_bold? : Bool = base_font_style().0,
doc_italic? : Bool = base_font_style().1,
color? : Color = base_font_color(),
) -> Style {
{
family,
size,
bold,
italic,
doc_bold,
doc_italic,
font_set: false,
text_transform: None,
color,
link: None,
background: None,
script: 0,
border_offset: 0.0,
underline: false,
strike: false,
wj: false,
image: -1,
}
}
///|
/// The (bold, italic) face a fragment is set in, as Prawn's arranger picks
/// it (`apply_font_settings`): a fragment whose markup names a font or a
/// bold or italic style takes exactly those styles; any other fragment is
/// set in the document font, whatever its style.
fn Style::face_style(self : Style) -> (Bool, Bool) {
if self.font_set || self.bold || self.italic {
(self.bold, self.italic)
} else {
(self.doc_bold, self.doc_italic)
}
}
///|
/// A run of text in one style. An `anchor` fragment has no text and marks
/// a named destination at its position.
priv struct Fragment {
text : String
style : Style
anchor : String?
}
///|
fn decode_entity(name : String) -> String? {
match name {
"amp" => Some("&")
"lt" => Some("<")
"gt" => Some(">")
"quot" => Some("\"")
"apos" => Some("'")
"nbsp" => Some(" ")
_ =>
if name.has_prefix("#x") || name.has_prefix("#X") {
let hex = name[2:].to_owned()
parse_int_radix(hex, 16).map(code_to_string)
} else if name.has_prefix("#") {
parse_int_radix(name[1:].to_owned(), 10).map(code_to_string)
} else {
None
}
}
}
///|
fn parse_int_radix(text : String, radix : Int) -> Int? {
if text.is_empty() {
return None
}
let mut value = 0
for c in text {
let digit = if c >= '0' && c <= '9' {
c.to_int() - '0'.to_int()
} else if c >= 'a' && c <= 'f' {
c.to_int() - 'a'.to_int() + 10
} else if c >= 'A' && c <= 'F' {
c.to_int() - 'A'.to_int() + 10
} else {
return None
}
if digit >= radix {
return None
}
value = value * radix + digit
}
Some(value)
}
///|
fn code_to_string(code : Int) -> String {
let sb = StringBuilder()
sb.write_char(code.unsafe_to_char())
sb.to_string()
}
///|
/// Decode character references in plain text.
fn decode_entities(text : String) -> String {
if !text.contains("&") {
return text
}
let sb = StringBuilder()
let mut i = 0
let n = text.length()
while i < n {
let c = text[i]
if c == '&' {
let mut j = i + 1
while j < n && j - i < 12 && text[j] != ';' && text[j] != '&' {
j += 1
}
if j < n && text[j] == ';' {
match decode_entity(text[i + 1:j].to_owned()) {
Some(decoded) => {
sb.write_string(decoded)
i = j + 1
continue
}
None => ()
}
}
}
sb.write_char(c.to_int().unsafe_to_char())
i += 1
}
sb.to_string()
}
///|
/// Parse `name="value"` pairs of a start tag.
fn parse_tag_attributes(source : String) -> Map[String, String] {
let attrs : Map[String, String] = Map([])
let mut i = 0
let n = source.length()
while i < n {
while i < n && (source[i] == ' ' || source[i] == '\n' || source[i] == '\t') {
i += 1
}
let name_start = i
while i < n && source[i] != '=' && source[i] != ' ' && source[i] != '/' {
i += 1
}
let name = source[name_start:i].to_owned()
if i < n && source[i] == '=' {
i += 1
if i < n && (source[i] == '"' || source[i] == '\'') {
let quote = source[i]
i += 1
let value_start = i
while i < n && source[i] != quote {
i += 1
}
attrs[name] = decode_entities(source[value_start:i].to_owned())
i += 1
} else {
let value_start = i
while i < n && source[i] != ' ' {
i += 1
}
attrs[name] = source[value_start:i].to_owned()
}
} else if !name.is_empty() {
attrs[name] = ""
} else {
i += 1
}
}
attrs
}
///|
/// Parse `#rrggbb` / `rrggbb`.
fn parse_color(value : String) -> Color? {
let hex = if value.has_prefix("#") { value[1:].to_owned() } else { value }
if hex.length() != 6 {
return None
}
parse_int_radix(hex, 16).map(rgb_hex)
}
///|
/// A font size attribute: `1.2em`, `85%` or points.
fn parse_size(value : String, current : Double) -> Double {
if value.has_suffix("em") {
parse_decimal(value[:value.length() - 2].to_owned()) * current
} else if value.has_suffix("%") {
parse_decimal(value[:value.length() - 1].to_owned()) / 100.0 * current
} else {
parse_decimal(value)
}
}
///|
fn parse_decimal(text : String) -> Double {
@string.parse_double(text) catch {
_ => 0.0
}
}
///|
/// Apply the style of an opening tag.
fn open_tag(tag : String, attrs : Map[String, String], style : Style) -> Style {
let tagged = tag_style(tag, attrs, style)
// then the roles of its classes the theme styles, whatever the tag
// (Transform#build_fragment)
match attrs.get("class") {
Some(classes) => apply_roles(tagged, classes)
None => tagged
}
}
///|
/// The tag names asciidoctor-pdf's markup grammar knows (`tag_name`,
/// `void_tag_name` in formatted_text/parser.treetop), in its order.
let markup_tag_names : Array[String] = [
"a", "strong", "em", "code", "font", "span", "button", "kbd", "sup", "sub", "mark",
"menu", "del",
]
///|
let markup_void_tag_names : Array[String] = ["br", "img"]
///|
/// Match the first of `names` that `s` has at `pos`, as a PEG ordered
/// choice does; the position after it.
fn match_name(s : String, pos : Int, names : Array[String]) -> Int? {
for name in names {
if s[pos:].has_prefix(name) {
return Some(pos + name.length())
}
}
None
}
///|
/// `attributes`: ` name="value"` pairs, names in [a-z_]; the position
/// after them.
fn match_markup_attributes(s : String, pos : Int) -> Int {
let n = s.length()
let mut p = pos
while true {
let mut q = p
while q < n && s[q] == ' ' {
q += 1
}
if q == p {
break
}
let name_start = q
while q < n && ((s[q] >= 'a' && s[q] <= 'z') || s[q] == '_') {
q += 1
}
if q == name_start || q + 1 >= n || s[q] != '=' || s[q + 1] != '"' {
break
}
q += 2
while q < n && s[q] != '"' {
q += 1
}
if q >= n {
break
}
p = q + 1
}
p
}
///|
/// Whether asciidoctor-pdf's markup grammar parses `s`: text, character
/// references (`&`, `’`, `’`), the known void and start
/// tags with double-quoted attributes, and end tags that close whatever
/// element is open. Formatted text it cannot parse is shown as it is,
/// markup and all (Formatter#format).
fn markup_parses(s : String) -> Bool {
let n = s.length()
let mut depth = 0
let mut i = 0
while i < n {
let c = s[i]
if c == '<' {
if i + 1 < n && s[i + 1] == '/' {
// an end tag closes the element open here; at the top level
// there is none
guard match_name(s, i + 2, markup_tag_names) is Some(p) &&
p < n &&
s[p] == '>' &&
depth > 0 else {
return false
}
depth -= 1
i = p + 1
continue
}
match match_name(s, i + 1, markup_void_tag_names) {
Some(p) => {
let mut q = match_markup_attributes(s, p)
let mut r = q
while r < n && s[r] == ' ' {
r += 1
}
if r < n && s[r] == '/' {
q = r + 1
}
if q < n && s[q] == '>' {
i = q + 1
continue
}
}
None => ()
}
guard match_name(s, i + 1, markup_tag_names) is Some(p) else {
return false
}
let q = match_markup_attributes(s, p)
guard q < n && s[q] == '>' else { return false }
depth += 1
i = q + 1
} else if c == '&' {
let mut j = i + 1
let digits = fn(from : Int, hex : Bool) {
let mut k = from
while k < n &&
(
(s[k] >= '0' && s[k] <= '9') ||
(
hex &&
((s[k] >= 'a' && s[k] <= 'f') || (s[k] >= 'A' && s[k] <= 'F'))
)
) {
k += 1
}
k
}
if s[j:].has_prefix("#x") {
let k = digits(j + 2, true)
guard k - (j + 2) >= 2 && k - (j + 2) <= 5 else { return false }
j = k
} else if s[j:].has_prefix("#") {
let k = digits(j + 1, false)
guard k - (j + 1) >= 2 && k - (j + 1) <= 6 else { return false }
j = k
} else {
guard match_name(s, j, ["amp", "apos", "gt", "lt", "nbsp", "quot"])
is Some(k) else {
return false
}
j = k
}
guard j < n && s[j] == ';' else { return false }
i = j + 1
} else {
i += 1
}
}
depth == 0
}
///|
/// Parse formatted text into fragments. `
` becomes a "\n" fragment;
/// with `normalize`, runs of whitespace (including newlines) collapse to
/// one space, as asciidoctor-pdf's formatter does for prose.
fn parse_formatted(
markup : String,
base : Style,
normalize? : Bool = true,
) -> Array[Fragment] {
if !markup_parses(markup) {
// shown as it is, markup and all
let text = if normalize { collapse_whitespace(markup) } else { markup }
return [{ text, style: base, anchor: None, }]
}
let fragments : Array[Fragment] = []
// each open element: its tag, the style outside it, how many fragments
// there were when it opened (-1 for an anchor, which is never empty),
// and the text fragment just before it (-1: none)
let stack : Array[(String, Style, Int, Int)] = []
let mut style = base
let text = StringBuilder()
fn flush() {
if !text.is_empty() {
let raw = decode_entities(text.to_string())
let content = if normalize { collapse_whitespace(raw) } else { raw }
let content = match style.text_transform {
Some(transform) => transform_text(content, transform)
None => content
}
if !content.is_empty() {
fragments.push({ text: content, style, anchor: None, })
}
text.reset()
}
}
let n = markup.length()
let mut i = 0
while i < n {
let c = markup[i]
if c == '<' {
// find the end of the tag
let mut j = i + 1
while j < n && markup[j] != '>' {
j += 1
}
if j >= n {
text.write_char('<')
i += 1
continue
}
let inner = markup[i + 1:j].to_owned()
i = j + 1
if inner.has_prefix("/") {
// an end tag closes the innermost open element, whatever its name
// (the markup grammar matches `start_tag complex end_tag` without
// comparing names)
let k = stack.length() - 1
if k >= 0 {
let (_, _, opened, before) = stack[k]
if opened >= 0 && text.is_empty() && fragments.length() == opened {
// an element with nothing in it takes the space before it
// away (Transform#apply)
if before >= 0 && fragments[before].text.has_suffix(" ") {
let t = fragments[before].text
fragments[before] = {
..fragments[before],
text: t[:t.length() - 1].to_owned(),
}
}
}
flush()
style = stack[k].1
while stack.length() > k {
ignore(stack.pop())
}
}
continue
}
let self_closing = inner.has_suffix("/")
let body = if self_closing {
inner[:inner.length() - 1].to_owned()
} else {
inner
}
let mut name_end = 0
while name_end < body.length() &&
body[name_end] != ' ' &&
body[name_end] != '\n' {
name_end += 1
}
let tag = body[:name_end].to_owned()
let attrs = parse_tag_attributes(body[name_end:].to_owned())
match tag {
"br" => {
flush()
fragments.push({ text: "\n", style, anchor: None, })
}
"a" if attrs.get("id") is Some(id) &&
!attrs.contains("href") &&
!attrs.contains("anchor") => {
// a concealed index term swallows the space before it
// (asciidoctor-pdf's Transform#apply)
if attrs.get("type") == Some("indexterm") &&
!attrs.contains("visible") &&
!text.is_empty() {
let pending = text.to_string()
let mut end = pending.length()
if normalize {
while end > 0 &&
(
pending[end - 1] == ' ' ||
pending[end - 1] == '\n' ||
pending[end - 1] == '\t'
) {
end -= 1
}
} else if pending[end - 1] == ' ' {
end -= 1
}
text.reset()
text.write_string(pending[:end].to_owned())
}
flush()
// a fresh fragment (Transform#build_fragment): none of the
// markup around it applies, so it is in the text's own font and
// size, which count towards its line's height
fragments.push({
text: "",
style: { ..base, wj: style.wj, },
anchor: Some(id),
})
if !self_closing {
stack.push((tag, style, -1, -1))
}
}
"img" => {
// an inline image: a placeholder the size of the image (see
// `inline_image_size`), drawn as the image
let index = match attrs.get("src") {
Some(s) => @string.parse_int(s) catch { _ => -1 }
None => -1
}
if index >= 0 && index < session().inline_images.length() {
flush()
fragments.push({
text: "\u{2063}",
style: { ..style, image: index, },
anchor: None,
})
}
}
_ =>
if !self_closing {
let before = if text.is_empty() { -1 } else { fragments.length() }
flush()
let before = if before >= 0 && fragments.length() > before {
before
} else {
-1
}
stack.push((tag, style, fragments.length(), before))
style = open_tag(tag, attrs, style)
}
}
} else {
text.write_char(c.to_int().unsafe_to_char())
i += 1
}
}
flush()
fragments
}
///|
fn collapse_whitespace(text : String) -> String {
let sb = StringBuilder()
let mut in_space = false
for c in text {
if c == ' ' || c == '\n' || c == '\t' || c == '\r' {
if !in_space {
sb.write_char(' ')
}
in_space = true
} else {
sb.write_char(c)
in_space = false
}
}
sb.to_string()
}