///|
fn feature_count_digits(text : String) -> Int {
let mut count = 0
for c in text {
if c.is_ascii_digit() {
count += 1
}
}
count
}
///|
fn feature_count_ascii_letters(text : String) -> Int {
let mut count = 0
for c in text {
if c.is_ascii_alphabetic() {
count += 1
}
}
count
}
///|
fn feature_count_non_ascii(text : String) -> Int {
let mut count = 0
for c in text {
if !c.is_ascii() && !c.is_whitespace() {
count += 1
}
}
count
}
///|
fn feature_count_punctuation(text : String) -> Int {
let mut count = 0
for c in text {
if c.is_ascii_punctuation() ||
is_sentence_boundary(c) ||
c == ',' ||
c == '。' ||
c == '、' ||
c == ':' {
count += 1
}
}
count
}
///|
fn feature_count_whitespace(text : String) -> Int {
let mut count = 0
for c in text {
if c.is_whitespace() {
count += 1
}
}
count
}
///|
fn feature_count_uppercase(text : String) -> Int {
let mut count = 0
for c in text {
if c.is_ascii_uppercase() {
count += 1
}
}
count
}
///|
fn feature_count_lines(text : String) -> Int {
if text.is_empty() {
0
} else {
text.split("\n").length()
}
}
///|
fn feature_safe_divide(left : Double, right : Double) -> Double {
if right.abs() < 0.000001 {
0.0
} else {
left / right
}
}
///|
fn feature_balance(left : Double, right : Double) -> Double {
let total = left.abs() + right.abs()
if total < 0.000001 {
1.0
} else {
1.0 - (left - right).abs() / total
}
}
///|
fn feature_move_one_hot(pair : AlignmentPair, move_kind : String) -> Double {
if pair.move_kind == move_kind {
1.0
} else {
0.0
}
}
///|
/// Names for the interpretable alignment feature vector.
pub fn alignment_feature_names() -> Array[String] {
[
"source_span", "target_span", "span_balance", "source_weight", "target_weight",
"weight_total", "weight_ratio", "weight_balance", "source_tokens", "target_tokens",
"token_ratio", "token_balance", "source_density", "target_density", "score",
"score_confidence", "move_1_1", "move_1_2", "move_2_1", "source_merged", "target_merged",
"span_product", "source_start", "target_start",
]
}
///|
/// Extract interpretable structural and scoring features from one pair.
pub fn alignment_feature_vector(pair : AlignmentPair) -> Array[Double] {
let source_span = (pair.source_end - pair.source_start).to_double()
let target_span = (pair.target_end - pair.target_start).to_double()
let source_weight = pair.source_char_weight
let target_weight = pair.target_char_weight
let source_tokens = pair.source_tokens.to_double()
let target_tokens = pair.target_tokens.to_double()
let source_density = feature_safe_divide(source_weight, source_tokens)
let target_density = feature_safe_divide(target_weight, target_tokens)
[
source_span,
target_span,
feature_balance(source_span, target_span),
source_weight,
target_weight,
source_weight + target_weight,
feature_safe_divide(target_weight, source_weight),
feature_balance(source_weight, target_weight),
source_tokens,
target_tokens,
feature_safe_divide(target_tokens, source_tokens),
feature_balance(source_tokens, target_tokens),
source_density,
target_density,
pair.score,
1.0 / (1.0 + pair.score.max(0.0)),
feature_move_one_hot(pair, "1-1"),
feature_move_one_hot(pair, "1-2"),
feature_move_one_hot(pair, "2-1"),
if source_span > 1.0 {
1.0
} else {
0.0
},
if target_span > 1.0 {
1.0
} else {
0.0
},
source_span * target_span,
pair.source_start.to_double(),
pair.target_start.to_double(),
]
}
///|
/// Names for the interpretable text-unit feature vector.
pub fn text_feature_names() -> Array[String] {
[
"char_weight", "token_count", "char_length", "weight_per_char", "weight_per_token",
"paragraph_index", "sentence_index", "digit_count", "ascii_letter_count", "non_ascii_count",
"punctuation_count", "whitespace_count", "uppercase_count", "line_count", "has_url",
"has_identifier", "is_short", "has_newline", "ascii_ratio", "cjk_ratio",
]
}
///|
/// Extract lexical, layout, and position features from one text unit.
pub fn text_feature_vector(unit : TextUnit) -> Array[Double] {
let text = unit.normalized
let char_length = text.char_length().to_double()
let token_count = unit.token_count.to_double()
let digit_count = feature_count_digits(text).to_double()
let ascii_letters = feature_count_ascii_letters(text).to_double()
let non_ascii = feature_count_non_ascii(text).to_double()
let punctuation = feature_count_punctuation(text).to_double()
let whitespace = feature_count_whitespace(text).to_double()
[
unit.char_weight,
token_count,
char_length,
feature_safe_divide(unit.char_weight, char_length),
feature_safe_divide(unit.char_weight, token_count),
unit.paragraph_index.to_double(),
unit.sentence_index.to_double(),
digit_count,
ascii_letters,
non_ascii,
punctuation,
whitespace,
feature_count_uppercase(text).to_double(),
feature_count_lines(text).to_double(),
if text.contains("http://") || text.contains("https://") {
1.0
} else {
0.0
},
if text.contains("_") || text.contains("/") {
1.0
} else {
0.0
},
if unit.char_weight < 2.0 {
1.0
} else {
0.0
},
if text.contains("\n") {
1.0
} else {
0.0
},
feature_safe_divide(ascii_letters, char_length),
feature_safe_divide(non_ascii, char_length),
]
}
///|
/// Return the mean value of the structural alignment features.
pub fn alignment_feature_mean(pair : AlignmentPair) -> Double {
let features = alignment_feature_vector(pair)
features.fold(init=0.0, (total, value) => total + value) /
features.length().to_double()
}
///|
/// Return the largest lexical/layout feature for a text unit.
pub fn text_feature_peak(unit : TextUnit) -> Double {
let features = text_feature_vector(unit)
features.fold(init=features[0], (largest, value) => largest.max(value))
}
///|
/// Convert a named feature vector into a stable CSV row.
pub fn feature_vector_to_csv(values : Array[Double]) -> String {
values.map(fn(value) { "\{value}" }).join(",")
}
///|
/// Render the feature schema and values as a reviewable CSV record.
pub fn named_feature_vector_to_csv(
names : Array[String],
values : Array[Double],
) -> String {
let width = names.length().min(values.length())
let fields = []
for i in 0..