///|
priv struct Package {
archive : @zip.Archive
audit : Audit
xml_nodes : Map[String, Array[Node]]
defaults : Map[String, String]
overrides : Map[String, String]
graph : GraphFacts
}
///|
fn get_part(archive : @zip.Archive, name : String) -> BytesView raise {
match archive.get(name[:]) {
Some(data) => data
None => raise Refused("required part missing: " + name)
}
}
///|
fn load_package(
input : Bytes,
limits : Limits,
enforce_profile? : Bool = true,
output? : Bool = false,
) -> Package raise {
limits.validate()
let package_limit = if output {
limits.output_bytes
} else {
limits.input_bytes
}
let zl = @zip.ReadLimits::default()
.with_package_limit(package_limit)
.with_entry_limit(limits.entry_bytes)
.with_total_limit(limits.total_bytes)
.with_entries_limit(limits.entries)
.with_preserved_source_limit(package_limit * 2)
let archive = @zip.read(input[:], limits=zl) catch {
@zip.ZipError(_, message) =>
raise Incomplete("ZIP parsing did not complete: " + message)
}
let names : Map[String, Bool] = Map([])
let aliases : Map[String, Bool] = Map([])
let ordered : Array[String] = []
for entry in archive.entries() {
let name = entry.name()
check_name(name)
guard !names.contains(name) && !aliases.contains(name.to_lower()) else {
raise Refused("duplicate or case-equivalent part name")
}
names[name] = true
aliases[name.to_lower()] = true
ordered.push(name)
guard @checksum.crc32(entry.data()) == entry.crc32() else {
raise Invalid("ZIP payload CRC mismatch")
}
}
ordered.sort()
let defaults : Map[String, String] = Map([])
let overrides : Map[String, String] = Map([])
let ct_nodes = metadata_nodes(
get_part(archive, "[Content_Types].xml"),
limits,
ct_ns,
"Types",
["Default", "Override"],
)
for i in 1.. t
None =>
match name.rev_split_once(".") {
Some((_, extension)) =>
match defaults.get(extension.to_owned()) {
Some(t) => t
None => raise Refused("part has no content type")
}
None => raise Refused("part has no content type")
}
}
}
let data = get_part(archive, name)
parts.push({ name, content_type: ct, sha256: digest(data), })
if name.has_suffix(".xml") || name.has_suffix(".rels") {
let parsed = read_xml(data, limits)
guard parsed.length() > 0 else { raise Refused("XML root missing") }
total_nodes += parsed.length()
for node in parsed {
total_attributes += node.element.attributes.length()
}
guard total_nodes <= 1000000 && total_attributes <= 1000000 else {
raise Incomplete("package XML aggregate limit")
}
nodes[name] = parsed
}
}
let relationships : Array[Relationship] = []
for name in ordered {
if !name.has_suffix(".rels") {
continue
}
let source = relationship_source(name)
guard source == "/" || names.contains(source) else {
raise Invalid("relationship source missing")
}
let local_ids : Map[String, Bool] = Map([])
let rels = metadata_nodes(
get_part(archive, name),
limits,
rel_ns,
"Relationships",
["Relationship"],
)
for i in 1.. 0 && id.length() <= 1024 && !local_ids.contains(id) else {
raise Invalid("duplicate or invalid local relationship ID")
}
local_ids[id] = true
let kind = attr(node, "Type")
let target = attr(node, "Target")
let mode = attr_opt(node, "TargetMode").unwrap_or("Internal")
guard mode == "Internal" || mode == "External" else {
raise Refused("invalid target mode")
}
let external = mode == "External"
let resolved = if external { "" } else { resolve_target(source, target) }
guard external || names.contains(resolved) else {
raise Invalid("internal target missing")
}
relationships.push({ source, id, kind, target, resolved, external, })
}
}
let mains = relationships.filter(fn(r) {
r.source == "/" && r.kind == office_rel + "officeDocument"
})
guard mains.length() == 1 && !mains[0].external else {
raise Invalid("exactly one internal office main required")
}
let main = mains[0].resolved
let main_ct = parts.filter(fn(p) { p.name == main })[0].content_type
guard !enforce_profile ||
main_ct == xlsm_ct ||
main_ct == xlsx_ct ||
main_ct == docm_ct ||
main_ct == docx_ct else {
raise Refused("unsupported main document type")
}
let graph = build_graph(parts, relationships, nodes)
let findings : Array[Finding] = []
for p in parts {
if p.content_type == vba_ct {
let incoming = graph_incoming(graph, p.name)
findings.push({
rule: "vba-v1",
capability: "VBA",
part: p.name,
evidence: ["content-type:" + vba_ct] +
incoming.map(fn(r) {
"relationship:" + r.source + "#" + r.id + ":" + r.kind
}),
})
}
}
let report : Audit = {
schema: if main_ct == docm_ct || main_ct == docx_ct {
"partsieve.audit.word-dev-v1"
} else {
"partsieve.audit.spike-v1"
},
input_hash: digest(input[:]),
format: if main_ct == xlsm_ct {
"XLSM"
} else if main_ct == xlsx_ct {
"XLSX"
} else if main_ct == docm_ct {
"DOCM"
} else if main_ct == docx_ct {
"DOCX"
} else {
"Unsupported"
},
main_part: main,
coverage: "CompleteWithinProfile",
profile: if main_ct == docm_ct || main_ct == docx_ct {
"simple-word-vba-dev-v1"
} else {
"simple-spreadsheet-spike-v1"
},
checked_rules: [
"bounded-zip-v1", "crc-v1", "opc-subset-v1", "namespace-xml-v1", "vba-v1",
"profile-v1",
],
limitations: [
"Restricted element/part allowlist; formulas, hyperlinks, drawings, controls, signatures and unknown extensions refused.",
"No VBA execution or malicious-code analysis. No visual fidelity claim. Client/SDK checks recorded separately.",
],
parts,
relationships,
findings,
}
let pkg = {
archive,
audit: report,
xml_nodes: nodes,
defaults,
overrides,
graph,
}
if enforce_profile {
require_supported_profile(pkg)
}
pkg
}
///|
fn validate_metadata_attributes(
node : Node,
permitted : Array[String],
) -> Unit raise {
for a in node.element.attributes {
guard a.name.namespace_uri is None && permitted.contains(a.name.local_name) else {
raise Refused("unsupported metadata attribute")
}
}
}
///|
pub fn audit(
input : Bytes,
limits? : Limits = Limits::default(),
) -> Audit raise {
load_package(input, limits).audit
}