Files
file-server/test/runtests.jl
Jeffrey Ward d9f32d9aaf Add stage-5 unknown-format discovery: header clustering + calibration
Implements phase A of the DESIGN_clustering.md design: a Dirichlet-process
mixture of per-position categoricals over the first 32 header bytes (257-symbol
alphabet) that clusters the binary/ pile by file format, plus signature
extraction and promotion nomination. All base-Julia (a Lanczos loggamma keeps
the Dirichlet-multinomial marginal dependency-free).

- src/cluster.jl: header_symbols feature extraction, collapsed Gibbs sampler
  (phase A), sequential CRP-predictive assignment (phase B core), signatures/
  promotion, and ARI/V-measure calibration metrics.
- bin/cluster_calibrate.jl: grid-tunes hyperparameters against magic-collapsed
  ground truth and cross-checks a model-free NCD (gzip) baseline.
- FS_CLUSTER_*/FS_PROMOTE_* config knobs; wire cluster.jl into the module.
- Tests for the three DESIGN §10 assertions plus the model primitives.

Calibrated defaults (n=32, alpha=1.0, beta=0.1) recover known formats at
ARI 0.77 (0.885 excl. tar); docx+zip and the ELF family merge correctly and the
NCD baseline agrees. DESIGN §11 records the results and three assumptions the
data corrected (tar/ELF header-zero merge, the cold-start seeding deadlock, and
the Bernoulli signature / Occam-penalized restart scoring).
2026-07-03 16:43:52 -04:00

529 lines
24 KiB
Julia
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

using Test
using FileServer
using JSON3
# Pull internals into scope. These aren't exported (only `run` is), but the
# whole risk profile of this pipeline lives in these functions, so we test them
# directly rather than only through the HTTP surface.
using FileServer: Job, Config, ChannelQueue, enqueue!, dequeue!, length,
sanitize_filename, recover_dir!, normalize_metadata,
build_metadata, finalize_known!, run_exiftool,
is_binary, handle_unknown_job,
detect_natural_language, run_linguist, detect_programming_language,
read_text_sample, build_text_metadata, finalize_text!, handle_text_job,
linguist_available,
header_symbols, header_matrix, ClusterStats, add!, remove!,
log_predictive, loggamma, gibbs_cluster, assign_file,
signature, magic_positions, is_promotable,
adjusted_rand_index, v_measure, HEADER_N, ALPHABET, PAST_EOF
using Random: MersenneTwister
using Languages: LanguageDetector
# A minimal, valid 1×1 PNG. Lets the real-exiftool tests assert stable facts
# (FileType == "PNG", 1×1 dimensions) that don't drift across exiftool versions.
const PNG_1x1 = UInt8[137,80,78,71,13,10,26,10,0,0,0,13,73,72,68,82,0,0,0,1,0,
0,0,1,8,6,0,0,0,31,21,196,137,0,0,0,11,73,68,65,84,120,218,99,100,248,255,
191,30,0,5,132,2,127,194,91,30,42,0,0,0,0,73,69,78,68,174,66,96,130]
"Build a Config whose data dirs all live under a fresh temp directory."
function tmp_config(root; kwargs...)
cfg = Config(;
spool_dir = joinpath(root, "spool"),
known_dir = joinpath(root, "known"),
unknown_dir = joinpath(root, "unknown"),
binary_dir = joinpath(root, "binary"),
text_dir = joinpath(root, "text"),
done_dir = joinpath(root, "done"),
text_done_dir = joinpath(root, "text_done"),
failed_dir = joinpath(root, "failed"),
kwargs...,
)
FileServer.ensure_dirs(cfg)
return cfg
end
@testset "FileServer" begin
@testset "sanitize_filename" begin
@test sanitize_filename("report.pdf") == "report.pdf"
# Directory components and traversal are stripped, not preserved.
@test sanitize_filename("../../etc/passwd") == "passwd"
@test sanitize_filename("/abs/path/x.txt") == "x.txt"
# Leading dots removed so "..", ".hidden" can't sneak through.
@test sanitize_filename("..") == "unnamed"
@test sanitize_filename(".hidden") == "hidden"
# Unsafe chars collapse to underscores; empty falls back to "unnamed".
@test sanitize_filename("a b&c*.d") == "a_b_c_.d"
@test sanitize_filename("") == "unnamed"
# Length is capped.
@test Base.length(sanitize_filename("a"^500)) == FileServer.MAX_NAME_LEN
end
@testset "normalize_metadata" begin
job = Job("id-1", "photo.jpg", "/data/known/id-1-photo.jpg", 4242, 0.0)
# Group-prefixed tags as exiftool -G emits them are already group-stripped
# by run_exiftool before reaching normalize_metadata, so keys are bare.
bytag = Dict{String,Any}(
"FileType" => "JPEG",
"MIMEType" => "image/jpeg",
"ImageWidth" => 800,
"ImageHeight"=> 600,
"Author" => "Ada Lovelace",
"Creator" => "Acrobat", # feeds created_by, not author
"CreateDate" => "2020:01:02 03:04:05",
"ModifyDate" => "2020:01:02 03:04:06",
"PageCount" => 12,
)
m = normalize_metadata(job, bytag)
@test m.file_type == "JPEG"
@test m.mime_type == "image/jpeg"
@test m.dimensions == (width = 800, height = 600)
@test m.author == "Ada Lovelace"
@test m.created_by == "Acrobat"
@test m.created_date == "2020:01:02 03:04:05"
@test m.page_count == 12
@test m.error === nothing
@test m.raw === bytag
# file_size is authoritative from the Job, never from exiftool.
@test m.file_size == 4242
end
@testset "normalize_metadata: missing tags degrade to nothing" begin
job = Job("id-2", "blob.bin", "/data/known/id-2-blob.bin", 7, 0.0)
m = normalize_metadata(job, Dict{String,Any}())
@test m.file_type === nothing
@test m.dimensions === nothing # neither width nor height present
@test m.author === nothing
@test m.file_size == 7
@test m.error === nothing # empty-but-present dict is still "success"
end
@testset "build_metadata: degraded on extraction failure" begin
mktempdir() do root
cfg = tmp_config(root; exiftool_timeout=5)
# Point at a nonexistent file → exiftool exits non-zero → degraded.
job = Job("id-3", "gone.dat", joinpath(cfg.known_dir, "id-3-gone.dat"), 99, 0.0)
m = build_metadata(job, cfg)
@test m.error !== nothing
@test m.file_type === nothing
@test m.raw === nothing
@test m.file_size == 99 # still authoritative from the Job
@test m.id == "id-3"
end
end
@testset "run_exiftool: real extraction on a PNG" begin
mktempdir() do root
p = joinpath(root, "pixel.png")
write(p, PNG_1x1)
bytag = run_exiftool(p, 30)
@test bytag !== nothing
@test bytag["FileType"] == "PNG"
@test bytag["ImageWidth"] == 1
@test bytag["ImageHeight"] == 1
end
end
@testset "finalize_known!: sidecar-first commit, end to end" begin
mktempdir() do root
cfg = tmp_config(root)
# A real known-stage file to enrich.
src = joinpath(cfg.known_dir, "id-9-pixel.png")
write(src, PNG_1x1)
job = Job("id-9", "pixel.png", src, Base.length(PNG_1x1), 0.0)
meta = build_metadata(job, cfg)
file_dest, sidecar = finalize_known!(cfg, job, meta)
# File moved into done/, original gone from known/.
@test isfile(file_dest)
@test dirname(file_dest) == cfg.done_dir
@test !isfile(src)
# Sidecar committed alongside it, valid JSON, no leftover .tmp.
@test isfile(sidecar)
@test endswith(sidecar, ".meta.json")
@test !isfile(string(sidecar, ".tmp"))
parsed = JSON3.read(read(sidecar, String))
@test parsed.file_type == "PNG"
@test parsed.file_size == Base.length(PNG_1x1)
@test parsed.error === nothing
end
end
@testset "is_binary: UTF-8 sniff" begin
mktempdir() do root
# Plain ASCII text → text.
txt = joinpath(root, "notes.txt")
write(txt, "hello, world\nsecond line\n")
@test is_binary(txt) == false
# Non-ASCII UTF-8 (accents, CJK, emoji) is valid text — the whole
# point of moving off the printable-ASCII/NUL heuristic.
uni = joinpath(root, "unicode.txt")
write(uni, "café — 日本語 — 🚀\n")
@test is_binary(uni) == false
# ANSI-colored log: ESC + other text control bytes are text-safe.
ansi = joinpath(root, "colored.log")
write(ansi, "\e[31merror\e[0m: tab\there\r\nnext\n")
@test is_binary(ansi) == false
# A NUL byte anywhere in the sniff window → binary (it's a control
# byte outside the text-safe set, even though it's valid UTF-8).
bin = joinpath(root, "blob.dat")
write(bin, UInt8[0x01, 0x02, 0x00, 0x03])
@test is_binary(bin) == true
# A non-NUL, non-text control byte (e.g. 0x07 BEL) → binary.
ctrl = joinpath(root, "ctrl.dat")
write(ctrl, UInt8[UInt8('h'), UInt8('i'), 0x07])
@test is_binary(ctrl) == true
# Malformed UTF-8 (lone continuation / bad lead byte) → binary.
bad = joinpath(root, "bad.dat")
write(bad, UInt8[UInt8('a'), 0xff, 0xfe, 0xc3, 0x28])
@test is_binary(bad) == true
# A multi-byte char split by the sniff boundary must NOT read as
# binary: pad to one byte short of the window, then a 2-byte 'é'
# (0xc3 0xa9) so only its lead byte lands inside the window.
split = joinpath(root, "split.txt")
write(split, vcat(fill(UInt8('a'), FileServer.CONTENT_SNIFF_BYTES - 1),
UInt8[0xc3, 0xa9]))
@test is_binary(split) == false
# Empty file → treated as text.
empty = joinpath(root, "empty")
write(empty, UInt8[])
@test is_binary(empty) == false
# Binary garbage past the sniff window is not seen → still text.
far = joinpath(root, "far.txt")
write(far, vcat(fill(UInt8('a'), FileServer.CONTENT_SNIFF_BYTES), UInt8[0x00]))
@test is_binary(far) == false
end
end
@testset "handle_unknown_job: binary terminal, text routed to stage 4" begin
mktempdir() do root
cfg = tmp_config(root)
text_queue = ChannelQueue(10)
# A binary file (embedded NUL) lands in binary/ and is NOT enqueued.
bpath = joinpath(cfg.unknown_dir, "id-b-blob.dat")
write(bpath, UInt8[0x00, 0xFF, 0x10])
bjob = Job("id-b", "blob.dat", bpath, filesize(bpath), 0.0)
handle_unknown_job(bjob, cfg, 1, text_queue)
@test isfile(joinpath(cfg.binary_dir, "id-b-blob.dat"))
@test !isfile(bpath)
@test length(text_queue) == 0
# A text file lands in text/ AND is routed onto the stage-4 queue,
# with its path updated to the new text/ location.
tpath = joinpath(cfg.unknown_dir, "id-t-notes.log")
write(tpath, "just some log text\n")
tjob = Job("id-t", "notes.log", tpath, filesize(tpath), 0.0)
handle_unknown_job(tjob, cfg, 1, text_queue)
moved = joinpath(cfg.text_dir, "id-t-notes.log")
@test isfile(moved)
@test !isfile(tpath)
@test length(text_queue) == 1
routed = dequeue!(text_queue)
@test routed.id == "id-t"
@test routed.path == moved
end
end
@testset "detect_natural_language" begin
d = LanguageDetector()
name, code, conf = detect_natural_language(d,
"The quick brown fox jumps over the lazy dog and then runs away quickly today.")
@test name == "English"
@test code == "eng"
@test conf isa Real && 0.0 <= conf <= 1.0
# Empty / whitespace-only text yields no result rather than throwing
# (the detector itself errors on empty input).
@test detect_natural_language(d, "") == (nothing, nothing, nothing)
@test detect_natural_language(d, " \n\t ") == (nothing, nothing, nothing)
end
@testset "read_text_sample: bounded, UTF-8 safe" begin
mktempdir() do root
p = joinpath(root, "notes.txt")
write(p, "café — 日本語 — hello\n")
@test read_text_sample(p) == "café — 日本語 — hello\n"
# Reads at most LANG_SAMPLE_BYTES, and doesn't choke on a multi-byte
# char straddling that boundary (trailing 'é' half-in the window).
big = joinpath(root, "big.txt")
write(big, vcat(fill(UInt8('a'), FileServer.LANG_SAMPLE_BYTES - 1),
UInt8[0xc3, 0xa9])) # 'é' split by the edge
s = read_text_sample(big)
@test Base.length(s) == FileServer.LANG_SAMPLE_BYTES - 1 # trailing half-char trimmed
@test all(==('a'), s)
end
end
@testset "run_linguist: real detection on source vs. prose" begin
if !linguist_available()
@info "github-linguist not on PATH; skipping run_linguist tests"
else
mktempdir() do root
# A Python source file → linguist names the language.
py = joinpath(root, "script.py")
write(py, "import sys\ndef main():\n print('hi')\nmain()\n")
@test run_linguist(py, 30) == "Python"
# Plain prose reports as "Text", which collapses to nothing.
prose = joinpath(root, "notes.txt")
write(prose, "The quarterly report shows steady growth this year.\n")
@test run_linguist(prose, 30) === nothing
end
end
end
@testset "build_text_metadata + finalize_text!: end to end" begin
mktempdir() do root
cfg = tmp_config(root)
d = LanguageDetector()
src = joinpath(cfg.text_dir, "id-x-script.py")
write(src, join(["# a short program in English prose comment",
"import sys",
"def greet(name):",
" print('hello ' + name + ' welcome to the show today')",
"greet('world')", ""], "\n"))
job = Job("id-x", "script.py", src, filesize(src), 0.0)
meta = build_text_metadata(d, job, cfg)
@test meta.id == "id-x"
@test meta.content_type == "text"
@test meta.file_size == filesize(src)
@test meta.language !== nothing # some natural language detected
@test meta.error === nothing
# programming_language is best-effort; present only when linguist is.
if linguist_available()
@test meta.programming_language == "Python"
end
file_dest, sidecar = finalize_text!(cfg, job, meta)
# File moved into text_done/, original gone from text/.
@test isfile(file_dest)
@test dirname(file_dest) == cfg.text_done_dir
@test !isfile(src)
# Sidecar committed alongside it, valid JSON, no leftover .tmp.
@test isfile(sidecar)
@test endswith(sidecar, ".meta.json")
@test !isfile(string(sidecar, ".tmp"))
parsed = JSON3.read(read(sidecar, String))
@test parsed.content_type == "text"
@test parsed.file_size == filesize(file_dest)
end
end
@testset "handle_text_job: enriches and commits to text_done/" begin
mktempdir() do root
cfg = tmp_config(root)
d = LanguageDetector()
src = joinpath(cfg.text_dir, "id-h-readme.md")
write(src, "# Project\n\nThis project does something useful and interesting for everyone.\n")
job = Job("id-h", "readme.md", src, filesize(src), 0.0)
handle_text_job(job, cfg, 1, d)
@test isfile(joinpath(cfg.text_done_dir, "id-h-readme.md"))
@test isfile(joinpath(cfg.text_done_dir, "id-h-readme.md.meta.json"))
@test !isfile(src)
end
end
@testset "cluster: header_symbols feature extraction" begin
mktempdir() do root
# Bytes map to 1-based symbols (b -> b+1); positions past EOF -> PAST_EOF.
p = joinpath(root, "f.bin")
write(p, UInt8[0x00, 0x7f, 0xff])
s = header_symbols(p; n=6)
@test s[1:3] == [1, 128, 256] # 0->1, 0x7f->128, 0xff->256
@test all(==(PAST_EOF), s[4:6]) # 3 bytes short of n=6 -> past EOF
@test PAST_EOF == ALPHABET == 257
@test Base.length(header_symbols(p)) == HEADER_N
# An empty file is all past-EOF (real signal, not an error).
e = joinpath(root, "empty"); write(e, UInt8[])
@test all(==(PAST_EOF), header_symbols(e; n=8))
# header_matrix stacks one column per file.
q = joinpath(root, "g.bin"); write(q, UInt8[0x41, 0x42])
X = header_matrix([p, q]; n=4)
@test size(X) == (4, 2)
@test X[:, 2] == [0x42, 0x43, PAST_EOF, PAST_EOF] # 'A'->66,'B'->67
end
end
@testset "cluster: loggamma matches known values" begin
@test loggamma(1.0) 0.0 atol=1e-10
@test loggamma(2.0) 0.0 atol=1e-10
@test loggamma(5.0) log(24) atol=1e-10 # Γ(5) = 4! = 24
@test loggamma(0.5) 0.5log(π) atol=1e-10 # Γ(1/2) = √π
@test loggamma(10.0) log(362880) atol=1e-8 # Γ(10) = 9!
end
@testset "cluster: sufficient stats and predictive" begin
c = ClusterStats(3)
x = [10, 20, 30]
# Empty cluster's predictive equals the uniform prior (1/ALPHABET)^n.
@test log_predictive(c, x, 0.5) -3 * log(ALPHABET) atol=1e-9
# add! then remove! is an exact round-trip back to empty.
add!(c, x); remove!(c, x)
@test c.members == 0
@test all(==(0), c.counts)
# A cluster holding a matching point scores it far above uniform.
add!(c, x)
@test log_predictive(c, x, 0.5) > -3 * log(ALPHABET)
end
@testset "cluster: ARI and V-measure" begin
# Identical labelings (up to relabeling) score 1.0.
@test adjusted_rand_index([1,1,2,2], [7,7,9,9]) 1.0
@test adjusted_rand_index(["a","a","b"], ["b","b","a"]) 1.0
v, h, comp = v_measure([1,1,2,2], [5,5,6,6])
@test v 1.0 && h 1.0 && comp 1.0
# A partition that merges two true classes into one is complete but not
# homogeneous, and ARI drops below 1.
@test adjusted_rand_index([1,1,2,2], [1,1,1,1]) < 1.0
_, h2, comp2 = v_measure([1,1,2,2], [1,1,1,1])
@test comp2 1.0 # everything from each class stays together
@test h2 < 1.0 # but the cluster mixes two classes
end
@testset "cluster: signature, magic length, promotability" begin
n = 8
c = ClusterStats(n)
# 30 files sharing bytes 0xDE 0xAD 0xBE 0xEF at positions 1-4, random after.
rng = MersenneTwister(1)
for _ in 1:30
x = vcat([0xDE, 0xAD, 0xBE, 0xEF] .+ 1, rand(rng, 1:256, 4))
add!(c, x)
end
sig = signature(c)
@test sig[1:4] == [0xDE, 0xAD, 0xBE, 0xEF] # spiked -> required bytes
@test all(isnothing, sig[5:8]) # flat -> wildcards
@test magic_positions(sig) == 4
@test is_promotable(c, sig; min_members=20, min_magic=3)
# Too few members, or too few magic positions, blocks nomination.
@test !is_promotable(c, sig; min_members=50, min_magic=3)
@test !is_promotable(c, sig; min_members=20, min_magic=5)
end
@testset "cluster: §10.1 discovers nothing from noise" begin
# 25 independent random blobs — the shape of data/binary (structureless
# junk). Correct output: ZERO promoted clusters (random headers never
# form a ≥20-member, ≥3-magic-byte signature). See DESIGN §10.1.
rng = MersenneTwister(20260703)
X = reduce(hcat, [rand(rng, 1:256, HEADER_N) for _ in 1:25])
r = gibbs_cluster(X; α=1.0, β=0.1, bg_mass=5.0, sweeps=60, restarts=3,
rng=MersenneTwister(1))
promoted = count(c -> is_promotable(c, signature(c); min_members=20, min_magic=3),
values(r.clusters))
@test promoted == 0
# And a lone structured file (a singleton, like the giant PDF in the pile)
# never promotes on its own: N=1 < min_members.
one = ClusterStats(HEADER_N)
add!(one, vcat([0x25,0x50,0x44,0x46] .+ 1, fill(1, HEADER_N - 4)))
@test !is_promotable(one, signature(one); min_members=20, min_magic=3)
end
@testset "cluster: §10.2 recovers known (synthetic) formats" begin
# Four synthetic "formats": a fixed magic prefix + random tail, mirroring
# gzip/PDF/JPEG/ELF. Calibrated settings must recover them as clean,
# promotable clusters at high ARI — the magic-collapsed recovery of §10.2,
# here with a hermetic, deterministic corpus.
# ~12-byte constant headers + random tails — the shape of a real file
# header (a fixed magic/version region, then variable content). A too-short
# magic over a fully-random tail is adversarially hard and lets a format
# over-split; real headers anchor a cluster with ~12+ constant bytes.
rng = MersenneTwister(7)
magics = Dict(
"gzip" => UInt8[0x1f,0x8b,0x08,0x00,0x00,0x00,0x00,0x00,0x00,0x03,0x2d,0x00],
"pdf" => UInt8[0x25,0x50,0x44,0x46,0x2d,0x31,0x2e,0x34,0x0a,0x25,0xe2,0xe3],
"jpeg" => UInt8[0xff,0xd8,0xff,0xe0,0x00,0x10,0x4a,0x46,0x49,0x46,0x00,0x01],
"elf" => UInt8[0x7f,0x45,0x4c,0x46,0x02,0x01,0x01,0x00,0x00,0x00,0x00,0x00],
)
cols = Vector{Int}[]; truth = String[]
for (label, magic) in magics, _ in 1:50
tail = rand(rng, 1:256, HEADER_N - Base.length(magic))
push!(cols, vcat(Int.(magic) .+ 1, tail))
push!(truth, label)
end
X = reduce(hcat, cols)
r = gibbs_cluster(X; α=1.0, β=0.1, bg_mass=5.0, sweeps=120, restarts=6,
rng=MersenneTwister(3))
@test adjusted_rand_index(truth, r.assignments) > 0.9
# Truth breakdown of each cluster, keyed by cluster id.
breakdown(id) = [truth[i] for i in eachindex(r.assignments) if r.assignments[i] == id]
# Nominations cover most formats (a format may over-split below the size
# threshold, but the recovery is not allowed to miss more than one)...
nominated_labels = Set{String}()
for (id, c) in r.clusters
sig = signature(c)
if is_promotable(c, sig; min_members=20, min_magic=3)
# ...and every nomination is PURE — the whole point of the human
# gate is that we never hand it a garbage merged signature.
labels = unique(breakdown(id))
@test Base.length(labels) == 1
push!(nominated_labels, only(labels))
end
end
@test Base.length(nominated_labels) >= 3
end
@testset "cluster: §5B sequential assignment (phase B)" begin
# Build a catalog with one strong cluster (magic 0xCA 0xFE ...).
n = 8
clusters = Dict{Int,ClusterStats}()
c = ClusterStats(n)
rng = MersenneTwister(2)
for _ in 1:40
add!(c, vcat([0xCA,0xFE,0xBA,0xBE] .+ 1, rand(rng, 1:256, 4)))
end
clusters[1] = c
ids = collect(keys(clusters))
# A file that matches the cluster's magic joins it.
match = vcat([0xCA,0xFE,0xBA,0xBE] .+ 1, rand(rng, 1:256, 4))
@test assign_file(match, clusters, ids; α=1.0, β=0.1, bg_mass=5.0) == 1
# A structured-but-novel file (different magic) spawns a new cluster (-1).
novel = vcat([0x12,0x34,0x56,0x78] .+ 1, fill(1, 4))
@test assign_file(novel, clusters, ids; α=1.0, β=0.1, bg_mass=5.0) in (-1, 0)
end
@testset "recover_dir!: re-enqueues work, skips sidecars" begin
mktempdir() do root
dir = joinpath(root, "known"); mkpath(dir)
uuid = "0123456789abcdef0123456789abcdef0123" # 36 chars
work = joinpath(dir, string(uuid, "-report.pdf"))
write(work, "x")
write(joinpath(dir, string(uuid, "-report.pdf.meta.json")), "{}") # sidecar
write(joinpath(dir, "shortname"), "y") # no uuid prefix
q = ChannelQueue(10)
n = recover_dir!(dir, q)
@test n == 2 # the two real files, not the sidecar
@test length(q) == 2
jobs = [dequeue!(q), dequeue!(q)] # sorted by filename on recovery
# "0123...-report.pdf" sorts before "shortname".
@test jobs[1].id == uuid
@test jobs[1].original_name == "report.pdf"
@test jobs[1].path == work
# File with no uuid prefix keeps its whole name; gets a minted id.
@test jobs[2].original_name == "shortname"
@test !isempty(jobs[2].id)
end
end
end