Wraps the assign_file scoring core (cluster.jl) in the live catalog the design's phase B calls for (DESIGN §5B/§9): - src/catalog.jl: Catalog durable state (frozen-id clusters + sufficient stats + processed set + examples); sparse, sidecar-first durable save/load; incremental catalog_sweep! (deterministic CRP-predictive assignment of new binary/ files); offline compact! that seeds on first run and recompacts later; write_nominations! emitting one JSON per promotable cluster with a hex magic template. - bin/cluster_sweep.jl: cron/periodic single-owner runner (--compact forces a recluster; first run auto-compacts to seed). - config.jl: cluster_catalog_path + nominated_dir knobs (FS_CLUSTER_CATALOG, FS_NOMINATED_DIR), wired into config_from_env and ensure_dirs. - Tests: durable round-trip, incremental sweep growth + idempotency, §10.1 nothing-from-noise end-to-end (zero promotions), a recurring format self-nominating, seed-then-live-assign (165 pass). - DESIGN_clustering.md: mark phase B built.
709 lines
32 KiB
Julia
709 lines
32 KiB
Julia
using Test
|
||
using FileServer
|
||
using JSON3
|
||
|
||
# Pull internals into scope. These aren't exported (only `run` is), but the
|
||
# whole risk profile of this pipeline lives in these functions, so we test them
|
||
# directly rather than only through the HTTP surface.
|
||
using FileServer: Job, Config, ChannelQueue, enqueue!, dequeue!, length,
|
||
sanitize_filename, recover_dir!, normalize_metadata,
|
||
build_metadata, finalize_known!, run_exiftool,
|
||
is_binary, handle_unknown_job,
|
||
detect_natural_language, run_linguist, detect_programming_language,
|
||
read_text_sample, build_text_metadata, finalize_text!, handle_text_job,
|
||
linguist_available,
|
||
header_symbols, header_matrix, ClusterStats, add!, remove!,
|
||
log_predictive, loggamma, gibbs_cluster, assign_file,
|
||
signature, magic_positions, is_promotable,
|
||
adjusted_rand_index, v_measure, HEADER_N, ALPHABET, PAST_EOF,
|
||
Catalog, load_catalog, save_catalog!, catalog_sweep!, compact!,
|
||
write_nominations!, run_cluster_sweep, binary_files, record_example!,
|
||
signature_hex, ensure_dirs
|
||
using Random: MersenneTwister
|
||
using Languages: LanguageDetector
|
||
|
||
# A minimal, valid 1×1 PNG. Lets the real-exiftool tests assert stable facts
|
||
# (FileType == "PNG", 1×1 dimensions) that don't drift across exiftool versions.
|
||
const PNG_1x1 = UInt8[137,80,78,71,13,10,26,10,0,0,0,13,73,72,68,82,0,0,0,1,0,
|
||
0,0,1,8,6,0,0,0,31,21,196,137,0,0,0,11,73,68,65,84,120,218,99,100,248,255,
|
||
191,30,0,5,132,2,127,194,91,30,42,0,0,0,0,73,69,78,68,174,66,96,130]
|
||
|
||
"Build a Config whose data dirs all live under a fresh temp directory."
|
||
function tmp_config(root; kwargs...)
|
||
cfg = Config(;
|
||
spool_dir = joinpath(root, "spool"),
|
||
known_dir = joinpath(root, "known"),
|
||
unknown_dir = joinpath(root, "unknown"),
|
||
binary_dir = joinpath(root, "binary"),
|
||
text_dir = joinpath(root, "text"),
|
||
done_dir = joinpath(root, "done"),
|
||
text_done_dir = joinpath(root, "text_done"),
|
||
failed_dir = joinpath(root, "failed"),
|
||
cluster_dir = joinpath(root, "binary"), # stage-5 sweeps the binary sink
|
||
cluster_catalog_path = joinpath(root, "catalog.json"),
|
||
nominated_dir = joinpath(root, "nominated"),
|
||
kwargs...,
|
||
)
|
||
FileServer.ensure_dirs(cfg)
|
||
return cfg
|
||
end
|
||
|
||
@testset "FileServer" begin
|
||
|
||
@testset "sanitize_filename" begin
|
||
@test sanitize_filename("report.pdf") == "report.pdf"
|
||
# Directory components and traversal are stripped, not preserved.
|
||
@test sanitize_filename("../../etc/passwd") == "passwd"
|
||
@test sanitize_filename("/abs/path/x.txt") == "x.txt"
|
||
# Leading dots removed so "..", ".hidden" can't sneak through.
|
||
@test sanitize_filename("..") == "unnamed"
|
||
@test sanitize_filename(".hidden") == "hidden"
|
||
# Unsafe chars collapse to underscores; empty falls back to "unnamed".
|
||
@test sanitize_filename("a b&c*.d") == "a_b_c_.d"
|
||
@test sanitize_filename("") == "unnamed"
|
||
# Length is capped.
|
||
@test Base.length(sanitize_filename("a"^500)) == FileServer.MAX_NAME_LEN
|
||
end
|
||
|
||
@testset "normalize_metadata" begin
|
||
job = Job("id-1", "photo.jpg", "/data/known/id-1-photo.jpg", 4242, 0.0)
|
||
# Group-prefixed tags as exiftool -G emits them are already group-stripped
|
||
# by run_exiftool before reaching normalize_metadata, so keys are bare.
|
||
bytag = Dict{String,Any}(
|
||
"FileType" => "JPEG",
|
||
"MIMEType" => "image/jpeg",
|
||
"ImageWidth" => 800,
|
||
"ImageHeight"=> 600,
|
||
"Author" => "Ada Lovelace",
|
||
"Creator" => "Acrobat", # feeds created_by, not author
|
||
"CreateDate" => "2020:01:02 03:04:05",
|
||
"ModifyDate" => "2020:01:02 03:04:06",
|
||
"PageCount" => 12,
|
||
)
|
||
m = normalize_metadata(job, bytag)
|
||
@test m.file_type == "JPEG"
|
||
@test m.mime_type == "image/jpeg"
|
||
@test m.dimensions == (width = 800, height = 600)
|
||
@test m.author == "Ada Lovelace"
|
||
@test m.created_by == "Acrobat"
|
||
@test m.created_date == "2020:01:02 03:04:05"
|
||
@test m.page_count == 12
|
||
@test m.error === nothing
|
||
@test m.raw === bytag
|
||
# file_size is authoritative from the Job, never from exiftool.
|
||
@test m.file_size == 4242
|
||
end
|
||
|
||
@testset "normalize_metadata: missing tags degrade to nothing" begin
|
||
job = Job("id-2", "blob.bin", "/data/known/id-2-blob.bin", 7, 0.0)
|
||
m = normalize_metadata(job, Dict{String,Any}())
|
||
@test m.file_type === nothing
|
||
@test m.dimensions === nothing # neither width nor height present
|
||
@test m.author === nothing
|
||
@test m.file_size == 7
|
||
@test m.error === nothing # empty-but-present dict is still "success"
|
||
end
|
||
|
||
@testset "build_metadata: degraded on extraction failure" begin
|
||
mktempdir() do root
|
||
cfg = tmp_config(root; exiftool_timeout=5)
|
||
# Point at a nonexistent file → exiftool exits non-zero → degraded.
|
||
job = Job("id-3", "gone.dat", joinpath(cfg.known_dir, "id-3-gone.dat"), 99, 0.0)
|
||
m = build_metadata(job, cfg)
|
||
@test m.error !== nothing
|
||
@test m.file_type === nothing
|
||
@test m.raw === nothing
|
||
@test m.file_size == 99 # still authoritative from the Job
|
||
@test m.id == "id-3"
|
||
end
|
||
end
|
||
|
||
@testset "run_exiftool: real extraction on a PNG" begin
|
||
mktempdir() do root
|
||
p = joinpath(root, "pixel.png")
|
||
write(p, PNG_1x1)
|
||
bytag = run_exiftool(p, 30)
|
||
@test bytag !== nothing
|
||
@test bytag["FileType"] == "PNG"
|
||
@test bytag["ImageWidth"] == 1
|
||
@test bytag["ImageHeight"] == 1
|
||
end
|
||
end
|
||
|
||
@testset "finalize_known!: sidecar-first commit, end to end" begin
|
||
mktempdir() do root
|
||
cfg = tmp_config(root)
|
||
# A real known-stage file to enrich.
|
||
src = joinpath(cfg.known_dir, "id-9-pixel.png")
|
||
write(src, PNG_1x1)
|
||
job = Job("id-9", "pixel.png", src, Base.length(PNG_1x1), 0.0)
|
||
|
||
meta = build_metadata(job, cfg)
|
||
file_dest, sidecar = finalize_known!(cfg, job, meta)
|
||
|
||
# File moved into done/, original gone from known/.
|
||
@test isfile(file_dest)
|
||
@test dirname(file_dest) == cfg.done_dir
|
||
@test !isfile(src)
|
||
|
||
# Sidecar committed alongside it, valid JSON, no leftover .tmp.
|
||
@test isfile(sidecar)
|
||
@test endswith(sidecar, ".meta.json")
|
||
@test !isfile(string(sidecar, ".tmp"))
|
||
parsed = JSON3.read(read(sidecar, String))
|
||
@test parsed.file_type == "PNG"
|
||
@test parsed.file_size == Base.length(PNG_1x1)
|
||
@test parsed.error === nothing
|
||
end
|
||
end
|
||
|
||
@testset "is_binary: UTF-8 sniff" begin
|
||
mktempdir() do root
|
||
# Plain ASCII text → text.
|
||
txt = joinpath(root, "notes.txt")
|
||
write(txt, "hello, world\nsecond line\n")
|
||
@test is_binary(txt) == false
|
||
|
||
# Non-ASCII UTF-8 (accents, CJK, emoji) is valid text — the whole
|
||
# point of moving off the printable-ASCII/NUL heuristic.
|
||
uni = joinpath(root, "unicode.txt")
|
||
write(uni, "café — 日本語 — 🚀\n")
|
||
@test is_binary(uni) == false
|
||
|
||
# ANSI-colored log: ESC + other text control bytes are text-safe.
|
||
ansi = joinpath(root, "colored.log")
|
||
write(ansi, "\e[31merror\e[0m: tab\there\r\nnext\n")
|
||
@test is_binary(ansi) == false
|
||
|
||
# A NUL byte anywhere in the sniff window → binary (it's a control
|
||
# byte outside the text-safe set, even though it's valid UTF-8).
|
||
bin = joinpath(root, "blob.dat")
|
||
write(bin, UInt8[0x01, 0x02, 0x00, 0x03])
|
||
@test is_binary(bin) == true
|
||
|
||
# A non-NUL, non-text control byte (e.g. 0x07 BEL) → binary.
|
||
ctrl = joinpath(root, "ctrl.dat")
|
||
write(ctrl, UInt8[UInt8('h'), UInt8('i'), 0x07])
|
||
@test is_binary(ctrl) == true
|
||
|
||
# Malformed UTF-8 (lone continuation / bad lead byte) → binary.
|
||
bad = joinpath(root, "bad.dat")
|
||
write(bad, UInt8[UInt8('a'), 0xff, 0xfe, 0xc3, 0x28])
|
||
@test is_binary(bad) == true
|
||
|
||
# A multi-byte char split by the sniff boundary must NOT read as
|
||
# binary: pad to one byte short of the window, then a 2-byte 'é'
|
||
# (0xc3 0xa9) so only its lead byte lands inside the window.
|
||
split = joinpath(root, "split.txt")
|
||
write(split, vcat(fill(UInt8('a'), FileServer.CONTENT_SNIFF_BYTES - 1),
|
||
UInt8[0xc3, 0xa9]))
|
||
@test is_binary(split) == false
|
||
|
||
# Empty file → treated as text.
|
||
empty = joinpath(root, "empty")
|
||
write(empty, UInt8[])
|
||
@test is_binary(empty) == false
|
||
|
||
# Binary garbage past the sniff window is not seen → still text.
|
||
far = joinpath(root, "far.txt")
|
||
write(far, vcat(fill(UInt8('a'), FileServer.CONTENT_SNIFF_BYTES), UInt8[0x00]))
|
||
@test is_binary(far) == false
|
||
end
|
||
end
|
||
|
||
@testset "handle_unknown_job: binary terminal, text routed to stage 4" begin
|
||
mktempdir() do root
|
||
cfg = tmp_config(root)
|
||
text_queue = ChannelQueue(10)
|
||
|
||
# A binary file (embedded NUL) lands in binary/ and is NOT enqueued.
|
||
bpath = joinpath(cfg.unknown_dir, "id-b-blob.dat")
|
||
write(bpath, UInt8[0x00, 0xFF, 0x10])
|
||
bjob = Job("id-b", "blob.dat", bpath, filesize(bpath), 0.0)
|
||
handle_unknown_job(bjob, cfg, 1, text_queue)
|
||
@test isfile(joinpath(cfg.binary_dir, "id-b-blob.dat"))
|
||
@test !isfile(bpath)
|
||
@test length(text_queue) == 0
|
||
|
||
# A text file lands in text/ AND is routed onto the stage-4 queue,
|
||
# with its path updated to the new text/ location.
|
||
tpath = joinpath(cfg.unknown_dir, "id-t-notes.log")
|
||
write(tpath, "just some log text\n")
|
||
tjob = Job("id-t", "notes.log", tpath, filesize(tpath), 0.0)
|
||
handle_unknown_job(tjob, cfg, 1, text_queue)
|
||
moved = joinpath(cfg.text_dir, "id-t-notes.log")
|
||
@test isfile(moved)
|
||
@test !isfile(tpath)
|
||
@test length(text_queue) == 1
|
||
routed = dequeue!(text_queue)
|
||
@test routed.id == "id-t"
|
||
@test routed.path == moved
|
||
end
|
||
end
|
||
|
||
@testset "detect_natural_language" begin
|
||
d = LanguageDetector()
|
||
name, code, conf = detect_natural_language(d,
|
||
"The quick brown fox jumps over the lazy dog and then runs away quickly today.")
|
||
@test name == "English"
|
||
@test code == "eng"
|
||
@test conf isa Real && 0.0 <= conf <= 1.0
|
||
|
||
# Empty / whitespace-only text yields no result rather than throwing
|
||
# (the detector itself errors on empty input).
|
||
@test detect_natural_language(d, "") == (nothing, nothing, nothing)
|
||
@test detect_natural_language(d, " \n\t ") == (nothing, nothing, nothing)
|
||
end
|
||
|
||
@testset "read_text_sample: bounded, UTF-8 safe" begin
|
||
mktempdir() do root
|
||
p = joinpath(root, "notes.txt")
|
||
write(p, "café — 日本語 — hello\n")
|
||
@test read_text_sample(p) == "café — 日本語 — hello\n"
|
||
|
||
# Reads at most LANG_SAMPLE_BYTES, and doesn't choke on a multi-byte
|
||
# char straddling that boundary (trailing 'é' half-in the window).
|
||
big = joinpath(root, "big.txt")
|
||
write(big, vcat(fill(UInt8('a'), FileServer.LANG_SAMPLE_BYTES - 1),
|
||
UInt8[0xc3, 0xa9])) # 'é' split by the edge
|
||
s = read_text_sample(big)
|
||
@test Base.length(s) == FileServer.LANG_SAMPLE_BYTES - 1 # trailing half-char trimmed
|
||
@test all(==('a'), s)
|
||
end
|
||
end
|
||
|
||
@testset "run_linguist: real detection on source vs. prose" begin
|
||
if !linguist_available()
|
||
@info "github-linguist not on PATH; skipping run_linguist tests"
|
||
else
|
||
mktempdir() do root
|
||
# A Python source file → linguist names the language.
|
||
py = joinpath(root, "script.py")
|
||
write(py, "import sys\ndef main():\n print('hi')\nmain()\n")
|
||
@test run_linguist(py, 30) == "Python"
|
||
|
||
# Plain prose reports as "Text", which collapses to nothing.
|
||
prose = joinpath(root, "notes.txt")
|
||
write(prose, "The quarterly report shows steady growth this year.\n")
|
||
@test run_linguist(prose, 30) === nothing
|
||
end
|
||
end
|
||
end
|
||
|
||
@testset "build_text_metadata + finalize_text!: end to end" begin
|
||
mktempdir() do root
|
||
cfg = tmp_config(root)
|
||
d = LanguageDetector()
|
||
|
||
src = joinpath(cfg.text_dir, "id-x-script.py")
|
||
write(src, join(["# a short program in English prose comment",
|
||
"import sys",
|
||
"def greet(name):",
|
||
" print('hello ' + name + ' welcome to the show today')",
|
||
"greet('world')", ""], "\n"))
|
||
job = Job("id-x", "script.py", src, filesize(src), 0.0)
|
||
|
||
meta = build_text_metadata(d, job, cfg)
|
||
@test meta.id == "id-x"
|
||
@test meta.content_type == "text"
|
||
@test meta.file_size == filesize(src)
|
||
@test meta.language !== nothing # some natural language detected
|
||
@test meta.error === nothing
|
||
# programming_language is best-effort; present only when linguist is.
|
||
if linguist_available()
|
||
@test meta.programming_language == "Python"
|
||
end
|
||
|
||
file_dest, sidecar = finalize_text!(cfg, job, meta)
|
||
# File moved into text_done/, original gone from text/.
|
||
@test isfile(file_dest)
|
||
@test dirname(file_dest) == cfg.text_done_dir
|
||
@test !isfile(src)
|
||
# Sidecar committed alongside it, valid JSON, no leftover .tmp.
|
||
@test isfile(sidecar)
|
||
@test endswith(sidecar, ".meta.json")
|
||
@test !isfile(string(sidecar, ".tmp"))
|
||
parsed = JSON3.read(read(sidecar, String))
|
||
@test parsed.content_type == "text"
|
||
@test parsed.file_size == filesize(file_dest)
|
||
end
|
||
end
|
||
|
||
@testset "handle_text_job: enriches and commits to text_done/" begin
|
||
mktempdir() do root
|
||
cfg = tmp_config(root)
|
||
d = LanguageDetector()
|
||
|
||
src = joinpath(cfg.text_dir, "id-h-readme.md")
|
||
write(src, "# Project\n\nThis project does something useful and interesting for everyone.\n")
|
||
job = Job("id-h", "readme.md", src, filesize(src), 0.0)
|
||
|
||
handle_text_job(job, cfg, 1, d)
|
||
@test isfile(joinpath(cfg.text_done_dir, "id-h-readme.md"))
|
||
@test isfile(joinpath(cfg.text_done_dir, "id-h-readme.md.meta.json"))
|
||
@test !isfile(src)
|
||
end
|
||
end
|
||
|
||
@testset "cluster: header_symbols feature extraction" begin
|
||
mktempdir() do root
|
||
# Bytes map to 1-based symbols (b -> b+1); positions past EOF -> PAST_EOF.
|
||
p = joinpath(root, "f.bin")
|
||
write(p, UInt8[0x00, 0x7f, 0xff])
|
||
s = header_symbols(p; n=6)
|
||
@test s[1:3] == [1, 128, 256] # 0->1, 0x7f->128, 0xff->256
|
||
@test all(==(PAST_EOF), s[4:6]) # 3 bytes short of n=6 -> past EOF
|
||
@test PAST_EOF == ALPHABET == 257
|
||
@test Base.length(header_symbols(p)) == HEADER_N
|
||
|
||
# An empty file is all past-EOF (real signal, not an error).
|
||
e = joinpath(root, "empty"); write(e, UInt8[])
|
||
@test all(==(PAST_EOF), header_symbols(e; n=8))
|
||
|
||
# header_matrix stacks one column per file.
|
||
q = joinpath(root, "g.bin"); write(q, UInt8[0x41, 0x42])
|
||
X = header_matrix([p, q]; n=4)
|
||
@test size(X) == (4, 2)
|
||
@test X[:, 2] == [0x42, 0x43, PAST_EOF, PAST_EOF] # 'A'->66,'B'->67
|
||
end
|
||
end
|
||
|
||
@testset "cluster: loggamma matches known values" begin
|
||
@test loggamma(1.0) ≈ 0.0 atol=1e-10
|
||
@test loggamma(2.0) ≈ 0.0 atol=1e-10
|
||
@test loggamma(5.0) ≈ log(24) atol=1e-10 # Γ(5) = 4! = 24
|
||
@test loggamma(0.5) ≈ 0.5log(π) atol=1e-10 # Γ(1/2) = √π
|
||
@test loggamma(10.0) ≈ log(362880) atol=1e-8 # Γ(10) = 9!
|
||
end
|
||
|
||
@testset "cluster: sufficient stats and predictive" begin
|
||
c = ClusterStats(3)
|
||
x = [10, 20, 30]
|
||
# Empty cluster's predictive equals the uniform prior (1/ALPHABET)^n.
|
||
@test log_predictive(c, x, 0.5) ≈ -3 * log(ALPHABET) atol=1e-9
|
||
# add! then remove! is an exact round-trip back to empty.
|
||
add!(c, x); remove!(c, x)
|
||
@test c.members == 0
|
||
@test all(==(0), c.counts)
|
||
# A cluster holding a matching point scores it far above uniform.
|
||
add!(c, x)
|
||
@test log_predictive(c, x, 0.5) > -3 * log(ALPHABET)
|
||
end
|
||
|
||
@testset "cluster: ARI and V-measure" begin
|
||
# Identical labelings (up to relabeling) score 1.0.
|
||
@test adjusted_rand_index([1,1,2,2], [7,7,9,9]) ≈ 1.0
|
||
@test adjusted_rand_index(["a","a","b"], ["b","b","a"]) ≈ 1.0
|
||
v, h, comp = v_measure([1,1,2,2], [5,5,6,6])
|
||
@test v ≈ 1.0 && h ≈ 1.0 && comp ≈ 1.0
|
||
# A partition that merges two true classes into one is complete but not
|
||
# homogeneous, and ARI drops below 1.
|
||
@test adjusted_rand_index([1,1,2,2], [1,1,1,1]) < 1.0
|
||
_, h2, comp2 = v_measure([1,1,2,2], [1,1,1,1])
|
||
@test comp2 ≈ 1.0 # everything from each class stays together
|
||
@test h2 < 1.0 # but the cluster mixes two classes
|
||
end
|
||
|
||
@testset "cluster: signature, magic length, promotability" begin
|
||
n = 8
|
||
c = ClusterStats(n)
|
||
# 30 files sharing bytes 0xDE 0xAD 0xBE 0xEF at positions 1-4, random after.
|
||
rng = MersenneTwister(1)
|
||
for _ in 1:30
|
||
x = vcat([0xDE, 0xAD, 0xBE, 0xEF] .+ 1, rand(rng, 1:256, 4))
|
||
add!(c, x)
|
||
end
|
||
sig = signature(c)
|
||
@test sig[1:4] == [0xDE, 0xAD, 0xBE, 0xEF] # spiked -> required bytes
|
||
@test all(isnothing, sig[5:8]) # flat -> wildcards
|
||
@test magic_positions(sig) == 4
|
||
@test is_promotable(c, sig; min_members=20, min_magic=3)
|
||
# Too few members, or too few magic positions, blocks nomination.
|
||
@test !is_promotable(c, sig; min_members=50, min_magic=3)
|
||
@test !is_promotable(c, sig; min_members=20, min_magic=5)
|
||
end
|
||
|
||
@testset "cluster: §10.1 discovers nothing from noise" begin
|
||
# 25 independent random blobs — the shape of data/binary (structureless
|
||
# junk). Correct output: ZERO promoted clusters (random headers never
|
||
# form a ≥20-member, ≥3-magic-byte signature). See DESIGN §10.1.
|
||
rng = MersenneTwister(20260703)
|
||
X = reduce(hcat, [rand(rng, 1:256, HEADER_N) for _ in 1:25])
|
||
r = gibbs_cluster(X; α=1.0, β=0.1, bg_mass=5.0, sweeps=60, restarts=3,
|
||
rng=MersenneTwister(1))
|
||
promoted = count(c -> is_promotable(c, signature(c); min_members=20, min_magic=3),
|
||
values(r.clusters))
|
||
@test promoted == 0
|
||
|
||
# And a lone structured file (a singleton, like the giant PDF in the pile)
|
||
# never promotes on its own: N=1 < min_members.
|
||
one = ClusterStats(HEADER_N)
|
||
add!(one, vcat([0x25,0x50,0x44,0x46] .+ 1, fill(1, HEADER_N - 4)))
|
||
@test !is_promotable(one, signature(one); min_members=20, min_magic=3)
|
||
end
|
||
|
||
@testset "cluster: §10.2 recovers known (synthetic) formats" begin
|
||
# Four synthetic "formats": a fixed magic prefix + random tail, mirroring
|
||
# gzip/PDF/JPEG/ELF. Calibrated settings must recover them as clean,
|
||
# promotable clusters at high ARI — the magic-collapsed recovery of §10.2,
|
||
# here with a hermetic, deterministic corpus.
|
||
# ~12-byte constant headers + random tails — the shape of a real file
|
||
# header (a fixed magic/version region, then variable content). A too-short
|
||
# magic over a fully-random tail is adversarially hard and lets a format
|
||
# over-split; real headers anchor a cluster with ~12+ constant bytes.
|
||
rng = MersenneTwister(7)
|
||
magics = Dict(
|
||
"gzip" => UInt8[0x1f,0x8b,0x08,0x00,0x00,0x00,0x00,0x00,0x00,0x03,0x2d,0x00],
|
||
"pdf" => UInt8[0x25,0x50,0x44,0x46,0x2d,0x31,0x2e,0x34,0x0a,0x25,0xe2,0xe3],
|
||
"jpeg" => UInt8[0xff,0xd8,0xff,0xe0,0x00,0x10,0x4a,0x46,0x49,0x46,0x00,0x01],
|
||
"elf" => UInt8[0x7f,0x45,0x4c,0x46,0x02,0x01,0x01,0x00,0x00,0x00,0x00,0x00],
|
||
)
|
||
cols = Vector{Int}[]; truth = String[]
|
||
for (label, magic) in magics, _ in 1:50
|
||
tail = rand(rng, 1:256, HEADER_N - Base.length(magic))
|
||
push!(cols, vcat(Int.(magic) .+ 1, tail))
|
||
push!(truth, label)
|
||
end
|
||
X = reduce(hcat, cols)
|
||
r = gibbs_cluster(X; α=1.0, β=0.1, bg_mass=5.0, sweeps=120, restarts=6,
|
||
rng=MersenneTwister(3))
|
||
@test adjusted_rand_index(truth, r.assignments) > 0.9
|
||
|
||
# Truth breakdown of each cluster, keyed by cluster id.
|
||
breakdown(id) = [truth[i] for i in eachindex(r.assignments) if r.assignments[i] == id]
|
||
# Nominations cover most formats (a format may over-split below the size
|
||
# threshold, but the recovery is not allowed to miss more than one)...
|
||
nominated_labels = Set{String}()
|
||
for (id, c) in r.clusters
|
||
sig = signature(c)
|
||
if is_promotable(c, sig; min_members=20, min_magic=3)
|
||
# ...and every nomination is PURE — the whole point of the human
|
||
# gate is that we never hand it a garbage merged signature.
|
||
labels = unique(breakdown(id))
|
||
@test Base.length(labels) == 1
|
||
push!(nominated_labels, only(labels))
|
||
end
|
||
end
|
||
@test Base.length(nominated_labels) >= 3
|
||
end
|
||
|
||
@testset "cluster: §5B sequential assignment (phase B)" begin
|
||
# Build a catalog with one strong cluster (magic 0xCA 0xFE ...).
|
||
n = 8
|
||
clusters = Dict{Int,ClusterStats}()
|
||
c = ClusterStats(n)
|
||
rng = MersenneTwister(2)
|
||
for _ in 1:40
|
||
add!(c, vcat([0xCA,0xFE,0xBA,0xBE] .+ 1, rand(rng, 1:256, 4)))
|
||
end
|
||
clusters[1] = c
|
||
ids = collect(keys(clusters))
|
||
# A file that matches the cluster's magic joins it.
|
||
match = vcat([0xCA,0xFE,0xBA,0xBE] .+ 1, rand(rng, 1:256, 4))
|
||
@test assign_file(match, clusters, ids; α=1.0, β=0.1, bg_mass=5.0) == 1
|
||
# A structured-but-novel file (different magic) spawns a new cluster (-1).
|
||
novel = vcat([0x12,0x34,0x56,0x78] .+ 1, fill(1, 4))
|
||
@test assign_file(novel, clusters, ids; α=1.0, β=0.1, bg_mass=5.0) in (-1, 0)
|
||
end
|
||
|
||
@testset "recover_dir!: re-enqueues work, skips sidecars" begin
|
||
mktempdir() do root
|
||
dir = joinpath(root, "known"); mkpath(dir)
|
||
uuid = "0123456789abcdef0123456789abcdef0123" # 36 chars
|
||
work = joinpath(dir, string(uuid, "-report.pdf"))
|
||
write(work, "x")
|
||
write(joinpath(dir, string(uuid, "-report.pdf.meta.json")), "{}") # sidecar
|
||
write(joinpath(dir, "shortname"), "y") # no uuid prefix
|
||
|
||
q = ChannelQueue(10)
|
||
n = recover_dir!(dir, q)
|
||
@test n == 2 # the two real files, not the sidecar
|
||
@test length(q) == 2
|
||
|
||
jobs = [dequeue!(q), dequeue!(q)] # sorted by filename on recovery
|
||
# "0123...-report.pdf" sorts before "shortname".
|
||
@test jobs[1].id == uuid
|
||
@test jobs[1].original_name == "report.pdf"
|
||
@test jobs[1].path == work
|
||
# File with no uuid prefix keeps its whole name; gets a minted id.
|
||
@test jobs[2].original_name == "shortname"
|
||
@test !isempty(jobs[2].id)
|
||
end
|
||
end
|
||
|
||
# Helper: write a "file" of raw bytes into a dir with a UUID-ish unique name,
|
||
# returning its path. Mirrors what stage-3 deposits into binary/.
|
||
function drop_binary(dir, bytes; name=string(rand(UInt128)))
|
||
mkpath(dir)
|
||
p = joinpath(dir, name)
|
||
open(p, "w") do io; write(io, Vector{UInt8}(bytes)); end
|
||
return p
|
||
end
|
||
|
||
@testset "catalog: durable save/load round-trip" begin
|
||
mktempdir() do root
|
||
n = 8
|
||
cat = Catalog(n)
|
||
c = ClusterStats(n)
|
||
add!(c, [0xCA+1, 0xFE+1, 0xBA+1, 0xBE+1, 1, 2, 3, 4])
|
||
add!(c, [0xCA+1, 0xFE+1, 0xBA+1, 0xBE+1, 5, 6, 7, 8])
|
||
cat.clusters[7] = c
|
||
cat.next_id = 8
|
||
record_example!(cat, 7, "alpha.bin")
|
||
push!(cat.processed, "alpha.bin"); push!(cat.processed, "beta.bin")
|
||
|
||
path = joinpath(root, "catalog.json")
|
||
save_catalog!(path, cat)
|
||
@test isfile(path)
|
||
|
||
back = load_catalog(path; n=n)
|
||
@test back.n == n
|
||
@test back.next_id == 8
|
||
@test back.processed == cat.processed
|
||
@test haskey(back.clusters, 7)
|
||
@test back.clusters[7].members == 2
|
||
@test back.clusters[7].counts == c.counts # sparse round-trips exactly
|
||
@test back.examples[7] == ["alpha.bin"]
|
||
end
|
||
end
|
||
|
||
@testset "catalog: load of a missing file is a fresh catalog" begin
|
||
mktempdir() do root
|
||
cat = load_catalog(joinpath(root, "nope.json"); n=16)
|
||
@test cat.n == 16
|
||
@test isempty(cat.clusters)
|
||
@test isempty(cat.processed)
|
||
@test cat.next_id == 1
|
||
end
|
||
end
|
||
|
||
@testset "catalog: binary_files skips sidecars, tmp, dirs; sorts" begin
|
||
mktempdir() do root
|
||
drop_binary(root, "a"; name="002-file")
|
||
drop_binary(root, "b"; name="001-file")
|
||
write(joinpath(root, "003-file.meta.json"), "{}") # sidecar
|
||
write(joinpath(root, "004-file.tmp"), "x") # scratch
|
||
mkpath(joinpath(root, "subdir")) # not a file
|
||
fs = binary_files(root)
|
||
@test basename.(fs) == ["001-file", "002-file"]
|
||
end
|
||
end
|
||
|
||
@testset "catalog: incremental sweep grows an existing cluster" begin
|
||
mktempdir() do root
|
||
n = 8
|
||
cfg = tmp_config(root; cluster_n=n, cluster_alpha=1.0,
|
||
cluster_pseudocount=0.1, cluster_bg_mass=5.0)
|
||
# Seed a strong cluster (magic 0xCA 0xFE 0xBA 0xBE, random tail).
|
||
cat = Catalog(n)
|
||
c = ClusterStats(n)
|
||
rng = MersenneTwister(3)
|
||
for _ in 1:40
|
||
add!(c, vcat([0xCA,0xFE,0xBA,0xBE] .+ 1, rand(rng, 1:256, 4)))
|
||
end
|
||
cat.clusters[1] = c
|
||
cat.next_id = 2
|
||
|
||
# A brand-new file that matches the magic must JOIN cluster 1.
|
||
drop_binary(cfg.cluster_dir, vcat(UInt8[0xCA,0xFE,0xBA,0xBE], rand(rng, UInt8, 4)); name="match-01")
|
||
# A structureless random blob must park in the background.
|
||
drop_binary(cfg.cluster_dir, rand(rng, UInt8, 64); name="blob-01")
|
||
|
||
s = catalog_sweep!(cat, cfg)
|
||
@test s.n_seen == 2
|
||
@test s.n_joined == 1
|
||
@test s.n_bg == 1
|
||
@test s.n_minted == 0
|
||
@test cat.clusters[1].members == 41 # grew by the matching file
|
||
@test "match-01" in cat.processed
|
||
@test "blob-01" in cat.processed
|
||
|
||
# Re-sweeping the same pile is idempotent — nothing new is seen.
|
||
s2 = catalog_sweep!(cat, cfg)
|
||
@test s2.n_seen == 0
|
||
@test cat.clusters[1].members == 41
|
||
end
|
||
end
|
||
|
||
@testset "catalog: §10.1 nothing from noise (end-to-end, no promotion)" begin
|
||
mktempdir() do root
|
||
n = 32
|
||
cfg = tmp_config(root; cluster_n=n, cluster_alpha=1.0,
|
||
cluster_pseudocount=0.1, cluster_bg_mass=5.0,
|
||
promote_min_members=20, promote_min_magic=3)
|
||
rng = MersenneTwister(10)
|
||
# The §10.1 pile: 20 small random blobs + 1 lone structured "PDF".
|
||
for i in 1:20
|
||
drop_binary(cfg.cluster_dir, rand(rng, UInt8, 40); name="blob-$(lpad(i,2,'0'))")
|
||
end
|
||
drop_binary(cfg.cluster_dir, vcat(UInt8[0x25,0x50,0x44,0x46], rand(rng, UInt8, 60)); name="lone-pdf")
|
||
|
||
# First run auto-compacts (empty catalog) to seed, then persists + nominates.
|
||
r = run_cluster_sweep(cfg; rng=MersenneTwister(10))
|
||
@test r.mode == :compact
|
||
@test isfile(cfg.cluster_catalog_path)
|
||
# The mission-critical assertion: ZERO promoted clusters from pure noise.
|
||
@test r.n_nominated == 0
|
||
@test isempty(readdir(cfg.nominated_dir))
|
||
# Every file was accounted for (clustered-as-singleton or background).
|
||
@test r.n_processed == 21
|
||
end
|
||
end
|
||
|
||
@testset "catalog: a real recurring format self-nominates" begin
|
||
mktempdir() do root
|
||
n = 32
|
||
# β=0.1 over-splits a format into pure sub-clusters (DESIGN §11 known
|
||
# limitation) — each still carries the full magic and nominates
|
||
# independently, so a modest min_members catches those sub-clusters.
|
||
cfg = tmp_config(root; cluster_n=n, cluster_alpha=1.0,
|
||
cluster_pseudocount=0.1, cluster_bg_mass=5.0,
|
||
promote_min_members=10, promote_min_magic=3)
|
||
rng = MersenneTwister(21)
|
||
# 30 files sharing a fixed 6-byte magic then random payload — a format.
|
||
magic = UInt8[0x89, 0x46, 0x4d, 0x54, 0x21, 0x0a]
|
||
for i in 1:30
|
||
drop_binary(cfg.cluster_dir, vcat(magic, rand(rng, UInt8, 40)); name="fmt-$(lpad(i,2,'0'))")
|
||
end
|
||
r = run_cluster_sweep(cfg; rng=MersenneTwister(21))
|
||
@test r.n_nominated >= 1
|
||
files = readdir(cfg.nominated_dir; join=true)
|
||
@test !isempty(files)
|
||
payload = JSON3.read(read(first(files), String))
|
||
@test payload.members >= 10
|
||
@test payload.magic_length >= 3
|
||
# The hex template exposes the shared magic bytes for the human gate.
|
||
@test occursin("89 46 4d 54", payload.signature_hex)
|
||
end
|
||
end
|
||
|
||
@testset "catalog: seeded catalog then live-assigns a matching arrival" begin
|
||
mktempdir() do root
|
||
n = 32
|
||
cfg = tmp_config(root; cluster_n=n, cluster_alpha=1.0,
|
||
cluster_pseudocount=0.1, cluster_bg_mass=5.0,
|
||
promote_min_members=20, promote_min_magic=3)
|
||
rng = MersenneTwister(31)
|
||
magic = UInt8[0x7a, 0x7a, 0x01, 0x02, 0x03]
|
||
for i in 1:25
|
||
drop_binary(cfg.cluster_dir, vcat(magic, rand(rng, UInt8, 40)); name="seed-$(lpad(i,2,'0'))")
|
||
end
|
||
# Seed pass.
|
||
run_cluster_sweep(cfg; rng=MersenneTwister(31))
|
||
cat = load_catalog(cfg.cluster_catalog_path; n=n)
|
||
@test !isempty(cat.clusters)
|
||
members_before = sum(c.members for c in values(cat.clusters))
|
||
|
||
# A new matching file arrives; an incremental sweep must fold it in
|
||
# (mode :sweep, not compact) without re-clustering the world.
|
||
drop_binary(cfg.cluster_dir, vcat(magic, rand(rng, UInt8, 40)); name="arrival-01")
|
||
r2 = run_cluster_sweep(cfg; rng=MersenneTwister(99))
|
||
@test r2.mode == :sweep
|
||
cat2 = load_catalog(cfg.cluster_catalog_path; n=n)
|
||
members_after = sum(c.members for c in values(cat2.clusters))
|
||
@test members_after == members_before + 1 # the arrival joined a cluster
|
||
end
|
||
end
|
||
|
||
end
|