using Test using FileServer using JSON3 # Pull internals into scope. These aren't exported (only `run` is), but the # whole risk profile of this pipeline lives in these functions, so we test them # directly rather than only through the HTTP surface. using FileServer: Job, Config, ChannelQueue, enqueue!, dequeue!, close!, length, sanitize_filename, recover_dir!, normalize_metadata, build_metadata, finalize_known!, run_exiftool, is_binary, handle_unknown_job, worker_loop, capacity, StageStats, IntakeStats, Metrics, METRICS, reset_metrics!, record_job!, enqueue_blocking!, stats_snapshot, STAGE_KEYS, detect_natural_language, run_linguist, detect_programming_language, read_text_sample, build_text_metadata, finalize_text!, handle_text_job, linguist_available, header_symbols, header_matrix, ClusterStats, add!, remove!, log_predictive, loggamma, gibbs_cluster, assign_file, signature, magic_positions, is_promotable, adjusted_rand_index, v_measure, HEADER_N, ALPHABET, PAST_EOF, Catalog, load_catalog, save_catalog!, catalog_sweep!, compact!, write_nominations!, run_cluster_sweep, binary_files, record_example!, signature_hex, ensure_dirs, MultipartReader, MultipartError, MultipartPart, next_part!, write_part_body!, skip_part_body!, multipart_boundary, parse_part_headers, spool_stream, UPLOAD_CHUNK_BYTES using Random: MersenneTwister using Languages: LanguageDetector # A minimal, valid 1×1 PNG. Lets the real-exiftool tests assert stable facts # (FileType == "PNG", 1×1 dimensions) that don't drift across exiftool versions. const PNG_1x1 = UInt8[137,80,78,71,13,10,26,10,0,0,0,13,73,72,68,82,0,0,0,1,0, 0,0,1,8,6,0,0,0,31,21,196,137,0,0,0,11,73,68,65,84,120,218,99,100,248,255, 191,30,0,5,132,2,127,194,91,30,42,0,0,0,0,73,69,78,68,174,66,96,130] """ Assemble a multipart/form-data body. `parts` are `(name, filename, content_type, data)` tuples; a `nothing` filename makes a plain form field rather than a file. """ function multipart_body(boundary, parts; preamble = "", terminate = true) io = IOBuffer() write(io, preamble) for (name, filename, content_type, data) in parts write(io, "--$boundary\r\n") write(io, "Content-Disposition: form-data; name=\"$name\"") filename === nothing || write(io, "; filename=\"$filename\"") write(io, "\r\n") content_type === nothing || write(io, "Content-Type: $content_type\r\n") write(io, "\r\n") write(io, data) write(io, "\r\n") end write(io, terminate ? "--$boundary--\r\n" : "--$boundary\r\n") return take!(io) end "Read every part out of `bytes`, returning `(part, body, nbytes)` triples." function read_all_parts(bytes, boundary; chunk_bytes = UPLOAD_CHUNK_BYTES) r = MultipartReader(IOBuffer(bytes), boundary; chunk_bytes = chunk_bytes) out = Tuple{MultipartPart,String,Int}[] while (part = next_part!(r)) !== nothing sink = IOBuffer() n = write_part_body!(sink, r) push!(out, (part, String(take!(sink)), n)) end return out end "Build a Config whose data dirs all live under a fresh temp directory." function tmp_config(root; kwargs...) cfg = Config(; spool_dir = joinpath(root, "spool"), known_dir = joinpath(root, "known"), unknown_dir = joinpath(root, "unknown"), binary_dir = joinpath(root, "binary"), text_dir = joinpath(root, "text"), done_dir = joinpath(root, "done"), text_done_dir = joinpath(root, "text_done"), failed_dir = joinpath(root, "failed"), cluster_dir = joinpath(root, "binary"), # stage-5 sweeps the binary sink cluster_catalog_path = joinpath(root, "catalog.json"), nominated_dir = joinpath(root, "nominated"), kwargs..., ) FileServer.ensure_dirs(cfg) return cfg end @testset "FileServer" begin @testset "sanitize_filename" begin @test sanitize_filename("report.pdf") == "report.pdf" # Directory components and traversal are stripped, not preserved. @test sanitize_filename("../../etc/passwd") == "passwd" @test sanitize_filename("/abs/path/x.txt") == "x.txt" # Leading dots removed so "..", ".hidden" can't sneak through. @test sanitize_filename("..") == "unnamed" @test sanitize_filename(".hidden") == "hidden" # Unsafe chars collapse to underscores; empty falls back to "unnamed". @test sanitize_filename("a b&c*.d") == "a_b_c_.d" @test sanitize_filename("") == "unnamed" # Length is capped. @test Base.length(sanitize_filename("a"^500)) == FileServer.MAX_NAME_LEN end @testset "multipart_boundary: extraction from Content-Type" begin @test multipart_boundary("multipart/form-data; boundary=abc") == "abc" @test multipart_boundary("multipart/form-data; boundary=\"a b;c\"") == "a b;c" @test multipart_boundary("MULTIPART/FORM-DATA; BOUNDARY=xyz") == "xyz" @test multipart_boundary("multipart/form-data; charset=utf-8; boundary=q1") == "q1" # Anything that isn't a usable multipart header is the same 400 to a caller. @test multipart_boundary("multipart/form-data") === nothing @test multipart_boundary("application/json") === nothing @test multipart_boundary(nothing) === nothing end @testset "parse_part_headers" begin p = parse_part_headers("Content-Disposition: form-data; name=\"f\"; filename=\"a b.txt\"\r\n" * "Content-Type: text/plain") @test p.name == "f" @test p.filename == "a b.txt" @test p.content_type == "text/plain" # `name=` must not match inside `filename=` — that would label every # file part with a bogus name and (worse) hide a missing real name. p = parse_part_headers("Content-Disposition: form-data; filename=\"only.txt\"") @test p.name === nothing @test p.filename == "only.txt" # No filename means a plain form field, which intake must not spool. p = parse_part_headers("Content-Disposition: form-data; name=\"note\"") @test p.filename === nothing end @testset "MultipartReader: parts, fields, and bodies" begin B = "----testboundary" body = multipart_body(B, [("f0", "a.txt", "text/plain", "hello world"), ("note", nothing, nothing, "just-a-field"), ("f1", "b.bin", nothing, "\x00\x01\x02")]) got = read_all_parts(body, B) @test length(got) == 3 @test got[1][1].filename == "a.txt" @test got[1][2] == "hello world" @test got[1][3] == 11 # reported byte count @test got[1][1].content_type == "text/plain" @test got[2][1].filename === nothing # the form field @test got[2][2] == "just-a-field" @test codeunits(got[3][2]) == UInt8[0x00, 0x01, 0x02] # A zero-byte file is legal and must survive as zero bytes. empty_got = read_all_parts(multipart_body(B, [("f", "empty.bin", nothing, "")]), B) @test empty_got[1][3] == 0 @test empty_got[1][2] == "" end @testset "MultipartReader: delimiter straddling every chunk offset" begin # The one thing a chunked parser can get catastrophically wrong is a # delimiter split across two reads. Parsing the same body at many chunk # sizes puts the split at every offset. The payload deliberately contains # CR, LF and '-' bytes, so a sloppy scan finds false delimiters. B = "----testboundary" rng = MersenneTwister(7) payload = String(rand(rng, UInt8[0x41:0x5a; 0x0d; 0x0a; 0x2d], 5000)) body = multipart_body(B, [("f", "big.bin", nothing, payload)]) for chunk in (1, 2, 3, 5, 7, 13, 16, 17, 64, 255, 4096, 10_000) got = read_all_parts(body, B; chunk_bytes = chunk) @test length(got) == 1 @test got[1][2] == payload end # A payload containing a *prefix* of the real delimiter must not end the part. tricky = "aaa\r\n--" * "----testboundar" * "bbb\r\n--x\r\nccc" body2 = multipart_body(B, [("f", "t.bin", nothing, tricky)]) for chunk in (1, 4, 9, 64, 4096) got = read_all_parts(body2, B; chunk_bytes = chunk) @test length(got) == 1 @test got[1][2] == tricky end end @testset "MultipartReader: memory stays bounded, not proportional to the part" begin # The whole point of the streaming reader. A 16 MiB part read with a # 64 KiB chunk must allocate on the order of the chunk, not the part. B = "----testboundary" payload = String(rand(MersenneTwister(11), UInt8, 16 * 1024 * 1024)) body = multipart_body(B, [("f", "huge.bin", nothing, payload)]) r = MultipartReader(IOBuffer(body), B; chunk_bytes = 64 * 1024) next_part!(r) GC.gc() allocated = @allocated write_part_body!(devnull, r) @test allocated < 4 * 1024 * 1024 end @testset "MultipartReader: malformed bodies raise MultipartError" begin B = "----testboundary" valid = multipart_body(B, [("f", "a.bin", nothing, "hello")]) @test_throws MultipartError read_all_parts(Vector{UInt8}("no delimiter here"), B) @test_throws MultipartError read_all_parts(valid[1:end-20], B) # truncated mid-part @test_throws MultipartError read_all_parts( multipart_body(B, [("f", "a.bin", nothing, "x")]; terminate = false), B) # Part headers must be bounded regardless of how the body was chunked, # since they are the one thing that has to be buffered whole to parse. oversized = Vector{UInt8}("--$B\r\nContent-Disposition: form-data; name=\"" * "x"^30_000 * "\"\r\n\r\ndata\r\n--$B--\r\n") @test_throws MultipartError read_all_parts(oversized, B) # A part's body must be consumed before advancing: the reader cannot skip # a body on its own, because a body only ends at the next delimiter. r = MultipartReader(IOBuffer(valid), B) next_part!(r) @test_throws MultipartError next_part!(r) end @testset "spool_stream: streams to disk, cleans up a failed write" begin mktempdir() do root cfg = tmp_config(root) job = spool_stream(cfg, "report v2.pdf") do io write(io, "abc") + write(io, "de") end @test isfile(job.path) @test read(job.path, String) == "abcde" @test job.size == 5 # from the bytes actually written @test job.original_name == "report v2.pdf" @test basename(job.path) == "$(job.id)-report_v2.pdf" # sanitized, uuid-prefixed # A write that throws must leave nothing behind: recovery on restart # re-enqueues whatever is in spool/, and a truncated upload there # would be silently processed as if it were complete. before = length(readdir(cfg.spool_dir)) @test_throws ErrorException spool_stream(cfg, "bad.bin") do io write(io, "partial") error("disk went away") end @test length(readdir(cfg.spool_dir)) == before end end @testset "normalize_metadata" begin job = Job("id-1", "photo.jpg", "/data/known/id-1-photo.jpg", 4242, 0.0) # Group-prefixed tags as exiftool -G emits them are already group-stripped # by run_exiftool before reaching normalize_metadata, so keys are bare. bytag = Dict{String,Any}( "FileType" => "JPEG", "MIMEType" => "image/jpeg", "ImageWidth" => 800, "ImageHeight"=> 600, "Author" => "Ada Lovelace", "Creator" => "Acrobat", # feeds created_by, not author "CreateDate" => "2020:01:02 03:04:05", "ModifyDate" => "2020:01:02 03:04:06", "PageCount" => 12, ) m = normalize_metadata(job, bytag) @test m.file_type == "JPEG" @test m.mime_type == "image/jpeg" @test m.dimensions == (width = 800, height = 600) @test m.author == "Ada Lovelace" @test m.created_by == "Acrobat" @test m.created_date == "2020:01:02 03:04:05" @test m.page_count == 12 @test m.error === nothing @test m.raw === bytag # file_size is authoritative from the Job, never from exiftool. @test m.file_size == 4242 end @testset "normalize_metadata: missing tags degrade to nothing" begin job = Job("id-2", "blob.bin", "/data/known/id-2-blob.bin", 7, 0.0) m = normalize_metadata(job, Dict{String,Any}()) @test m.file_type === nothing @test m.dimensions === nothing # neither width nor height present @test m.author === nothing @test m.file_size == 7 @test m.error === nothing # empty-but-present dict is still "success" end @testset "build_metadata: degraded on extraction failure" begin mktempdir() do root cfg = tmp_config(root; exiftool_timeout=5) # Point at a nonexistent file → exiftool exits non-zero → degraded. job = Job("id-3", "gone.dat", joinpath(cfg.known_dir, "id-3-gone.dat"), 99, 0.0) m = build_metadata(job, cfg) @test m.error !== nothing @test m.file_type === nothing @test m.raw === nothing @test m.file_size == 99 # still authoritative from the Job @test m.id == "id-3" end end @testset "run_exiftool: real extraction on a PNG" begin mktempdir() do root p = joinpath(root, "pixel.png") write(p, PNG_1x1) bytag = run_exiftool(p, 30) @test bytag !== nothing @test bytag["FileType"] == "PNG" @test bytag["ImageWidth"] == 1 @test bytag["ImageHeight"] == 1 end end @testset "finalize_known!: sidecar-first commit, end to end" begin mktempdir() do root cfg = tmp_config(root) # A real known-stage file to enrich. src = joinpath(cfg.known_dir, "id-9-pixel.png") write(src, PNG_1x1) job = Job("id-9", "pixel.png", src, Base.length(PNG_1x1), 0.0) meta = build_metadata(job, cfg) file_dest, sidecar = finalize_known!(cfg, job, meta) # File moved into done/, original gone from known/. @test isfile(file_dest) @test dirname(file_dest) == cfg.done_dir @test !isfile(src) # Sidecar committed alongside it, valid JSON, no leftover .tmp. @test isfile(sidecar) @test endswith(sidecar, ".meta.json") @test !isfile(string(sidecar, ".tmp")) parsed = JSON3.read(read(sidecar, String)) @test parsed.file_type == "PNG" @test parsed.file_size == Base.length(PNG_1x1) @test parsed.error === nothing end end @testset "is_binary: UTF-8 sniff" begin mktempdir() do root # Plain ASCII text → text. txt = joinpath(root, "notes.txt") write(txt, "hello, world\nsecond line\n") @test is_binary(txt) == false # Non-ASCII UTF-8 (accents, CJK, emoji) is valid text — the whole # point of moving off the printable-ASCII/NUL heuristic. uni = joinpath(root, "unicode.txt") write(uni, "café — 日本語 — 🚀\n") @test is_binary(uni) == false # ANSI-colored log: ESC + other text control bytes are text-safe. ansi = joinpath(root, "colored.log") write(ansi, "\e[31merror\e[0m: tab\there\r\nnext\n") @test is_binary(ansi) == false # A NUL byte anywhere in the sniff window → binary (it's a control # byte outside the text-safe set, even though it's valid UTF-8). bin = joinpath(root, "blob.dat") write(bin, UInt8[0x01, 0x02, 0x00, 0x03]) @test is_binary(bin) == true # A non-NUL, non-text control byte (e.g. 0x07 BEL) → binary. ctrl = joinpath(root, "ctrl.dat") write(ctrl, UInt8[UInt8('h'), UInt8('i'), 0x07]) @test is_binary(ctrl) == true # Malformed UTF-8 (lone continuation / bad lead byte) → binary. bad = joinpath(root, "bad.dat") write(bad, UInt8[UInt8('a'), 0xff, 0xfe, 0xc3, 0x28]) @test is_binary(bad) == true # A multi-byte char split by the sniff boundary must NOT read as # binary: pad to one byte short of the window, then a 2-byte 'é' # (0xc3 0xa9) so only its lead byte lands inside the window. split = joinpath(root, "split.txt") write(split, vcat(fill(UInt8('a'), FileServer.CONTENT_SNIFF_BYTES - 1), UInt8[0xc3, 0xa9])) @test is_binary(split) == false # Empty file → treated as text. empty = joinpath(root, "empty") write(empty, UInt8[]) @test is_binary(empty) == false # Binary garbage past the sniff window is not seen → still text. far = joinpath(root, "far.txt") write(far, vcat(fill(UInt8('a'), FileServer.CONTENT_SNIFF_BYTES), UInt8[0x00])) @test is_binary(far) == false end end @testset "handle_unknown_job: binary terminal, text routed to stage 4" begin mktempdir() do root cfg = tmp_config(root) text_queue = ChannelQueue(10) stats = StageStats() # A binary file (embedded NUL) lands in binary/ and is NOT enqueued. bpath = joinpath(cfg.unknown_dir, "id-b-blob.dat") write(bpath, UInt8[0x00, 0xFF, 0x10]) bjob = Job("id-b", "blob.dat", bpath, filesize(bpath), 0.0) handle_unknown_job(bjob, cfg, 1, text_queue, stats) @test isfile(joinpath(cfg.binary_dir, "id-b-blob.dat")) @test !isfile(bpath) @test length(text_queue) == 0 # A text file lands in text/ AND is routed onto the stage-4 queue, # with its path updated to the new text/ location. tpath = joinpath(cfg.unknown_dir, "id-t-notes.log") write(tpath, "just some log text\n") tjob = Job("id-t", "notes.log", tpath, filesize(tpath), 0.0) handle_unknown_job(tjob, cfg, 1, text_queue, stats) moved = joinpath(cfg.text_dir, "id-t-notes.log") @test isfile(moved) @test !isfile(tpath) @test length(text_queue) == 1 routed = dequeue!(text_queue) @test routed.id == "id-t" @test routed.path == moved end end @testset "detect_natural_language" begin d = LanguageDetector() name, code, conf = detect_natural_language(d, "The quick brown fox jumps over the lazy dog and then runs away quickly today.") @test name == "English" @test code == "eng" @test conf isa Real && 0.0 <= conf <= 1.0 # Empty / whitespace-only text yields no result rather than throwing # (the detector itself errors on empty input). @test detect_natural_language(d, "") == (nothing, nothing, nothing) @test detect_natural_language(d, " \n\t ") == (nothing, nothing, nothing) end @testset "read_text_sample: bounded, UTF-8 safe" begin mktempdir() do root p = joinpath(root, "notes.txt") write(p, "café — 日本語 — hello\n") @test read_text_sample(p) == "café — 日本語 — hello\n" # Reads at most LANG_SAMPLE_BYTES, and doesn't choke on a multi-byte # char straddling that boundary (trailing 'é' half-in the window). big = joinpath(root, "big.txt") write(big, vcat(fill(UInt8('a'), FileServer.LANG_SAMPLE_BYTES - 1), UInt8[0xc3, 0xa9])) # 'é' split by the edge s = read_text_sample(big) @test Base.length(s) == FileServer.LANG_SAMPLE_BYTES - 1 # trailing half-char trimmed @test all(==('a'), s) end end @testset "run_linguist: real detection on source vs. prose" begin if !linguist_available() @info "github-linguist not on PATH; skipping run_linguist tests" else mktempdir() do root # A Python source file → linguist names the language. py = joinpath(root, "script.py") write(py, "import sys\ndef main():\n print('hi')\nmain()\n") @test run_linguist(py, 30) == "Python" # Plain prose reports as "Text", which collapses to nothing. prose = joinpath(root, "notes.txt") write(prose, "The quarterly report shows steady growth this year.\n") @test run_linguist(prose, 30) === nothing end end end @testset "build_text_metadata + finalize_text!: end to end" begin mktempdir() do root cfg = tmp_config(root) d = LanguageDetector() src = joinpath(cfg.text_dir, "id-x-script.py") write(src, join(["# a short program in English prose comment", "import sys", "def greet(name):", " print('hello ' + name + ' welcome to the show today')", "greet('world')", ""], "\n")) job = Job("id-x", "script.py", src, filesize(src), 0.0) meta = build_text_metadata(d, job, cfg) @test meta.id == "id-x" @test meta.content_type == "text" @test meta.file_size == filesize(src) @test meta.language !== nothing # some natural language detected @test meta.error === nothing # programming_language is best-effort; present only when linguist is. if linguist_available() @test meta.programming_language == "Python" end file_dest, sidecar = finalize_text!(cfg, job, meta) # File moved into text_done/, original gone from text/. @test isfile(file_dest) @test dirname(file_dest) == cfg.text_done_dir @test !isfile(src) # Sidecar committed alongside it, valid JSON, no leftover .tmp. @test isfile(sidecar) @test endswith(sidecar, ".meta.json") @test !isfile(string(sidecar, ".tmp")) parsed = JSON3.read(read(sidecar, String)) @test parsed.content_type == "text" @test parsed.file_size == filesize(file_dest) end end @testset "handle_text_job: enriches and commits to text_done/" begin mktempdir() do root cfg = tmp_config(root) d = LanguageDetector() src = joinpath(cfg.text_dir, "id-h-readme.md") write(src, "# Project\n\nThis project does something useful and interesting for everyone.\n") job = Job("id-h", "readme.md", src, filesize(src), 0.0) handle_text_job(job, cfg, 1, d) @test isfile(joinpath(cfg.text_done_dir, "id-h-readme.md")) @test isfile(joinpath(cfg.text_done_dir, "id-h-readme.md.meta.json")) @test !isfile(src) end end @testset "cluster: header_symbols feature extraction" begin mktempdir() do root # Bytes map to 1-based symbols (b -> b+1); positions past EOF -> PAST_EOF. p = joinpath(root, "f.bin") write(p, UInt8[0x00, 0x7f, 0xff]) s = header_symbols(p; n=6) @test s[1:3] == [1, 128, 256] # 0->1, 0x7f->128, 0xff->256 @test all(==(PAST_EOF), s[4:6]) # 3 bytes short of n=6 -> past EOF @test PAST_EOF == ALPHABET == 257 @test Base.length(header_symbols(p)) == HEADER_N # An empty file is all past-EOF (real signal, not an error). e = joinpath(root, "empty"); write(e, UInt8[]) @test all(==(PAST_EOF), header_symbols(e; n=8)) # header_matrix stacks one column per file. q = joinpath(root, "g.bin"); write(q, UInt8[0x41, 0x42]) X = header_matrix([p, q]; n=4) @test size(X) == (4, 2) @test X[:, 2] == [0x42, 0x43, PAST_EOF, PAST_EOF] # 'A'->66,'B'->67 end end @testset "cluster: loggamma matches known values" begin @test loggamma(1.0) ≈ 0.0 atol=1e-10 @test loggamma(2.0) ≈ 0.0 atol=1e-10 @test loggamma(5.0) ≈ log(24) atol=1e-10 # Γ(5) = 4! = 24 @test loggamma(0.5) ≈ 0.5log(π) atol=1e-10 # Γ(1/2) = √π @test loggamma(10.0) ≈ log(362880) atol=1e-8 # Γ(10) = 9! end @testset "cluster: sufficient stats and predictive" begin c = ClusterStats(3) x = [10, 20, 30] # Empty cluster's predictive equals the uniform prior (1/ALPHABET)^n. @test log_predictive(c, x, 0.5) ≈ -3 * log(ALPHABET) atol=1e-9 # add! then remove! is an exact round-trip back to empty. add!(c, x); remove!(c, x) @test c.members == 0 @test all(==(0), c.counts) # A cluster holding a matching point scores it far above uniform. add!(c, x) @test log_predictive(c, x, 0.5) > -3 * log(ALPHABET) end @testset "cluster: ARI and V-measure" begin # Identical labelings (up to relabeling) score 1.0. @test adjusted_rand_index([1,1,2,2], [7,7,9,9]) ≈ 1.0 @test adjusted_rand_index(["a","a","b"], ["b","b","a"]) ≈ 1.0 v, h, comp = v_measure([1,1,2,2], [5,5,6,6]) @test v ≈ 1.0 && h ≈ 1.0 && comp ≈ 1.0 # A partition that merges two true classes into one is complete but not # homogeneous, and ARI drops below 1. @test adjusted_rand_index([1,1,2,2], [1,1,1,1]) < 1.0 _, h2, comp2 = v_measure([1,1,2,2], [1,1,1,1]) @test comp2 ≈ 1.0 # everything from each class stays together @test h2 < 1.0 # but the cluster mixes two classes end @testset "cluster: signature, magic length, promotability" begin n = 8 c = ClusterStats(n) # 30 files sharing bytes 0xDE 0xAD 0xBE 0xEF at positions 1-4, random after. rng = MersenneTwister(1) for _ in 1:30 x = vcat([0xDE, 0xAD, 0xBE, 0xEF] .+ 1, rand(rng, 1:256, 4)) add!(c, x) end sig = signature(c) @test sig[1:4] == [0xDE, 0xAD, 0xBE, 0xEF] # spiked -> required bytes @test all(isnothing, sig[5:8]) # flat -> wildcards @test magic_positions(sig) == 4 @test is_promotable(c, sig; min_members=20, min_magic=3) # Too few members, or too few magic positions, blocks nomination. @test !is_promotable(c, sig; min_members=50, min_magic=3) @test !is_promotable(c, sig; min_members=20, min_magic=5) end @testset "cluster: §10.1 discovers nothing from noise" begin # 25 independent random blobs — the shape of data/binary (structureless # junk). Correct output: ZERO promoted clusters (random headers never # form a ≥20-member, ≥3-magic-byte signature). See DESIGN §10.1. rng = MersenneTwister(20260703) X = reduce(hcat, [rand(rng, 1:256, HEADER_N) for _ in 1:25]) r = gibbs_cluster(X; α=1.0, β=0.1, bg_mass=5.0, sweeps=60, restarts=3, rng=MersenneTwister(1)) promoted = count(c -> is_promotable(c, signature(c); min_members=20, min_magic=3), values(r.clusters)) @test promoted == 0 # And a lone structured file (a singleton, like the giant PDF in the pile) # never promotes on its own: N=1 < min_members. one = ClusterStats(HEADER_N) add!(one, vcat([0x25,0x50,0x44,0x46] .+ 1, fill(1, HEADER_N - 4))) @test !is_promotable(one, signature(one); min_members=20, min_magic=3) end @testset "cluster: §10.2 recovers known (synthetic) formats" begin # Four synthetic "formats": a fixed magic prefix + random tail, mirroring # gzip/PDF/JPEG/ELF. Calibrated settings must recover them as clean, # promotable clusters at high ARI — the magic-collapsed recovery of §10.2, # here with a hermetic, deterministic corpus. # ~12-byte constant headers + random tails — the shape of a real file # header (a fixed magic/version region, then variable content). A too-short # magic over a fully-random tail is adversarially hard and lets a format # over-split; real headers anchor a cluster with ~12+ constant bytes. rng = MersenneTwister(7) magics = Dict( "gzip" => UInt8[0x1f,0x8b,0x08,0x00,0x00,0x00,0x00,0x00,0x00,0x03,0x2d,0x00], "pdf" => UInt8[0x25,0x50,0x44,0x46,0x2d,0x31,0x2e,0x34,0x0a,0x25,0xe2,0xe3], "jpeg" => UInt8[0xff,0xd8,0xff,0xe0,0x00,0x10,0x4a,0x46,0x49,0x46,0x00,0x01], "elf" => UInt8[0x7f,0x45,0x4c,0x46,0x02,0x01,0x01,0x00,0x00,0x00,0x00,0x00], ) cols = Vector{Int}[]; truth = String[] for (label, magic) in magics, _ in 1:50 tail = rand(rng, 1:256, HEADER_N - Base.length(magic)) push!(cols, vcat(Int.(magic) .+ 1, tail)) push!(truth, label) end X = reduce(hcat, cols) r = gibbs_cluster(X; α=1.0, β=0.1, bg_mass=5.0, sweeps=120, restarts=6, rng=MersenneTwister(3)) @test adjusted_rand_index(truth, r.assignments) > 0.9 # Truth breakdown of each cluster, keyed by cluster id. breakdown(id) = [truth[i] for i in eachindex(r.assignments) if r.assignments[i] == id] # Nominations cover most formats (a format may over-split below the size # threshold, but the recovery is not allowed to miss more than one)... nominated_labels = Set{String}() for (id, c) in r.clusters sig = signature(c) if is_promotable(c, sig; min_members=20, min_magic=3) # ...and every nomination is PURE — the whole point of the human # gate is that we never hand it a garbage merged signature. labels = unique(breakdown(id)) @test Base.length(labels) == 1 push!(nominated_labels, only(labels)) end end @test Base.length(nominated_labels) >= 3 end @testset "cluster: §5B sequential assignment (phase B)" begin # Build a catalog with one strong cluster (magic 0xCA 0xFE ...). n = 8 clusters = Dict{Int,ClusterStats}() c = ClusterStats(n) rng = MersenneTwister(2) for _ in 1:40 add!(c, vcat([0xCA,0xFE,0xBA,0xBE] .+ 1, rand(rng, 1:256, 4))) end clusters[1] = c ids = collect(keys(clusters)) # A file that matches the cluster's magic joins it. match = vcat([0xCA,0xFE,0xBA,0xBE] .+ 1, rand(rng, 1:256, 4)) @test assign_file(match, clusters, ids; α=1.0, β=0.1, bg_mass=5.0) == 1 # A structured-but-novel file (different magic) spawns a new cluster (-1). novel = vcat([0x12,0x34,0x56,0x78] .+ 1, fill(1, 4)) @test assign_file(novel, clusters, ids; α=1.0, β=0.1, bg_mass=5.0) in (-1, 0) end @testset "recover_dir!: re-enqueues work, skips sidecars" begin mktempdir() do root dir = joinpath(root, "known"); mkpath(dir) uuid = "0123456789abcdef0123456789abcdef0123" # 36 chars work = joinpath(dir, string(uuid, "-report.pdf")) write(work, "x") write(joinpath(dir, string(uuid, "-report.pdf.meta.json")), "{}") # sidecar write(joinpath(dir, "shortname"), "y") # no uuid prefix q = ChannelQueue(10) n = recover_dir!(dir, q) @test n == 2 # the two real files, not the sidecar @test length(q) == 2 jobs = [dequeue!(q), dequeue!(q)] # sorted by filename on recovery # "0123...-report.pdf" sorts before "shortname". @test jobs[1].id == uuid @test jobs[1].original_name == "report.pdf" @test jobs[1].path == work # File with no uuid prefix keeps its whole name; gets a minted id. @test jobs[2].original_name == "shortname" @test !isempty(jobs[2].id) end end # Helper: write a "file" of raw bytes into a dir with a UUID-ish unique name, # returning its path. Mirrors what stage-3 deposits into binary/. function drop_binary(dir, bytes; name=string(rand(UInt128))) mkpath(dir) p = joinpath(dir, name) open(p, "w") do io; write(io, Vector{UInt8}(bytes)); end return p end @testset "catalog: durable save/load round-trip" begin mktempdir() do root n = 8 cat = Catalog(n) c = ClusterStats(n) add!(c, [0xCA+1, 0xFE+1, 0xBA+1, 0xBE+1, 1, 2, 3, 4]) add!(c, [0xCA+1, 0xFE+1, 0xBA+1, 0xBE+1, 5, 6, 7, 8]) cat.clusters[7] = c cat.next_id = 8 record_example!(cat, 7, "alpha.bin") push!(cat.processed, "alpha.bin"); push!(cat.processed, "beta.bin") path = joinpath(root, "catalog.json") save_catalog!(path, cat) @test isfile(path) back = load_catalog(path; n=n) @test back.n == n @test back.next_id == 8 @test back.processed == cat.processed @test haskey(back.clusters, 7) @test back.clusters[7].members == 2 @test back.clusters[7].counts == c.counts # sparse round-trips exactly @test back.examples[7] == ["alpha.bin"] end end @testset "catalog: load of a missing file is a fresh catalog" begin mktempdir() do root cat = load_catalog(joinpath(root, "nope.json"); n=16) @test cat.n == 16 @test isempty(cat.clusters) @test isempty(cat.processed) @test cat.next_id == 1 end end @testset "catalog: binary_files skips sidecars, tmp, dirs; sorts" begin mktempdir() do root drop_binary(root, "a"; name="002-file") drop_binary(root, "b"; name="001-file") write(joinpath(root, "003-file.meta.json"), "{}") # sidecar write(joinpath(root, "004-file.tmp"), "x") # scratch mkpath(joinpath(root, "subdir")) # not a file fs = binary_files(root) @test basename.(fs) == ["001-file", "002-file"] end end @testset "catalog: incremental sweep grows an existing cluster" begin mktempdir() do root n = 8 cfg = tmp_config(root; cluster_n=n, cluster_alpha=1.0, cluster_pseudocount=0.1, cluster_bg_mass=5.0) # Seed a strong cluster (magic 0xCA 0xFE 0xBA 0xBE, random tail). cat = Catalog(n) c = ClusterStats(n) rng = MersenneTwister(3) for _ in 1:40 add!(c, vcat([0xCA,0xFE,0xBA,0xBE] .+ 1, rand(rng, 1:256, 4))) end cat.clusters[1] = c cat.next_id = 2 # A brand-new file that matches the magic must JOIN cluster 1. drop_binary(cfg.cluster_dir, vcat(UInt8[0xCA,0xFE,0xBA,0xBE], rand(rng, UInt8, 4)); name="match-01") # A structureless random blob must park in the background. drop_binary(cfg.cluster_dir, rand(rng, UInt8, 64); name="blob-01") s = catalog_sweep!(cat, cfg) @test s.n_seen == 2 @test s.n_joined == 1 @test s.n_bg == 1 @test s.n_minted == 0 @test cat.clusters[1].members == 41 # grew by the matching file @test "match-01" in cat.processed @test "blob-01" in cat.processed # Re-sweeping the same pile is idempotent — nothing new is seen. s2 = catalog_sweep!(cat, cfg) @test s2.n_seen == 0 @test cat.clusters[1].members == 41 end end @testset "catalog: §10.1 nothing from noise (end-to-end, no promotion)" begin mktempdir() do root n = 32 cfg = tmp_config(root; cluster_n=n, cluster_alpha=1.0, cluster_pseudocount=0.1, cluster_bg_mass=5.0, promote_min_members=20, promote_min_magic=3) rng = MersenneTwister(10) # The §10.1 pile: 20 small random blobs + 1 lone structured "PDF". for i in 1:20 drop_binary(cfg.cluster_dir, rand(rng, UInt8, 40); name="blob-$(lpad(i,2,'0'))") end drop_binary(cfg.cluster_dir, vcat(UInt8[0x25,0x50,0x44,0x46], rand(rng, UInt8, 60)); name="lone-pdf") # First run auto-compacts (empty catalog) to seed, then persists + nominates. r = run_cluster_sweep(cfg; rng=MersenneTwister(10)) @test r.mode == :compact @test isfile(cfg.cluster_catalog_path) # The mission-critical assertion: ZERO promoted clusters from pure noise. @test r.n_nominated == 0 @test isempty(readdir(cfg.nominated_dir)) # Every file was accounted for (clustered-as-singleton or background). @test r.n_processed == 21 end end @testset "catalog: a real recurring format self-nominates" begin mktempdir() do root n = 32 # β=0.1 over-splits a format into pure sub-clusters (DESIGN §11 known # limitation) — each still carries the full magic and nominates # independently, so a modest min_members catches those sub-clusters. cfg = tmp_config(root; cluster_n=n, cluster_alpha=1.0, cluster_pseudocount=0.1, cluster_bg_mass=5.0, promote_min_members=10, promote_min_magic=3) rng = MersenneTwister(21) # 30 files sharing a fixed 6-byte magic then random payload — a format. magic = UInt8[0x89, 0x46, 0x4d, 0x54, 0x21, 0x0a] for i in 1:30 drop_binary(cfg.cluster_dir, vcat(magic, rand(rng, UInt8, 40)); name="fmt-$(lpad(i,2,'0'))") end r = run_cluster_sweep(cfg; rng=MersenneTwister(21)) @test r.n_nominated >= 1 files = readdir(cfg.nominated_dir; join=true) @test !isempty(files) payload = JSON3.read(read(first(files), String)) @test payload.members >= 10 @test payload.magic_length >= 3 # The hex template exposes the shared magic bytes for the human gate. @test occursin("89 46 4d 54", payload.signature_hex) end end @testset "catalog: seeded catalog then live-assigns a matching arrival" begin mktempdir() do root n = 32 cfg = tmp_config(root; cluster_n=n, cluster_alpha=1.0, cluster_pseudocount=0.1, cluster_bg_mass=5.0, promote_min_members=20, promote_min_magic=3) rng = MersenneTwister(31) magic = UInt8[0x7a, 0x7a, 0x01, 0x02, 0x03] for i in 1:25 drop_binary(cfg.cluster_dir, vcat(magic, rand(rng, UInt8, 40)); name="seed-$(lpad(i,2,'0'))") end # Seed pass. run_cluster_sweep(cfg; rng=MersenneTwister(31)) cat = load_catalog(cfg.cluster_catalog_path; n=n) @test !isempty(cat.clusters) members_before = sum(c.members for c in values(cat.clusters)) # A new matching file arrives; an incremental sweep must fold it in # (mode :sweep, not compact) without re-clustering the world. drop_binary(cfg.cluster_dir, vcat(magic, rand(rng, UInt8, 40)); name="arrival-01") r2 = run_cluster_sweep(cfg; rng=MersenneTwister(99)) @test r2.mode == :sweep cat2 = load_catalog(cfg.cluster_catalog_path; n=n) members_after = sum(c.members for c in values(cat2.clusters)) @test members_after == members_before + 1 # the arrival joined a cluster end end # ---------------------------------------------------------------- stats # # The counters exist to answer "which stage is the bottleneck", and every # wrong answer they could give is a wrong *attribution*: time credited to the # stage that was waiting rather than the stage that was slow. So these tests # care less about exact numbers than about what is charged to whom. @testset "per-stage stats" begin @testset "record_job! separates completions from quarantines" begin s = StageStats() record_job!(s, true, 100, 5_000_000) record_job!(s, true, 200, 5_000_000) record_job!(s, false, 50, 1_000_000) @test s.completed[] == 2 @test s.failed[] == 1 @test s.bytes[] == 350 # a quarantined job still moved bytes @test s.busy_ns[] == 11_000_000 end @testset "reset_metrics! zeroes counters and restarts the window" begin m = Metrics() Threads.atomic_add!(m.intake.files, 7) record_job!(m.stages.enrich, true, 10, 1000) m.since[] = 0.0 reset_metrics!(m) @test m.intake.files[] == 0 @test m.stages.enrich.completed[] == 0 @test m.since[] > 0.0 end @testset "worker_loop records service time, failures, and drains in_flight" begin mktempdir() do root cfg = tmp_config(root) q = ChannelQueue(10) stats = StageStats() # Two jobs that succeed, one that throws. The thrower is # quarantined by worker_loop, and must still be counted. for (i, name) in enumerate(("ok-1", "ok-2", "boom")) p = joinpath(cfg.spool_dir, "id-$i-$name") write(p, "x" ^ 10) @test enqueue!(q, Job("id-$i", name, p, filesize(p), 0.0)) end close!(q) worker_loop(1, cfg, q, (job, _, _) -> begin sleep(0.02) job.original_name == "boom" && error("handler blew up") nothing end, stats) @test stats.completed[] == 2 @test stats.failed[] == 1 @test stats.bytes[] == 30 # Each of the three handlers slept 20ms before its outcome, so # busy time covers the failure too — the work was done either way. @test stats.busy_ns[] > 3 * 15_000_000 @test stats.blocked_ns[] == 0 # nothing downstream to block on @test stats.in_flight[] == 0 # the finally in worker_loop @test isfile(joinpath(cfg.failed_dir, "id-3-boom")) end end @testset "enqueue_blocking! charges only the parked time to blocked_ns" begin s = StageStats() q = ChannelQueue(1) job = Job("id-1", "a.bin", "/tmp/a.bin", 1, 0.0) # Room available → no wait, and nothing charged. This is the common # case, and it must not pay for the instrumentation. enqueue_blocking!(q, job, s; retry_seconds = 0.01) @test length(q) == 1 @test s.blocked_ns[] == 0 # Queue full → the call parks until a consumer makes room, and that # time lands in blocked_ns, NOT in the caller's service time (which # worker_loop measures separately around the whole handler). drainer = Threads.@spawn begin sleep(0.1) dequeue!(q) end enqueue_blocking!(q, Job("id-2", "b.bin", "/tmp/b.bin", 1, 0.0), s; retry_seconds = 0.01) wait(drainer) @test length(q) == 1 @test s.blocked_ns[] > 50_000_000 # parked for ~100ms end @testset "a routing handler charges a full downstream queue as blocked" begin mktempdir() do root cfg = tmp_config(root) stats = StageStats() # Stage 3 routing a text file with the stage-4 queue already # full: it must park rather than drop, and the wait must land in # blocked_ns instead of masquerading as slow triage work. text_queue = ChannelQueue(1) @test enqueue!(text_queue, Job("filler", "f", "/tmp/f", 1, 0.0)) p = joinpath(cfg.unknown_dir, "id-t-notes.log") write(p, "plain text\n") job = Job("id-t", "notes.log", p, filesize(p), 0.0) drainer = Threads.@spawn begin sleep(0.1) dequeue!(text_queue) end handle_unknown_job(job, cfg, 1, text_queue, stats) wait(drainer) @test stats.blocked_ns[] > 50_000_000 @test length(text_queue) == 1 # the file did get through @test isfile(joinpath(cfg.text_dir, "id-t-notes.log")) end end @testset "stats_snapshot reports depth against capacity" begin mktempdir() do root cfg = tmp_config(root; worker_count = 3, known_worker_count = 4, unknown_worker_count = 5, text_worker_count = 6, queue_capacity = 11, known_queue_capacity = 12, unknown_queue_capacity = 13, text_queue_capacity = 14) m = Metrics() queues = (classify = ChannelQueue(11), enrich = ChannelQueue(12), triage = ChannelQueue(13), language = ChannelQueue(14)) @test enqueue!(queues.enrich, Job("id", "n", "/tmp/n", 1, 0.0)) record_job!(m.stages.enrich, true, 4096, 2_000_000_000) Threads.atomic_add!(m.intake.files, 9) snap = stats_snapshot(cfg, queues, m) @test length(snap.stages) == 4 @test [s.name for s in snap.stages] == ["classify", "enrich", "triage", "language"] @test [s.stage for s in snap.stages] == [1, 2, 3, 4] @test [s.workers for s in snap.stages] == [3, 4, 5, 6] @test [s.queue_capacity for s in snap.stages] == [11, 12, 13, 14] enrich = snap.stages[2] @test enrich.queue_depth == 1 @test enrich.completed == 1 @test enrich.bytes == 4096 @test enrich.busy_seconds ≈ 2.0 @test snap.intake.files == 9 @test snap.uptime_seconds >= 0 # It has to survive the trip through JSON — /stats is the only # consumer, and bin/bench.jl reads these exact field names. round_tripped = JSON3.read(JSON3.write(snap)) @test round_tripped.stages[2].busy_seconds ≈ 2.0 @test round_tripped.stages[2].blocked_seconds == 0.0 @test round_tripped.stages[2].queue_depth == 1 end end @testset "capacity is part of the queue seam" begin @test capacity(ChannelQueue(7)) == 7 end end end