Compare commits
2 Commits
fac3adbaf6
...
d9f32d9aaf
| Author | SHA1 | Date | |
|---|---|---|---|
| d9f32d9aaf | |||
| 2c8de488a1 |
@@ -2,7 +2,7 @@
|
|||||||
|
|
||||||
julia_version = "1.12.6"
|
julia_version = "1.12.6"
|
||||||
manifest_format = "2.0"
|
manifest_format = "2.0"
|
||||||
project_hash = "a623ff56053e3a56c1799a1cb2080ec48d933b73"
|
project_hash = "ed6bd1b772452682c906ce1236b89ccb1b0876fc"
|
||||||
|
|
||||||
[[deps.ADTypes]]
|
[[deps.ADTypes]]
|
||||||
git-tree-sha1 = "d9aaef7c63466eee4de23b4d9dad03629df54bea"
|
git-tree-sha1 = "d9aaef7c63466eee4de23b4d9dad03629df54bea"
|
||||||
@@ -309,7 +309,7 @@ weakdeps = ["HTTP"]
|
|||||||
HTTPExt = "HTTP"
|
HTTPExt = "HTTP"
|
||||||
|
|
||||||
[[deps.FileServer]]
|
[[deps.FileServer]]
|
||||||
deps = ["HTTP", "JLD2", "JSON3", "Logging", "Lux", "Optimisers", "Oxygen", "UUIDs", "Zygote"]
|
deps = ["HTTP", "JLD2", "JSON3", "Languages", "Logging", "Lux", "Optimisers", "Oxygen", "Random", "UUIDs", "Zygote"]
|
||||||
path = "."
|
path = "."
|
||||||
uuid = "b3f1c2d4-5e6a-4b7c-8d9e-0f1a2b3c4d5e"
|
uuid = "b3f1c2d4-5e6a-4b7c-8d9e-0f1a2b3c4d5e"
|
||||||
version = "0.1.0"
|
version = "0.1.0"
|
||||||
|
|||||||
@@ -12,6 +12,7 @@ Logging = "56ddb016-857b-54e1-b83d-db4d58db5568"
|
|||||||
Lux = "b2108857-7c20-44ae-9111-449ecde12c47"
|
Lux = "b2108857-7c20-44ae-9111-449ecde12c47"
|
||||||
Optimisers = "3bd65402-5787-11e9-1adc-39752487f4e2"
|
Optimisers = "3bd65402-5787-11e9-1adc-39752487f4e2"
|
||||||
Oxygen = "df9a0d86-3283-4920-82dc-4555fc0d1d8b"
|
Oxygen = "df9a0d86-3283-4920-82dc-4555fc0d1d8b"
|
||||||
|
Random = "9a3f8284-a2c9-5f02-9a11-845980a1fd5c"
|
||||||
UUIDs = "cf7118a7-6976-5b1a-9a39-7adc72f591a4"
|
UUIDs = "cf7118a7-6976-5b1a-9a39-7adc72f591a4"
|
||||||
Zygote = "e88e6eb3-aa80-5325-afca-941959d7151f"
|
Zygote = "e88e6eb3-aa80-5325-afca-941959d7151f"
|
||||||
|
|
||||||
@@ -24,6 +25,7 @@ Logging = "1.11.0"
|
|||||||
Lux = "1.31.4"
|
Lux = "1.31.4"
|
||||||
Optimisers = "0.4.7"
|
Optimisers = "0.4.7"
|
||||||
Oxygen = "1.10.2"
|
Oxygen = "1.10.2"
|
||||||
|
Random = "1.11.0"
|
||||||
UUIDs = "1.11.0"
|
UUIDs = "1.11.0"
|
||||||
Zygote = "0.7.11"
|
Zygote = "0.7.11"
|
||||||
|
|
||||||
|
|||||||
65
README.md
65
README.md
@@ -149,8 +149,9 @@ or printable-ASCII heuristics, it keeps non-ASCII text (accents, CJK, emoji) in
|
|||||||
UTF-8 near their start — still land in `binary/`. A NUL byte is valid UTF-8 but
|
UTF-8 near their start — still land in `binary/`. A NUL byte is valid UTF-8 but
|
||||||
not a text control byte, so it still reads as binary. A multi-byte character
|
not a text control byte, so it still reads as binary. A multi-byte character
|
||||||
split by the 8000-byte boundary is trimmed before the check so it isn't mistaken
|
split by the 8000-byte boundary is trimmed before the check so it isn't mistaken
|
||||||
for malformed bytes. An empty file is treated as text. `binary/` is terminal;
|
for malformed bytes. An empty file is treated as text. `binary/` is terminal on
|
||||||
`text/` is handed to stage 4 (`src/content.jl`).
|
the live path (but is the input the offline **stage-5 discovery** sweeps — see
|
||||||
|
below); `text/` is handed to stage 4 (`src/content.jl`).
|
||||||
|
|
||||||
### Language enrichment (stage 4)
|
### Language enrichment (stage 4)
|
||||||
|
|
||||||
@@ -199,6 +200,50 @@ Like stage 2, the sidecar is committed **before** the file is moved into
|
|||||||
`data/text_done/`, so the file's presence there always implies its sidecar is
|
`data/text_done/`, so the file's presence there always implies its sidecar is
|
||||||
present; recovery re-enriches idempotently (`src/language.jl`).
|
present; recovery re-enriches idempotently (`src/language.jl`).
|
||||||
|
|
||||||
|
### Unknown-format discovery (stage 5, offline)
|
||||||
|
|
||||||
|
The `binary/` sink from stage 3 is the pile of genuinely *unrecognized* files.
|
||||||
|
Stage 5 mines it for **recurring new file formats** by clustering files on their
|
||||||
|
header bytes — a growing catalog of discovered formats, each with a magic-byte
|
||||||
|
signature that can eventually be promoted into the classifier's fast path. Unlike
|
||||||
|
stages 1–4 it is **not on the request hot path**: it is a single-owner *batch*
|
||||||
|
process (the catalog is mutable shared state, the opposite of the stateless
|
||||||
|
classifier), and because promotion is human-gated nothing here is
|
||||||
|
latency-sensitive. The full rationale — and the assumptions we deliberately
|
||||||
|
rejected — live in [`model/DESIGN_clustering.md`](model/DESIGN_clustering.md).
|
||||||
|
|
||||||
|
The model (`src/cluster.jl`, base-Julia, no extra deps) is a Dirichlet-process
|
||||||
|
mixture of **per-position categoricals** over the first 32 header bytes, on a
|
||||||
|
257-symbol alphabet (byte `0–255` plus a `past-EOF` symbol so short fixed-length
|
||||||
|
formats are modeled honestly). Bytes are treated as **categorical, not numeric**
|
||||||
|
— `0x89` and `0x88` are not "close" — so this deliberately does *not* reuse the
|
||||||
|
classifier's `[0,1]` byte scaling. A fixed uniform **background** component
|
||||||
|
absorbs structureless (compressed/encrypted) blobs so they don't mint spurious
|
||||||
|
clusters. A cluster's spiked positions become a libmagic-style signature;
|
||||||
|
clusters with enough members and enough fixed positions self-**nominate** for
|
||||||
|
promotion (a human does the one irreversible step, redefining "known").
|
||||||
|
|
||||||
|
**Status:** the offline science (phase A) is implemented and calibrated; the live
|
||||||
|
catalog process (phase B) is designed and its scoring core (`assign_file`) is in
|
||||||
|
place, but its batch-runner plumbing is not yet built.
|
||||||
|
|
||||||
|
Calibration is its own offline script (like training — never in the request
|
||||||
|
path), scored against magic-collapsed ground truth (so `docx`≡`zip` and the whole
|
||||||
|
ELF family count as one format each, which is the *correct* answer, not an error):
|
||||||
|
|
||||||
|
```bash
|
||||||
|
julia --project=. bin/cluster_calibrate.jl [training_set_dir] # defaults to ../training_set
|
||||||
|
```
|
||||||
|
|
||||||
|
It grid-tunes the hyperparameters to maximize Adjusted Rand Index against known
|
||||||
|
formats and cross-checks against a model-free NCD (gzip) baseline. On the 700-file
|
||||||
|
training corpus the calibrated defaults (`n=32`, `α=1.0`, `β=0.1`) recover the
|
||||||
|
known formats at **ARI 0.77** (0.885 excluding tar), with `gzip`, `pkzip`
|
||||||
|
(`docx`+`zip` merged), and `jpeg` forming clean, promotable clusters; the NCD
|
||||||
|
baseline agrees. See `DESIGN_clustering.md` §11 for the full results, including the
|
||||||
|
one known limitation (ELF and these tarballs share a long run of header zero-
|
||||||
|
padding and merge — the v2 fix is inverse-entropy position weighting).
|
||||||
|
|
||||||
## The queue seam (→ RabbitMQ later)
|
## The queue seam (→ RabbitMQ later)
|
||||||
|
|
||||||
The HTTP handler and workers only ever call `enqueue!`, `dequeue!`, and
|
The HTTP handler and workers only ever call `enqueue!`, `dequeue!`, and
|
||||||
@@ -307,6 +352,13 @@ init, so the artifact is exactly regenerable from the same inputs.
|
|||||||
| `FS_MODEL_PATH` | `model/classifier.jld2` | Classifier artifact loaded at startup |
|
| `FS_MODEL_PATH` | `model/classifier.jld2` | Classifier artifact loaded at startup |
|
||||||
| `FS_EXIFTOOL_TIMEOUT` | `30` | Seconds before a stuck exiftool is killed |
|
| `FS_EXIFTOOL_TIMEOUT` | `30` | Seconds before a stuck exiftool is killed |
|
||||||
| `FS_LINGUIST_TIMEOUT` | `30` | Seconds before a stuck github-linguist is killed |
|
| `FS_LINGUIST_TIMEOUT` | `30` | Seconds before a stuck github-linguist is killed |
|
||||||
|
| `FS_CLUSTER_DIR` | `data/binary` | Stage-5 input: the unknown/binary pile to sweep |
|
||||||
|
| `FS_CLUSTER_N` | `32` | Header bytes modeled per file |
|
||||||
|
| `FS_CLUSTER_ALPHA` | `1.0` | CRP concentration (propensity to spawn new formats) |
|
||||||
|
| `FS_CLUSTER_PSEUDOCOUNT` | `0.1` | Dirichlet pseudocount β (calibrated) |
|
||||||
|
| `FS_CLUSTER_BG_MASS` | `5.0` | Mass of the uniform background component |
|
||||||
|
| `FS_PROMOTE_MIN_MEMBERS` | `20` | Cluster size threshold for promotion nomination |
|
||||||
|
| `FS_PROMOTE_MIN_MAGIC` | `3` | Required fixed signature positions to nominate |
|
||||||
|
|
||||||
> To get real parallelism, start Julia with enough threads (`-t N`) to cover all
|
> To get real parallelism, start Julia with enough threads (`-t N`) to cover all
|
||||||
> pools. If `FS_WORKERS + FS_KNOWN_WORKERS + FS_UNKNOWN_WORKERS + FS_TEXT_WORKERS`
|
> pools. If `FS_WORKERS + FS_KNOWN_WORKERS + FS_UNKNOWN_WORKERS + FS_TEXT_WORKERS`
|
||||||
@@ -346,11 +398,14 @@ src/
|
|||||||
metadata.jl exiftool extraction + normalized sidecar (stage 2)
|
metadata.jl exiftool extraction + normalized sidecar (stage 2)
|
||||||
content.jl binary-vs-text sniff for unknown files (stage 3)
|
content.jl binary-vs-text sniff for unknown files (stage 3)
|
||||||
language.jl natural + programming language enrichment for text (stage 4)
|
language.jl natural + programming language enrichment for text (stage 4)
|
||||||
|
cluster.jl header-byte clustering for unknown-format discovery (stage 5, offline)
|
||||||
worker.jl parametrized worker loop + classify/enrich/triage/language handlers
|
worker.jl parametrized worker loop + classify/enrich/triage/language handlers
|
||||||
server.jl HTTP routes/handlers
|
server.jl HTTP routes/handlers
|
||||||
bin/
|
bin/
|
||||||
server.jl entry point
|
server.jl entry point
|
||||||
train.jl offline training script → model/classifier.jld2
|
train.jl offline training script → model/classifier.jld2
|
||||||
|
cluster_calibrate.jl offline stage-5 hyperparameter calibration + NCD baseline
|
||||||
model/
|
model/
|
||||||
classifier.jld2 committed trained weights (loaded at startup)
|
classifier.jld2 committed trained weights (loaded at startup)
|
||||||
|
DESIGN_clustering.md stage-5 design rationale + calibration results
|
||||||
```
|
```
|
||||||
|
|||||||
217
bin/cluster_calibrate.jl
Normal file
217
bin/cluster_calibrate.jl
Normal file
@@ -0,0 +1,217 @@
|
|||||||
|
#!/usr/bin/env julia
|
||||||
|
#
|
||||||
|
# Phase-A calibration for stage-5 header clustering (model/DESIGN_clustering.md
|
||||||
|
# §7). Runs labeled known files through the exact clustering pipeline, scores the
|
||||||
|
# recovered partition against magic-collapsed ground truth with ARI / V-measure,
|
||||||
|
# grid-tunes (α, β, bg_mass, n), and cross-checks the winning config against a
|
||||||
|
# model-free NCD (gzip) baseline (§8). The settings printed here are the ones the
|
||||||
|
# machine rediscovers known formats at — copy the winner into config.jl.
|
||||||
|
#
|
||||||
|
# julia --project=. bin/cluster_calibrate.jl [training_set_dir]
|
||||||
|
#
|
||||||
|
# Defaults to ../training_set. Prints a report; writes nothing.
|
||||||
|
|
||||||
|
using Random
|
||||||
|
using Printf
|
||||||
|
|
||||||
|
include(joinpath(@__DIR__, "..", "src", "cluster.jl"))
|
||||||
|
|
||||||
|
# --- ground truth: magic-collapsed classes, NOT extensions (DESIGN §7.2) -----
|
||||||
|
|
||||||
|
"""
|
||||||
|
truth_label(path) -> String
|
||||||
|
|
||||||
|
The magic-collapsed format class of a file, read from its actual bytes (so
|
||||||
|
docx≡zip and the whole ELF family merge, exactly the answer we want the
|
||||||
|
clustering to reproduce). `tar` is detected by the `ustar` magic at offset 257 —
|
||||||
|
outside the model's front window, so tars are the accepted blind spot that
|
||||||
|
scatters to background.
|
||||||
|
"""
|
||||||
|
function truth_label(path::AbstractString)
|
||||||
|
b = zeros(UInt8, 262)
|
||||||
|
open(path) do io
|
||||||
|
chunk = read(io, 262)
|
||||||
|
copyto!(b, 1, chunk, 1, length(chunk))
|
||||||
|
end
|
||||||
|
b[1] == 0x1f && b[2] == 0x8b && return "gzip"
|
||||||
|
b[1] == 0x50 && b[2] == 0x4b && return "pkzip"
|
||||||
|
b[1] == 0x25 && b[2] == 0x50 && b[3] == 0x44 && b[4] == 0x46 && return "pdf"
|
||||||
|
b[1] == 0xff && b[2] == 0xd8 && b[3] == 0xff && return "jpeg"
|
||||||
|
b[1] == 0x7f && b[2] == 0x45 && b[3] == 0x4c && b[4] == 0x46 && return "elf"
|
||||||
|
(b[258] == 0x75 && b[259] == 0x73 && b[260] == 0x74 && b[261] == 0x61 && b[262] == 0x72) && return "tar"
|
||||||
|
return "other"
|
||||||
|
end
|
||||||
|
|
||||||
|
# --- NCD (Normalized Compression Distance) baseline, model-free (DESIGN §8) ---
|
||||||
|
|
||||||
|
"gzip-compressed size of a byte buffer, via the gzip CLI (no CodecZlib dep)."
|
||||||
|
function gz_size(bytes::Vector{UInt8})
|
||||||
|
out = IOBuffer()
|
||||||
|
open(pipeline(`gzip -c`; stdout=out); write=true) do io
|
||||||
|
write(io, bytes)
|
||||||
|
end
|
||||||
|
return length(take!(out))
|
||||||
|
end
|
||||||
|
|
||||||
|
"NCD(x,y) = (C(xy) - min(C(x),C(y))) / max(C(x),C(y)) — 0 = identical, ~1 = unrelated."
|
||||||
|
function ncd(xb, yb, cx, cy)
|
||||||
|
cxy = gz_size(vcat(xb, yb))
|
||||||
|
return (cxy - min(cx, cy)) / max(cx, cy)
|
||||||
|
end
|
||||||
|
|
||||||
|
"""
|
||||||
|
ncd_1nn_purity(paths, truth; head_bytes) -> Float64
|
||||||
|
|
||||||
|
Fraction of files whose NCD-nearest neighbour shares its true label — a cheap,
|
||||||
|
O(N²) sanity read on how well raw gzip-similarity alone separates formats on the
|
||||||
|
same input. The Bayesian clusters should broadly agree; a big gap is a red flag
|
||||||
|
(DESIGN §10.3). Uses each file's first `head_bytes` so the giant files don't
|
||||||
|
dominate compression time.
|
||||||
|
"""
|
||||||
|
function ncd_1nn_purity(paths::Vector{String}, truth::Vector{String}; head_bytes::Int=4096)
|
||||||
|
bufs = map(paths) do p
|
||||||
|
open(io -> read(io, head_bytes), p)
|
||||||
|
end
|
||||||
|
csz = gz_size.(bufs)
|
||||||
|
N = length(paths)
|
||||||
|
correct = 0
|
||||||
|
for i in 1:N
|
||||||
|
best_j = 0; best_d = Inf
|
||||||
|
for j in 1:N
|
||||||
|
i == j && continue
|
||||||
|
d = ncd(bufs[i], bufs[j], csz[i], csz[j])
|
||||||
|
if d < best_d
|
||||||
|
best_d = d; best_j = j
|
||||||
|
end
|
||||||
|
end
|
||||||
|
best_j != 0 && truth[best_j] == truth[i] && (correct += 1)
|
||||||
|
end
|
||||||
|
return correct / N
|
||||||
|
end
|
||||||
|
|
||||||
|
"1-NN label purity of a *cluster* assignment vs truth (same yardstick as NCD's)."
|
||||||
|
function cluster_1nn_purity(pred::Vector{Int}, truth::Vector{String})
|
||||||
|
# For each file, its 'nearest neighbour' is any other file in the same
|
||||||
|
# cluster; purity = P(a random same-cluster neighbour shares the true label).
|
||||||
|
groups = Dict{Int,Vector{Int}}()
|
||||||
|
for (i, k) in enumerate(pred)
|
||||||
|
push!(get!(groups, k, Int[]), i)
|
||||||
|
end
|
||||||
|
correct = 0; total = 0
|
||||||
|
for (_, idxs) in groups
|
||||||
|
length(idxs) < 2 && continue
|
||||||
|
for i in idxs
|
||||||
|
same = count(j -> j != i && truth[j] == truth[i], idxs)
|
||||||
|
total += 1
|
||||||
|
same > 0 && (correct += 1)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return total == 0 ? 0.0 : correct / total
|
||||||
|
end
|
||||||
|
|
||||||
|
# --- data ---------------------------------------------------------------------
|
||||||
|
|
||||||
|
function load_corpus(dir::AbstractString)
|
||||||
|
paths = String[]
|
||||||
|
for name in readdir(dir; join=true)
|
||||||
|
isfile(name) && push!(paths, name)
|
||||||
|
end
|
||||||
|
truth = truth_label.(paths)
|
||||||
|
return paths, truth
|
||||||
|
end
|
||||||
|
|
||||||
|
# --- grid search --------------------------------------------------------------
|
||||||
|
|
||||||
|
function evaluate(X, truth; α, β, bg_mass, sweeps, restarts, seed)
|
||||||
|
r = gibbs_cluster(X; α=α, β=β, bg_mass=bg_mass, sweeps=sweeps,
|
||||||
|
restarts=restarts, rng=MersenneTwister(seed))
|
||||||
|
pred = r.assignments
|
||||||
|
ari = adjusted_rand_index(truth, pred)
|
||||||
|
keep = truth .!= "tar"
|
||||||
|
ari_notar = adjusted_rand_index(truth[keep], pred[keep])
|
||||||
|
v, h, comp = v_measure(truth, pred)
|
||||||
|
return (; ari, ari_notar, v, h, comp, k=length(r.clusters),
|
||||||
|
bg=count(==(0), pred), result=r)
|
||||||
|
end
|
||||||
|
|
||||||
|
function main()
|
||||||
|
dir = length(ARGS) >= 1 ? ARGS[1] : joinpath(@__DIR__, "..", "..", "training_set")
|
||||||
|
isdir(dir) || error("training set dir not found: $dir")
|
||||||
|
paths, truth = load_corpus(dir)
|
||||||
|
classes = sort(unique(truth))
|
||||||
|
counts = [(c, count(==(c), truth)) for c in classes]
|
||||||
|
@printf("corpus: %d files from %s\n", length(paths), dir)
|
||||||
|
println("magic-collapsed truth classes: ", join(["$c=$n" for (c, n) in counts], " "))
|
||||||
|
println()
|
||||||
|
|
||||||
|
sweeps = 150
|
||||||
|
restarts = 6
|
||||||
|
seed = 20260703
|
||||||
|
|
||||||
|
# Grid. n is expensive to re-featurize, so loop it outermost. Ranges are
|
||||||
|
# centred where the coarse sweep found the optimum: small β (peaked
|
||||||
|
# per-position priors) is what separates formats whose headers differ in only
|
||||||
|
# a few magic bytes; large β over-merges. bg_mass barely moves the result
|
||||||
|
# here (almost nothing lands in background on this corpus), so it is fixed.
|
||||||
|
αs = [1.0, 2.0]
|
||||||
|
βs = [0.05, 0.08, 0.1, 0.15, 0.2]
|
||||||
|
bgs = [5.0]
|
||||||
|
ns = [32, 64]
|
||||||
|
|
||||||
|
println("grid search (sweeps=$sweeps, restarts=$restarts):")
|
||||||
|
@printf(" %-4s %-5s %-5s %-6s | %-6s %-8s %-6s %-6s %-6s %-4s %-4s\n",
|
||||||
|
"n", "alpha", "beta", "bgmss", "ARI", "ARI-tar", "V", "homog", "compl", "k", "bg")
|
||||||
|
results = Vector{Any}()
|
||||||
|
for n in ns
|
||||||
|
X = header_matrix(paths; n=n)
|
||||||
|
for α in αs, β in βs, bg in bgs
|
||||||
|
e = evaluate(X, truth; α=α, β=β, bg_mass=bg, sweeps=sweeps, restarts=restarts, seed=seed)
|
||||||
|
push!(results, (; n, α, β, bg, e))
|
||||||
|
@printf(" %-4d %-5.1f %-5.2f %-6.1f | %-6.3f %-8.3f %-6.3f %-6.3f %-6.3f %-4d %-4d\n",
|
||||||
|
n, α, β, bg, e.ari, e.ari_notar, e.v, e.h, e.comp, e.k, e.bg)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
# Rank by ARI-excluding-tar (tar is the accepted blind spot; scoring it would
|
||||||
|
# penalise the correct answer of scattering tars to background — DESIGN §7.2).
|
||||||
|
sort!(results; by=r -> r.e.ari_notar, rev=true)
|
||||||
|
best = results[1]
|
||||||
|
println()
|
||||||
|
@printf("BEST (by ARI excl. tar): n=%d α=%.1f β=%.2f bg_mass=%.1f\n",
|
||||||
|
best.n, best.α, best.β, best.bg)
|
||||||
|
@printf(" ARI=%.3f ARI(excl tar)=%.3f V=%.3f homogeneity=%.3f completeness=%.3f clusters=%d background=%d\n",
|
||||||
|
best.e.ari, best.e.ari_notar, best.e.v, best.e.h, best.e.comp, best.e.k, best.e.bg)
|
||||||
|
|
||||||
|
# Per-cluster composition of the winning partition, and promotion nominations.
|
||||||
|
pred = best.e.result.assignments
|
||||||
|
println("\nwinning partition — cluster composition (truth breakdown):")
|
||||||
|
for (id, c) in sort(collect(best.e.result.clusters); by=x -> -x[2].members)
|
||||||
|
members = [truth[i] for i in eachindex(pred) if pred[i] == id]
|
||||||
|
comp = sort([(l, count(==(l), members)) for l in unique(members)]; by=x -> -x[2])
|
||||||
|
sig = signature(c)
|
||||||
|
promo = is_promotable(c, sig; min_members=20, min_magic=3) ? " ✓NOMINATE" : ""
|
||||||
|
@printf(" cluster %-4d n=%-3d magic=%-2d %s%s\n",
|
||||||
|
id, c.members, magic_positions(sig),
|
||||||
|
join(["$l:$k" for (l, k) in comp], " "), promo)
|
||||||
|
end
|
||||||
|
nbg = count(==(0), pred)
|
||||||
|
bg_truth = [truth[i] for i in eachindex(pred) if pred[i] == 0]
|
||||||
|
bgc = sort([(l, count(==(l), bg_truth)) for l in unique(bg_truth)]; by=x -> -x[2])
|
||||||
|
@printf(" background n=%-3d %s\n", nbg, join(["$l:$k" for (l, k) in bgc], " "))
|
||||||
|
|
||||||
|
# NCD baseline cross-check on a subsample (O(N²), so keep it small).
|
||||||
|
println("\nNCD (gzip) baseline cross-check:")
|
||||||
|
subn = min(150, length(paths))
|
||||||
|
sub = shuffle(MersenneTwister(seed), collect(1:length(paths)))[1:subn]
|
||||||
|
subpaths = paths[sub]; subtruth = truth[sub]
|
||||||
|
ncd_pur = ncd_1nn_purity(subpaths, subtruth)
|
||||||
|
Xsub = header_matrix(subpaths; n=best.n)
|
||||||
|
rsub = gibbs_cluster(Xsub; α=best.α, β=best.β, bg_mass=best.bg,
|
||||||
|
sweeps=sweeps, restarts=restarts, rng=MersenneTwister(seed))
|
||||||
|
bay_pur = cluster_1nn_purity(rsub.assignments, subtruth)
|
||||||
|
@printf(" subsample=%d NCD 1-NN label purity=%.3f Bayesian same-cluster purity=%.3f\n",
|
||||||
|
subn, ncd_pur, bay_pur)
|
||||||
|
println(" (both high ⇒ header-byte signal agrees with model-free gzip similarity — DESIGN §10.3)")
|
||||||
|
end
|
||||||
|
|
||||||
|
main()
|
||||||
305
model/DESIGN_clustering.md
Normal file
305
model/DESIGN_clustering.md
Normal file
@@ -0,0 +1,305 @@
|
|||||||
|
# Stage-5: Unknown-format discovery by Bayesian header clustering
|
||||||
|
|
||||||
|
Status: **phase A implemented and calibrated** (`src/cluster.jl`,
|
||||||
|
`bin/cluster_calibrate.jl`, tests in `test/runtests.jl`); phase-B core scoring
|
||||||
|
implemented (`assign_file`), its live batch-process plumbing still to do. Product
|
||||||
|
of a design interview; captures the decisions and — as important — the
|
||||||
|
assumptions we *rejected* so they don't get silently reintroduced. §11 records
|
||||||
|
what building it actually taught us, including three assumptions in this document
|
||||||
|
that the data corrected.
|
||||||
|
|
||||||
|
## 1. Goal
|
||||||
|
|
||||||
|
Discover **recurring new file formats** hiding in the `binary/` bucket (the
|
||||||
|
`:unknown` sink from `classify.jl` → stage-3 triage). A genuinely novel format is
|
||||||
|
a plausible proxy for a genuinely novel producing application, but we do **not**
|
||||||
|
try to identify producers directly (see §3). The output is a **growing catalog of
|
||||||
|
discovered formats**, each with a magic-byte signature that can be promoted into
|
||||||
|
the classifier's fast path.
|
||||||
|
|
||||||
|
Task shape (settled): **unsupervised clustering with an unknown number of
|
||||||
|
clusters.** Not pairwise "same producer" scoring, not classification against a
|
||||||
|
fixed label set.
|
||||||
|
|
||||||
|
## 2. Two phases — build (A) then run (B)
|
||||||
|
|
||||||
|
**(A) Batch, offline — the science.** Cluster the accumulated pile from scratch.
|
||||||
|
Its job is *not* to be the catalog; it is to (i) prove the header-byte signal
|
||||||
|
actually separates formats, cross-checked against an NCD baseline (§8), and
|
||||||
|
(ii) **calibrate hyperparameters** against known formats (§7). Ship this first —
|
||||||
|
it de-risks (B). If (A)'s clusters are garbage, (B)'s machinery is wasted.
|
||||||
|
|
||||||
|
**(B) Online, live — the catalog.** The target deliverable. A persistent catalog
|
||||||
|
where each discovered format has a **durable, frozen ID** and stored sufficient
|
||||||
|
statistics. New unknown files are scored against existing clusters; only
|
||||||
|
genuinely novel ones spawn a new entry. Clusters that accumulate enough evidence
|
||||||
|
are **nominated for promotion** into the classifier (§6).
|
||||||
|
|
||||||
|
## 3. What we are and are NOT clustering
|
||||||
|
|
||||||
|
We cluster by **file format**, not by producer. The first-*n* header bytes are
|
||||||
|
format-mandated and producer-invariant: every valid PNG shares the same magic
|
||||||
|
regardless of which program wrote it; a PDF's producer string lives deep inside
|
||||||
|
the file, not in the header. Producer identity, where recoverable at all, is
|
||||||
|
`exiftool`'s job (stage 2), not this stage's.
|
||||||
|
|
||||||
|
Corollary already visible in `../training_set`: extension labels are **not**
|
||||||
|
header-format labels. `docx` *is* a PK zip; `so`/`o`/`elf`/`out` are all ELF.
|
||||||
|
Merging those is **correct**, not error (see §7).
|
||||||
|
|
||||||
|
## 4. Model: DP mixture of per-position categoricals
|
||||||
|
|
||||||
|
A cluster is a **product of independent per-position categorical distributions**
|
||||||
|
over the first *n* header bytes. Position *i* carries a distribution `θᵢ` over a
|
||||||
|
**257-symbol alphabet**: byte values `0–255`, plus symbol `256 = "past EOF"`.
|
||||||
|
|
||||||
|
- Invariant positions (magic bytes) learn a spiked `θᵢ`; variable positions
|
||||||
|
(lengths, timestamps) learn a flat one. A cluster's signature = the vector of
|
||||||
|
modal symbols + per-position peakedness. That signature **is a magic-number
|
||||||
|
template** — this is the entire reason for the categorical choice.
|
||||||
|
- `257` alphabet handles short files honestly: a format that is always 20 bytes
|
||||||
|
produces a spiked "past-EOF" at positions 20–31, which is real, discriminative
|
||||||
|
signal. No zero-padding (would collide `0x00` padding with real `0x00` bytes).
|
||||||
|
|
||||||
|
**Priors:** Dirichlet on each `θᵢ` (conjugate to Categorical); **Dirichlet
|
||||||
|
process (CRP)** over cluster assignments → unknown *k* falls out natively.
|
||||||
|
|
||||||
|
**Why categorical, not Euclidean.** Bytes are categorical, not ordinal: `0x89`
|
||||||
|
and `0x88` are not "close," `0x00` and `0xFF` are not "far." k-means / Gaussian
|
||||||
|
mixtures over scaled bytes assert a metric that does not exist in header space.
|
||||||
|
**Do not reuse `model.jl`'s `[0,1]` byte scaling here** — that scaling is correct
|
||||||
|
for the Lux net and wrong for this model. We need the raw `0–255` byte as a
|
||||||
|
categorical index.
|
||||||
|
|
||||||
|
### 4a. Background component (high-entropy handling)
|
||||||
|
|
||||||
|
Add a fixed, **non-adaptive uniform component** (each position uniform over 257)
|
||||||
|
as the "junk drawer." Compressed/encrypted/structureless blobs are ~uniform after
|
||||||
|
any magic and would otherwise either (i) mint a singleton per file or (ii)
|
||||||
|
collapse into one flat cluster that then matches everything. The background
|
||||||
|
absorbs them cleanly.
|
||||||
|
|
||||||
|
Two populations, to be precise:
|
||||||
|
- **Structured prefix + random tail** (gzip `1f 8b`, PK zip, zstd, most encrypted
|
||||||
|
*containers*): peaked at positions 0–3, flat after. These form **real clusters
|
||||||
|
for free** — genuine discoveries, no special handling.
|
||||||
|
- **Uniform from byte 0** (raw encrypted streams, key material): nothing in the
|
||||||
|
header to cluster on → absorbed by background.
|
||||||
|
|
||||||
|
The background is **never promotable**. But it is **not a silent sink**: its
|
||||||
|
size / growth / entropy histogram is surfaced as a first-class signal ("12% of
|
||||||
|
this week's unknowns are structureless"). If sub-clustering the structureless
|
||||||
|
residue ever matters, that needs a *different* feature (byte histogram / entropy),
|
||||||
|
a separate v3 model — header bytes genuinely cannot do it.
|
||||||
|
|
||||||
|
### 4b. Feature window
|
||||||
|
|
||||||
|
**Front-only, `n = 32`** (config knob; try 64 if under-resolved). Magic lives at
|
||||||
|
offset 0. Tail window **deferred to v2** — a minority of formats have trailers
|
||||||
|
(ZIP EOCD, ID3v1, PDF `%%EOF`); add as an independent *second block* of positions
|
||||||
|
only if real trailer-formats show up in the residue.
|
||||||
|
|
||||||
|
**Known blind spot: tar.** `ustar` magic is at **offset 257**, outside the
|
||||||
|
window, so all 100 training tars scatter to background. Accepted for v1 — tar is
|
||||||
|
already a *known* format, so discovery doesn't need it. General lesson: a minority
|
||||||
|
of formats put magic at a fixed deeper offset; the fix (if ever needed) is a
|
||||||
|
**sparse probe window** at that offset (e.g. bytes 257–262 as a third block), not
|
||||||
|
densely modeling 257 front bytes — that would 8× every cluster's `n×257`
|
||||||
|
sufficient-stat table to catch one format.
|
||||||
|
|
||||||
|
## 5. Inference: different mode per phase (resolves the Bayesian-vs-catalog tension)
|
||||||
|
|
||||||
|
A sampler yields a *posterior over partitions*; a catalog needs *one partition
|
||||||
|
with durable IDs*. Two MCMC gotchas: **label switching** (cluster #3 is not a
|
||||||
|
stable identity across iterations/runs) and **distribution-not-answer** (1000
|
||||||
|
partitions, not one). We sidestep both by using two inference modes:
|
||||||
|
|
||||||
|
- **Phase (A), offline:** full **collapsed Gibbs** sampler over the
|
||||||
|
Dirichlet-Categorical (conjugacy → ~100 lines, no continuous approximation,
|
||||||
|
unknown *k* native). Used to validate signal, tune `α` + Dirichlet strength,
|
||||||
|
and seed the initial catalog (summarize to a point partition **once**, via a
|
||||||
|
VI/Binder loss over the posterior similarity matrix — tolerated because it is
|
||||||
|
offline, never in the hot path).
|
||||||
|
- **Phase (B), live:** **deterministic sequential CRP-predictive assignment.**
|
||||||
|
Each catalog cluster stores per-position 257-count vectors (sufficient stats).
|
||||||
|
A new file's CRP predictive probability of joining each existing cluster vs.
|
||||||
|
the background vs. spawning a new cluster is computed; assign to the argmax.
|
||||||
|
A new cluster is minted only if the new-cluster evidence beats the background
|
||||||
|
by a margin. **IDs are frozen at birth → no label switching.** This is exactly
|
||||||
|
the Gibbs predictive rule with existing assignments held fixed — same math, not
|
||||||
|
an ad-hoc hack.
|
||||||
|
- **Periodic compaction, offline:** re-run Gibbs seeded from the current catalog
|
||||||
|
to merge drifted clusters / split bloated ones.
|
||||||
|
|
||||||
|
## 6. Promotion (closing the loop to the classifier)
|
||||||
|
|
||||||
|
**Layered known-check at ingest** becomes:
|
||||||
|
1. Match against **promoted signatures** (exact, fast) — runs *before* the net.
|
||||||
|
2. Else the Lux `:known` / `:unknown` classifier.
|
||||||
|
3. Else route to `binary/` for this stage.
|
||||||
|
|
||||||
|
**Promotion = append a magic-byte signature to a registry.** A cluster's spiked
|
||||||
|
positions (posterior max-prob `> ~0.9`) become required bytes; flat positions
|
||||||
|
become wildcards — a libmagic-style signature. This is a **data change, not a
|
||||||
|
retrain**; interpretable, auditable, reversible. Retraining the Lux net is a
|
||||||
|
separate, *optional periodic* activity using accumulated signature-labeled files,
|
||||||
|
never the promotion mechanism itself.
|
||||||
|
|
||||||
|
**Nominate automatically, activate by hand.** A cluster crossing thresholds —
|
||||||
|
`≥ N` members (start `N ≈ 20–50`, loose dial since a human is the backstop) **and**
|
||||||
|
`≥ ~3` magic positions **and** not the background — is written to a `nominated/`
|
||||||
|
registry with its signature, member count, and example files. A human glance
|
||||||
|
promotes it into the active set. Human gate guards the one hard-to-reverse action
|
||||||
|
(redefining "known"); everything upstream stays automatic.
|
||||||
|
|
||||||
|
## 7. Calibration: recover known formats, then trust on unknowns
|
||||||
|
|
||||||
|
Do not pick priors blind. We have ground truth: `../training_set` (100 each of
|
||||||
|
tgz/tar/pdf/docx, 98 zip, 93 jpg, ELF family) and the `data/done` corpus.
|
||||||
|
|
||||||
|
1. Run **labeled known files** through the exact clustering pipeline.
|
||||||
|
2. Ground truth = **magic-collapsed classes**, *not* extensions:
|
||||||
|
`{gzip (tgz), PKzip (docx≡zip), ELF (so/o/elf/out/x86_64), JPEG, PDF, tar}`.
|
||||||
|
Merging docx+zip and the ELF family is the **correct** answer — scoring
|
||||||
|
against raw extensions would penalize correctness and mistune `α`.
|
||||||
|
3. Measure recovered-vs-truth agreement with **Adjusted Rand Index / V-measure**.
|
||||||
|
4. **Grid-tune `α` and the Dirichlet pseudocount to maximize agreement** — the
|
||||||
|
settings at which the machine rediscovers formats we already know.
|
||||||
|
5. Freeze, deploy on the `:unknown` pile.
|
||||||
|
|
||||||
|
Splitting docx from zip is a **later tier**: the discriminating info
|
||||||
|
(central-directory filenames like `word/document.xml`) sits at a *variable
|
||||||
|
offset*, not a fixed position — a different feature problem, deferred.
|
||||||
|
|
||||||
|
## 8. Julia package surface
|
||||||
|
|
||||||
|
- **Hand-rolled collapsed Gibbs** over Dirichlet-Categorical — recommended. The
|
||||||
|
conjugacy makes it short/fast; we own the online + promotion logic; no library
|
||||||
|
impedance. `Distributions.jl` for `Dirichlet`/`Categorical` primitives.
|
||||||
|
- **`CodecZlib`** for the **NCD (Normalized Compression Distance)** baseline —
|
||||||
|
model-free gzip-similarity clustering. Excellent at format grouping and a great
|
||||||
|
step-(A) sanity check, but O(N²), non-generative (no signature, no online
|
||||||
|
scoring, no promotion) → **baseline only, cannot be the catalog.**
|
||||||
|
- **`Clustering.jl`** — `randindex` / `vmeasure` for the §7 calibration metric,
|
||||||
|
plus a throwaway k-modes-ish baseline. **Not** the real model (its k-means is
|
||||||
|
the Euclidean trap of §4).
|
||||||
|
- **`Turing.jl`** — considered and rejected: discrete assignment latents + DP are
|
||||||
|
awkward, and we'd still hand-roll the online path. Overkill.
|
||||||
|
|
||||||
|
## 9. Architecture: single-owner batch stage, NOT inline inference
|
||||||
|
|
||||||
|
The classifier is stateless, immutable, shared read-only across worker threads
|
||||||
|
(see `classify.jl`). **The catalog is the opposite: mutable, learned, shared** —
|
||||||
|
every assigned file updates a cluster's counts. It therefore must **not** copy the
|
||||||
|
classifier's concurrency model (concurrent workers → lock contention, torn reads
|
||||||
|
of sufficient stats, CRP assignment against stale mass).
|
||||||
|
|
||||||
|
Because **promotion is human-gated, nothing here is latency-sensitive.** So:
|
||||||
|
|
||||||
|
- Workers stay stateless — they deposit `:unknown` files into `binary/` exactly as
|
||||||
|
today. **No catalog access on the hot path.**
|
||||||
|
- A **separate stage-5 process** (periodic / cron, single-threaded) owns the
|
||||||
|
catalog **exclusively**: sweeps newly-arrived `binary/` files, runs sequential
|
||||||
|
CRP-predictive assignment, updates sufficient stats, writes nominations.
|
||||||
|
**One writer, zero locks, no cross-thread shared mutable state.**
|
||||||
|
- The catalog is a **durable file** mutated by one process — reuse the stage-2
|
||||||
|
**sidecar-first durable-commit** discipline (`commit_enriched!`: temp write →
|
||||||
|
fsync bytes → rename → fsync dir) so a crash can't corrupt it or lose a rename.
|
||||||
|
|
||||||
|
This slots in as a batch stage, matching how stages 2/3/4 already work. New
|
||||||
|
`Config` knobs follow the existing `FS_*` env-override convention (e.g.
|
||||||
|
`FS_CLUSTER_DIR`, `FS_CLUSTER_N`, `FS_CLUSTER_ALPHA`, `FS_CLUSTER_PSEUDOCOUNT`,
|
||||||
|
`FS_PROMOTE_MIN_MEMBERS`).
|
||||||
|
|
||||||
|
## 10. Concrete test assertions (write these first)
|
||||||
|
|
||||||
|
1. **Discovers nothing from noise.** Current `data/binary` = 20 small random
|
||||||
|
blobs + 1 giant PDF. Correct output: PDF is a singleton that **never promotes**
|
||||||
|
(N=1), 20 blobs absorbed by background, **zero promoted clusters.** Any
|
||||||
|
promoted cluster from this pile = broken.
|
||||||
|
2. **Recovers known formats.** On a `../training_set` sample, calibrated settings
|
||||||
|
cluster into the ~6 magic-collapsed classes with high ARI (docx+zip merged,
|
||||||
|
ELF family merged, tar in background as the accepted blind spot).
|
||||||
|
3. **NCD agreement.** Step-(A) Bayesian clusters broadly agree with the NCD
|
||||||
|
baseline on the same input; large disagreement is a red flag to investigate
|
||||||
|
before trusting the generative model.
|
||||||
|
|
||||||
|
## 11. Implementation status & calibration results (v1)
|
||||||
|
|
||||||
|
**Shipped.** `src/cluster.jl` — feature extraction (`header_symbols`, 257-symbol
|
||||||
|
alphabet), collapsed Gibbs (`gibbs_cluster`, phase A), the sequential
|
||||||
|
CRP-predictive rule (`assign_file`, phase B core), signatures/promotion
|
||||||
|
(`signature`, `is_promotable`), and calibration metrics (`adjusted_rand_index`,
|
||||||
|
`v_measure`). All base-Julia — a base-only Lanczos `loggamma` keeps the
|
||||||
|
Dirichlet-multinomial marginal dependency-free (no Manifest churn). Config knobs
|
||||||
|
`FS_CLUSTER_*` (§9) added. `bin/cluster_calibrate.jl` runs the §7 grid and the §8
|
||||||
|
NCD baseline. Concrete §10 assertions are in the test suite (hermetic synthetic
|
||||||
|
corpora, so they need neither `../training_set` nor gzip).
|
||||||
|
|
||||||
|
**Calibrated defaults** (grid over the 700-file `training_set`, ranked by ARI
|
||||||
|
excluding tar): **n=32, α=1.0, β=0.1, bg_mass=5.0** → ARI **0.77** (0.885 excl.
|
||||||
|
tar), V-measure 0.83, homogeneity 0.87. Clusters are clean and promotable:
|
||||||
|
`pkzip:197` (docx+zip correctly merged, §7.2 ✓), `gzip:100`, `jpeg`, and several
|
||||||
|
`pdf` clusters all self-nominate. The **NCD baseline agrees** (§10.3): on a
|
||||||
|
150-file subsample, NCD 1-NN label purity 0.90 vs. the model's same-cluster
|
||||||
|
purity 0.987 — the generative header model separates formats at least as well as
|
||||||
|
model-free gzip similarity.
|
||||||
|
|
||||||
|
### Three assumptions the data corrected
|
||||||
|
|
||||||
|
1. **Tar is not in the background here; it merges into ELF.** §4b/§7 assumed
|
||||||
|
tar's `ustar`-at-257 magic is out of window so tars scatter to background. But
|
||||||
|
98/100 tars in the corpus are Hex/Elixir package tarballs whose *first
|
||||||
|
archived file is named `VERSION`* → a constant, strongly-peaked `VERSION\0`
|
||||||
|
prefix at offset 0. They do form a peaked cluster — but it **merges with ELF**,
|
||||||
|
because ELF's ident padding and tar's name-field zero-padding give the two a
|
||||||
|
long shared run of `0x00` in bytes 5–31; they differ in only ~3 magic bytes,
|
||||||
|
and 32 equally-weighted positions let ~20 shared zeros outvote 3 real ones. No
|
||||||
|
β both separates ELF/tar and keeps the other formats whole. The honest v1
|
||||||
|
position: this is the *same* "tar is hard" reality §4b flagged, just wearing a
|
||||||
|
different mask. **Fix (v2):** weight positions by inverse entropy so a
|
||||||
|
low-information shared-zero run stops dominating a few high-information magic
|
||||||
|
bytes — this generalizes beyond tar and is the highest-value next lever.
|
||||||
|
|
||||||
|
2. **You cannot cold-start every point in the background.** A natural reading of
|
||||||
|
§4a/§5 is "everything starts in the junk drawer, real clusters condense out."
|
||||||
|
That **deadlocks**: at a format's first file a fresh cluster and the background
|
||||||
|
are *both* uniform, so with the `bg_mass ≥ α` that §4a needs for absorption,
|
||||||
|
the background always wins and no cluster is ever seeded. Fix: **initialize
|
||||||
|
every file in its own singleton**; same-format singletons merge and snowball,
|
||||||
|
while a lone random-blob singleton dissolves on resample and is reclaimed by
|
||||||
|
the (stickier) background. Absorption still works — just not as the *initial*
|
||||||
|
state.
|
||||||
|
|
||||||
|
3. **Two pieces of math that look optional but aren't.** (a) Signature peakedness
|
||||||
|
is a **Bernoulli** question ("is this position fixed to byte v?"), so it uses a
|
||||||
|
2-way posterior `(count+β)/(members+2β)`, **not** the 257-way mixture
|
||||||
|
predictive — the alphabet-wide denominator drags even a unanimous position
|
||||||
|
below 0.9 once β<1, which would make promotion *impossible*. (b) Ranking Gibbs
|
||||||
|
restarts needs the **collapsed Dirichlet-multinomial marginal** (with its
|
||||||
|
`loggamma` normalizer / Occam penalty); a plain product-of-predictives score
|
||||||
|
omits the penalty and actively **rewards merging** everything into one blob
|
||||||
|
(observed, then fixed).
|
||||||
|
|
||||||
|
### Known v1 limitations (accepted)
|
||||||
|
|
||||||
|
- **β=0.1 over-splits** PDF and JPEG into several *pure* sub-clusters (e.g. PDF by
|
||||||
|
version byte). This costs completeness/ARI but not the mission: each sub-cluster
|
||||||
|
still carries valid magic and promotes independently, and a human dedupes
|
||||||
|
overlapping `%PDF-1.x` nominations at the gate.
|
||||||
|
- The point partition is the **best of N Gibbs restarts by marginal likelihood**,
|
||||||
|
a MAP-style stand-in for the VI/Binder posterior summary §5 defers — adequate
|
||||||
|
because the formats are strongly separated; revisit if compaction (§5) needs it.
|
||||||
|
- Phase B's **live single-owner batch process** (§9) and the durable catalog file
|
||||||
|
are not yet built; `assign_file` is the scoring core they will wrap.
|
||||||
|
|
||||||
|
## Open items (deferred, intentionally)
|
||||||
|
|
||||||
|
- **v2, now top priority: inverse-entropy position weighting** (unblocks ELF/tar
|
||||||
|
and any format pair that shares a long constant run — see §11).
|
||||||
|
|
||||||
|
- v2: tail-window block; sparse deep-offset probe (tar-class).
|
||||||
|
- v3: sub-clustering structureless high-entropy residue (needs entropy/histogram
|
||||||
|
feature, not header bytes).
|
||||||
|
- Later tier: docx-vs-zip split via variable-offset central-directory names.
|
||||||
|
- Periodic Lux retrain from accumulated signature-labeled files.
|
||||||
@@ -1,6 +1,7 @@
|
|||||||
module FileServer
|
module FileServer
|
||||||
|
|
||||||
using Logging
|
using Logging
|
||||||
|
using Random
|
||||||
using UUIDs
|
using UUIDs
|
||||||
using HTTP
|
using HTTP
|
||||||
using JSON3
|
using JSON3
|
||||||
@@ -18,6 +19,7 @@ include("classify.jl") # Classifier + load_classifier/classify (needs model.jl)
|
|||||||
include("metadata.jl") # exiftool extraction + sidecar enrichment (stage 2)
|
include("metadata.jl") # exiftool extraction + sidecar enrichment (stage 2)
|
||||||
include("content.jl") # binary-vs-text triage for unknown files (stage 3)
|
include("content.jl") # binary-vs-text triage for unknown files (stage 3)
|
||||||
include("language.jl") # natural + programming language enrichment for text (stage 4)
|
include("language.jl") # natural + programming language enrichment for text (stage 4)
|
||||||
|
include("cluster.jl") # unknown-format discovery by header clustering (stage 5)
|
||||||
include("worker.jl")
|
include("worker.jl")
|
||||||
|
|
||||||
# Globals the HTTP handlers read at request time. Set once in `run`, before the
|
# Globals the HTTP handlers read at request time. Set once in `run`, before the
|
||||||
|
|||||||
518
src/cluster.jl
Normal file
518
src/cluster.jl
Normal file
@@ -0,0 +1,518 @@
|
|||||||
|
# Stage-5: unknown-format discovery by Bayesian header clustering.
|
||||||
|
#
|
||||||
|
# See model/DESIGN_clustering.md for the full rationale. In brief: files that
|
||||||
|
# stage-3 sorted into `binary/` are the `:unknown` sink — genuinely
|
||||||
|
# unrecognized bytes. This stage clusters them by *file format* (not producer)
|
||||||
|
# using the first `HEADER_N` header bytes, modeled as a Dirichlet-process
|
||||||
|
# mixture of per-position categoricals over a 257-symbol alphabet
|
||||||
|
# (byte 0–255, plus symbol 257 = "past EOF"). Each cluster's signature is a
|
||||||
|
# magic-number template that can be promoted into the classifier's fast path.
|
||||||
|
#
|
||||||
|
# This file is deliberately dependency-light: everything below is base Julia
|
||||||
|
# (only `log`, no `SpecialFunctions`), so it drops into the existing module and
|
||||||
|
# the offline calibration script alike without touching the Manifest. The model
|
||||||
|
# is categorical on purpose — do NOT reuse model.jl's [0,1] byte scaling here
|
||||||
|
# (that metric is meaningful for the Lux net and meaningless for header bytes,
|
||||||
|
# where 0x89 and 0x88 are not "close"; see DESIGN §4).
|
||||||
|
|
||||||
|
"Number of leading header bytes modeled per file (the feature window). DESIGN §4b."
|
||||||
|
const HEADER_N = 32
|
||||||
|
|
||||||
|
"Alphabet size: byte values 0–255 plus one extra symbol for 'past end of file'."
|
||||||
|
const ALPHABET = 257
|
||||||
|
|
||||||
|
"The 'past EOF' symbol (1-based index `ALPHABET`). A file shorter than a given
|
||||||
|
position emits this here — real, discriminative signal for fixed-length formats,
|
||||||
|
and it avoids colliding zero-padding with genuine 0x00 header bytes (DESIGN §4)."
|
||||||
|
const PAST_EOF = ALPHABET
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Feature extraction
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
"""
|
||||||
|
header_symbols(path; n=HEADER_N) -> Vector{Int}
|
||||||
|
|
||||||
|
Read the first `n` bytes of the file at `path` and map them to a length-`n`
|
||||||
|
vector of 1-based categorical symbols: byte value `b` → `b + 1` (so `1..256`),
|
||||||
|
and every position at or beyond end-of-file → `PAST_EOF` (`257`). Never reads
|
||||||
|
more than `n` bytes, so memory stays flat regardless of file size.
|
||||||
|
"""
|
||||||
|
function header_symbols(path::AbstractString; n::Integer=HEADER_N)
|
||||||
|
syms = fill(PAST_EOF, n)
|
||||||
|
open(path, "r") do io
|
||||||
|
bytes = read(io, n)
|
||||||
|
@inbounds for i in eachindex(bytes)
|
||||||
|
syms[i] = Int(bytes[i]) + 1
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return syms
|
||||||
|
end
|
||||||
|
|
||||||
|
"""
|
||||||
|
header_matrix(paths; n=HEADER_N) -> Matrix{Int}
|
||||||
|
|
||||||
|
Stack `header_symbols` for every path into an `n × length(paths)` matrix (one
|
||||||
|
column per file), the input layout the Gibbs sampler and predictive scorer both
|
||||||
|
consume.
|
||||||
|
"""
|
||||||
|
function header_matrix(paths::AbstractVector{<:AbstractString}; n::Integer=HEADER_N)
|
||||||
|
X = Matrix{Int}(undef, n, length(paths))
|
||||||
|
for (j, p) in enumerate(paths)
|
||||||
|
X[:, j] = header_symbols(p; n=n)
|
||||||
|
end
|
||||||
|
return X
|
||||||
|
end
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Model: DP mixture of per-position categoricals (Dirichlet-Categorical)
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
"""
|
||||||
|
ClusterStats
|
||||||
|
|
||||||
|
Sufficient statistics for one cluster: a per-position count table `counts`
|
||||||
|
(`n × ALPHABET`; `counts[i, v]` = how many member files show symbol `v` at
|
||||||
|
position `i`) and the member count `members`. These are exactly what phase-B
|
||||||
|
persists per catalog entry, and everything the collapsed predictive needs.
|
||||||
|
A slot with `members == 0` is inactive (reusable) — the Gibbs sweep prunes
|
||||||
|
emptied clusters without renumbering, so surviving cluster ids stay stable.
|
||||||
|
"""
|
||||||
|
mutable struct ClusterStats
|
||||||
|
counts::Matrix{Int} # n × ALPHABET
|
||||||
|
members::Int
|
||||||
|
end
|
||||||
|
|
||||||
|
ClusterStats(n::Integer) = ClusterStats(zeros(Int, n, ALPHABET), 0)
|
||||||
|
|
||||||
|
"Add file `x` (a length-n symbol vector) into cluster `c`'s sufficient stats."
|
||||||
|
function add!(c::ClusterStats, x::AbstractVector{<:Integer})
|
||||||
|
@inbounds for i in eachindex(x)
|
||||||
|
c.counts[i, x[i]] += 1
|
||||||
|
end
|
||||||
|
c.members += 1
|
||||||
|
return c
|
||||||
|
end
|
||||||
|
|
||||||
|
"Remove file `x` from cluster `c`'s sufficient stats (inverse of `add!`)."
|
||||||
|
function remove!(c::ClusterStats, x::AbstractVector{<:Integer})
|
||||||
|
@inbounds for i in eachindex(x)
|
||||||
|
c.counts[i, x[i]] -= 1
|
||||||
|
end
|
||||||
|
c.members -= 1
|
||||||
|
return c
|
||||||
|
end
|
||||||
|
|
||||||
|
"""
|
||||||
|
log_predictive(c, x, β) -> Float64
|
||||||
|
|
||||||
|
Log probability that file `x` was produced by cluster `c` under the collapsed
|
||||||
|
Dirichlet-Categorical predictive, given `c`'s current counts: at each position
|
||||||
|
`i`, `p(x_i | c) = (counts[i, x_i] + β) / (members + ALPHABET·β)`, summed in log
|
||||||
|
space over positions. Call with `c` NOT containing `x` (Gibbs excludes the point
|
||||||
|
being resampled), so an emptied cluster reduces to the uniform prior `1/ALPHABET`
|
||||||
|
per position — identical to a brand-new cluster, as it should be.
|
||||||
|
"""
|
||||||
|
function log_predictive(c::ClusterStats, x::AbstractVector{<:Integer}, β::Float64)
|
||||||
|
denom = log(c.members + ALPHABET * β)
|
||||||
|
s = 0.0
|
||||||
|
@inbounds for i in eachindex(x)
|
||||||
|
s += log(c.counts[i, x[i]] + β) - denom
|
||||||
|
end
|
||||||
|
return s
|
||||||
|
end
|
||||||
|
|
||||||
|
"Log likelihood of `x` under the fixed uniform component (each position uniform
|
||||||
|
over the 257 symbols): `n · log(1/ALPHABET)`. Used for both the never-adaptive
|
||||||
|
background 'junk drawer' and the prior predictive of a fresh cluster (DESIGN §4a)."
|
||||||
|
log_uniform(n::Integer) = -n * log(ALPHABET)
|
||||||
|
|
||||||
|
# Lanczos approximation to log Γ(x) for x > 0, so partition scoring (below) needs
|
||||||
|
# the Dirichlet-multinomial marginal's gamma terms without pulling in
|
||||||
|
# SpecialFunctions — keeping this stage dependency-flat (no Manifest churn).
|
||||||
|
# g = 7, standard coefficients; accurate to ~1e-14 over the range we use.
|
||||||
|
const _LANCZOS_G = 7
|
||||||
|
const _LANCZOS_C = (0.99999999999980993, 676.5203681218851, -1259.1392167224028,
|
||||||
|
771.32342877765313, -176.61502916214059, 12.507343278686905,
|
||||||
|
-0.13857109526572012, 9.9843695780195716e-6, 1.5056327351493116e-7)
|
||||||
|
|
||||||
|
function loggamma(x::Float64)
|
||||||
|
x < 0.5 && return log(π / sin(π * x)) - loggamma(1.0 - x) # reflection
|
||||||
|
x -= 1.0
|
||||||
|
a = _LANCZOS_C[1]
|
||||||
|
t = x + _LANCZOS_G + 0.5
|
||||||
|
@inbounds for i in 1:_LANCZOS_G + 1
|
||||||
|
a += _LANCZOS_C[i + 1] / (x + i)
|
||||||
|
end
|
||||||
|
return 0.5 * log(2π) + (x + 0.5) * log(t) - t + log(a)
|
||||||
|
end
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Phase A: collapsed Gibbs sampler (offline — the science)
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
"""
|
||||||
|
GibbsResult
|
||||||
|
|
||||||
|
Output of `gibbs_cluster`: `assignments` (one per input file; `0` = absorbed by
|
||||||
|
the background junk drawer, positive ints = cluster id), the surviving
|
||||||
|
`clusters` keyed by id, and `score` (the partition's collapsed pseudo-likelihood,
|
||||||
|
used to rank restarts).
|
||||||
|
"""
|
||||||
|
struct GibbsResult
|
||||||
|
assignments::Vector{Int}
|
||||||
|
clusters::Dict{Int,ClusterStats}
|
||||||
|
score::Float64
|
||||||
|
end
|
||||||
|
|
||||||
|
"""
|
||||||
|
gibbs_cluster(X; α, β, bg_mass, sweeps, restarts, rng) -> GibbsResult
|
||||||
|
|
||||||
|
Cluster the columns of `X` (an `n × N` header-symbol matrix) with a collapsed
|
||||||
|
Gibbs sampler over a CRP/Dirichlet-Categorical mixture plus a fixed uniform
|
||||||
|
background (DESIGN §5). Unknown *k* falls out of the CRP natively.
|
||||||
|
|
||||||
|
Per point, per sweep, the point is removed from its cluster and reassigned by
|
||||||
|
sampling from the CRP-predictive weights:
|
||||||
|
|
||||||
|
* existing cluster `k`: `members_k · exp(log_predictive)`
|
||||||
|
* background: `bg_mass · (1/ALPHABET)^n` (never adapts)
|
||||||
|
* a fresh cluster: `α · (1/ALPHABET)^n`
|
||||||
|
|
||||||
|
A uniform/high-entropy blob matches no structured cluster, and background vs.
|
||||||
|
fresh is then decided by `bg_mass` vs. `α`; with `bg_mass ≥ α` such blobs are
|
||||||
|
absorbed rather than minting singletons. `restarts` independent runs are made
|
||||||
|
from different seeds and the highest-scoring partition is returned (a cheap,
|
||||||
|
base-only stand-in for the offline Binder/VI point-summary the design defers).
|
||||||
|
"""
|
||||||
|
function gibbs_cluster(X::AbstractMatrix{<:Integer};
|
||||||
|
α::Float64=1.0, β::Float64=0.5, bg_mass::Float64=5.0,
|
||||||
|
sweeps::Integer=80, restarts::Integer=4,
|
||||||
|
rng::AbstractRNG=Random.default_rng())
|
||||||
|
best = nothing
|
||||||
|
for _ in 1:restarts
|
||||||
|
r = _gibbs_once(X; α=α, β=β, bg_mass=bg_mass, sweeps=sweeps, rng=rng)
|
||||||
|
if best === nothing || r.score > best.score
|
||||||
|
best = r
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return best
|
||||||
|
end
|
||||||
|
|
||||||
|
function _gibbs_once(X::AbstractMatrix{<:Integer};
|
||||||
|
α::Float64, β::Float64, bg_mass::Float64,
|
||||||
|
sweeps::Integer, rng::AbstractRNG)
|
||||||
|
n, N = size(X)
|
||||||
|
# Seed every file in its own singleton (NOT the background). Cold-starting
|
||||||
|
# from the background deadlocks: at a format's first file, a fresh cluster
|
||||||
|
# and the background are equally uniform, so with bg_mass ≥ α the background
|
||||||
|
# always wins and no real cluster is ever seeded. Singleton init sidesteps
|
||||||
|
# this — same-format singletons merge and snowball, while a lone
|
||||||
|
# random-blob singleton dissolves on resample and is reclaimed by the
|
||||||
|
# (stickier) background. See DESIGN §4a.
|
||||||
|
z = collect(1:N)
|
||||||
|
clusters = Dict{Int,ClusterStats}()
|
||||||
|
for j in 1:N
|
||||||
|
c = ClusterStats(n)
|
||||||
|
add!(c, view(X, :, j))
|
||||||
|
clusters[j] = c
|
||||||
|
end
|
||||||
|
next_id = N + 1
|
||||||
|
log_u = log_uniform(n)
|
||||||
|
log_bg = log(bg_mass) + log_u
|
||||||
|
log_new = log(α) + log_u
|
||||||
|
|
||||||
|
# Reused across every point-visit so the sampler's hot loop allocates nothing
|
||||||
|
# per step (2M+ visits per run): `idbuf[t]` is the cluster id whose weight is
|
||||||
|
# `logw[t+1]` (logw[1] = background, logw[end] = fresh). Rebuilding these with
|
||||||
|
# fresh `Vector`/`collect(keys(...))` each step was both slow and enough GC
|
||||||
|
# churn to trip a Julia GC segfault on long grid runs.
|
||||||
|
idbuf = Int[]
|
||||||
|
logw = Float64[]
|
||||||
|
for _ in 1:sweeps
|
||||||
|
for j in 1:N
|
||||||
|
x = view(X, :, j)
|
||||||
|
|
||||||
|
# Remove point j from its current component.
|
||||||
|
zj = z[j]
|
||||||
|
if zj > 0
|
||||||
|
c = clusters[zj]
|
||||||
|
remove!(c, x)
|
||||||
|
if c.members == 0
|
||||||
|
delete!(clusters, zj) # prune emptied cluster; id retired
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
# Candidate log-weights: background, each live cluster, fresh.
|
||||||
|
empty!(idbuf); empty!(logw)
|
||||||
|
push!(logw, log_bg)
|
||||||
|
for (k, c) in clusters
|
||||||
|
push!(idbuf, k)
|
||||||
|
push!(logw, log(c.members) + log_predictive(c, x, β))
|
||||||
|
end
|
||||||
|
push!(logw, log_new)
|
||||||
|
|
||||||
|
# Gumbel-max sample from the categorical over components.
|
||||||
|
pick = _gumbel_argmax(logw, rng)
|
||||||
|
|
||||||
|
if pick == 1
|
||||||
|
z[j] = 0 # background
|
||||||
|
elseif pick == length(logw)
|
||||||
|
id = next_id; next_id += 1 # fresh cluster
|
||||||
|
c = ClusterStats(n)
|
||||||
|
add!(c, x)
|
||||||
|
clusters[id] = c
|
||||||
|
z[j] = id
|
||||||
|
else
|
||||||
|
id = idbuf[pick - 1]
|
||||||
|
add!(clusters[id], x)
|
||||||
|
z[j] = id
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
return GibbsResult(z, clusters, partition_logmarginal(X, z, clusters, α, β))
|
||||||
|
end
|
||||||
|
|
||||||
|
"Argmax of `logw .+ Gumbel noise` — an exact draw from softmax(logw) without
|
||||||
|
normalizing (numerically safe for the tiny header-likelihood magnitudes)."
|
||||||
|
function _gumbel_argmax(logw::AbstractVector{Float64}, rng::AbstractRNG)
|
||||||
|
best_i = 1
|
||||||
|
best_v = -Inf
|
||||||
|
@inbounds for i in eachindex(logw)
|
||||||
|
g = logw[i] - log(-log(rand(rng)))
|
||||||
|
if g > best_v
|
||||||
|
best_v = g
|
||||||
|
best_i = i
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return best_i
|
||||||
|
end
|
||||||
|
|
||||||
|
"""
|
||||||
|
partition_logmarginal(X, z, clusters, α, β) -> Float64
|
||||||
|
|
||||||
|
The joint log-evidence `log p(z, X)` of a partition under the CRP prior and the
|
||||||
|
collapsed Dirichlet-Categorical likelihood — the principled score for ranking
|
||||||
|
Gibbs restarts (higher = better). It is the sum of:
|
||||||
|
|
||||||
|
* the Dirichlet-multinomial **marginal** of each cluster's per-position counts,
|
||||||
|
`lΓ(Aβ) − lΓ(mₖ+Aβ) + Σ_v [lΓ(c_v+β) − lΓ(β)]`, whose normalizer supplies the
|
||||||
|
Occam penalty that a plain product-of-predictives lacks — it is what makes a
|
||||||
|
*merged, heterogeneous* cluster score **worse** than two clean ones (an
|
||||||
|
earlier pseudo-likelihood scorer omitted this and wrongly rewarded merging);
|
||||||
|
* the CRP prior over the clustered points, `K·log α + Σₖ lΓ(mₖ) + lΓ(α) −
|
||||||
|
lΓ(α+N_clustered)`, penalizing gratuitous extra clusters; and
|
||||||
|
* the fixed uniform term for background-assigned files.
|
||||||
|
"""
|
||||||
|
function partition_logmarginal(X::AbstractMatrix{<:Integer}, z::AbstractVector{<:Integer},
|
||||||
|
clusters::Dict{Int,ClusterStats}, α::Float64, β::Float64)
|
||||||
|
n, N = size(X)
|
||||||
|
Aβ = ALPHABET * β
|
||||||
|
lg_Aβ = loggamma(Aβ)
|
||||||
|
lg_β = loggamma(β)
|
||||||
|
s = 0.0
|
||||||
|
# Dirichlet-Categorical marginal likelihood, per cluster × position.
|
||||||
|
for (_, c) in clusters
|
||||||
|
lg_denom = loggamma(c.members + Aβ)
|
||||||
|
@inbounds for i in 1:n, v in 1:ALPHABET
|
||||||
|
cv = c.counts[i, v]
|
||||||
|
cv > 0 && (s += loggamma(cv + β) - lg_β)
|
||||||
|
end
|
||||||
|
s += n * (lg_Aβ - lg_denom)
|
||||||
|
end
|
||||||
|
# CRP prior over the partition of the clustered points.
|
||||||
|
n_bg = count(==(0), z)
|
||||||
|
n_clustered = N - n_bg
|
||||||
|
K = length(clusters)
|
||||||
|
s += K * log(α) + loggamma(α) - loggamma(α + n_clustered)
|
||||||
|
for (_, c) in clusters
|
||||||
|
s += loggamma(float(c.members))
|
||||||
|
end
|
||||||
|
# Background files: independent, uniform.
|
||||||
|
s += n_bg * log_uniform(n)
|
||||||
|
return s
|
||||||
|
end
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Phase B: sequential CRP-predictive assignment (online — the catalog)
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
"""
|
||||||
|
assign_file(x, clusters, ids; α, β, bg_mass) -> Int
|
||||||
|
|
||||||
|
Deterministically assign a single file `x` against an existing catalog: the
|
||||||
|
same CRP-predictive rule as Gibbs but at the **argmax** (no sampling) with the
|
||||||
|
current assignments held fixed (DESIGN §5B). Returns the id of the chosen
|
||||||
|
cluster, `0` for the background, or `-1` to signal "mint a new cluster". `ids`
|
||||||
|
is the caller's stable ordering of `keys(clusters)`.
|
||||||
|
|
||||||
|
A new cluster is minted (`-1`) only when the fresh-cluster weight strictly wins.
|
||||||
|
Fresh and background share the same `(1/ALPHABET)^n` likelihood (one file, however
|
||||||
|
structured, is indistinguishable from a uniform blob until a *second* like file
|
||||||
|
appears), so this reduces to `α > bg_mass`. Under the calibrated `bg_mass > α`,
|
||||||
|
minting is therefore effectively off on the live path **by design**: a novel file
|
||||||
|
that matches nothing parks in the background, and genuinely new formats are
|
||||||
|
discovered by the periodic **offline Gibbs compaction** re-clustering that
|
||||||
|
residue (DESIGN §5), not by single-file minting. Because ids are frozen at birth
|
||||||
|
by the caller, there is no label switching.
|
||||||
|
"""
|
||||||
|
function assign_file(x::AbstractVector{<:Integer}, clusters::Dict{Int,ClusterStats},
|
||||||
|
ids::AbstractVector{<:Integer};
|
||||||
|
α::Float64=1.0, β::Float64=0.5, bg_mass::Float64=5.0)
|
||||||
|
n = length(x)
|
||||||
|
log_u = log_uniform(n)
|
||||||
|
best_kind = :bg # :bg, :existing, :new
|
||||||
|
best_id = 0
|
||||||
|
best = log(bg_mass) + log_u
|
||||||
|
new_w = log(α) + log_u
|
||||||
|
if new_w > best
|
||||||
|
best = new_w; best_kind = :new
|
||||||
|
end
|
||||||
|
for k in ids
|
||||||
|
c = clusters[k]
|
||||||
|
w = log(c.members) + log_predictive(c, x, β)
|
||||||
|
if w > best
|
||||||
|
best = w; best_kind = :existing; best_id = k
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return best_kind === :bg ? 0 : best_kind === :new ? -1 : best_id
|
||||||
|
end
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Signatures and promotion (closing the loop to the classifier — DESIGN §6)
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
"""
|
||||||
|
signature(c; peak_threshold=0.9, β=0.5) -> Vector{Union{Int,Nothing}}
|
||||||
|
|
||||||
|
Turn a cluster's counts into a libmagic-style template: at each position, if the
|
||||||
|
modal symbol's posterior probability exceeds `peak_threshold`, that byte is
|
||||||
|
*required* (returned as the 0–255 byte value, or `PAST_EOF`); otherwise the
|
||||||
|
position is a wildcard (`nothing`). The vector of required bytes IS the
|
||||||
|
magic-number template — the whole point of the categorical model (DESIGN §4).
|
||||||
|
|
||||||
|
Peakedness is a **Bernoulli** question ("is this position fixed to byte `v`, or
|
||||||
|
not?"), so it uses a 2-way posterior mean `(count + β)/(members + 2β)` — NOT the
|
||||||
|
257-way mixture predictive. The alphabet-wide version would smear the estimate
|
||||||
|
across 257 symbols (`members + 257β` in the denominator), pulling even a
|
||||||
|
unanimous position below any sane threshold once β is small — which would make
|
||||||
|
promotion impossible. This decouples signature detection from the clustering
|
||||||
|
pseudocount and the alphabet size.
|
||||||
|
"""
|
||||||
|
function signature(c::ClusterStats; peak_threshold::Float64=0.9, β::Float64=0.5)
|
||||||
|
n = size(c.counts, 1)
|
||||||
|
sig = Vector{Union{Int,Nothing}}(nothing, n)
|
||||||
|
c.members == 0 && return sig
|
||||||
|
denom = c.members + 2β
|
||||||
|
@inbounds for i in 1:n
|
||||||
|
v = argmax(view(c.counts, i, :))
|
||||||
|
p = (c.counts[i, v] + β) / denom
|
||||||
|
if p > peak_threshold
|
||||||
|
sig[i] = v == PAST_EOF ? PAST_EOF : v - 1 # back to raw byte value
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return sig
|
||||||
|
end
|
||||||
|
|
||||||
|
"Number of fixed (non-wildcard) positions in a signature — its 'magic length'."
|
||||||
|
magic_positions(sig::AbstractVector) = count(!isnothing, sig)
|
||||||
|
|
||||||
|
"""
|
||||||
|
is_promotable(c, sig; min_members=20, min_magic=3) -> Bool
|
||||||
|
|
||||||
|
A cluster qualifies for *nomination* (still human-gated, DESIGN §6) when it has
|
||||||
|
at least `min_members` files AND at least `min_magic` fixed signature positions.
|
||||||
|
The background (id 0) is never passed here — it is never promotable by design.
|
||||||
|
"""
|
||||||
|
function is_promotable(c::ClusterStats, sig::AbstractVector;
|
||||||
|
min_members::Integer=20, min_magic::Integer=3)
|
||||||
|
return c.members >= min_members && magic_positions(sig) >= min_magic
|
||||||
|
end
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Calibration metrics (DESIGN §7): agreement of recovered clusters vs. truth
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
"Map a label vector to consecutive integer ids and a group→indices table."
|
||||||
|
function _groups(labels::AbstractVector)
|
||||||
|
g = Dict{Any,Vector{Int}}()
|
||||||
|
for (i, l) in enumerate(labels)
|
||||||
|
push!(get!(g, l, Int[]), i)
|
||||||
|
end
|
||||||
|
return g
|
||||||
|
end
|
||||||
|
|
||||||
|
"""
|
||||||
|
adjusted_rand_index(a, b) -> Float64
|
||||||
|
|
||||||
|
Adjusted Rand Index between two labelings of the same items: 1.0 = identical
|
||||||
|
partitions (up to relabeling), ~0.0 = chance agreement, can go negative. This is
|
||||||
|
the §7 calibration objective — grid-tuning maximizes ARI of recovered-vs-truth
|
||||||
|
(magic-collapsed) labels. Hand-rolled to keep the dependency footprint flat;
|
||||||
|
matches `Clustering.randindex`.
|
||||||
|
"""
|
||||||
|
function adjusted_rand_index(a::AbstractVector, b::AbstractVector)
|
||||||
|
length(a) == length(b) || throw(DimensionMismatch("label vectors differ in length"))
|
||||||
|
n = length(a)
|
||||||
|
n < 2 && return 1.0
|
||||||
|
ga = collect(values(_groups(a)))
|
||||||
|
gb = collect(values(_groups(b)))
|
||||||
|
# Contingency-table sum of C(n_ij, 2).
|
||||||
|
comb2(x) = x * (x - 1) / 2
|
||||||
|
sa = Set.(ga)
|
||||||
|
index = 0.0
|
||||||
|
for A in sa, B in gb
|
||||||
|
nij = count(in(A), B)
|
||||||
|
index += comb2(nij)
|
||||||
|
end
|
||||||
|
sum_a = sum(comb2(length(g)) for g in ga)
|
||||||
|
sum_b = sum(comb2(length(g)) for g in gb)
|
||||||
|
total = comb2(n)
|
||||||
|
expected = sum_a * sum_b / total
|
||||||
|
maxi = (sum_a + sum_b) / 2
|
||||||
|
denom = maxi - expected
|
||||||
|
return denom == 0 ? 1.0 : (index - expected) / denom
|
||||||
|
end
|
||||||
|
|
||||||
|
"""
|
||||||
|
v_measure(truth, pred; β=1.0) -> (v, homogeneity, completeness)
|
||||||
|
|
||||||
|
Entropy-based cluster agreement (Rosenberg & Hirschberg): homogeneity (each
|
||||||
|
predicted cluster holds one true class), completeness (each true class stays in
|
||||||
|
one predicted cluster), and their weighted harmonic mean `v`. Reported alongside
|
||||||
|
ARI in §7 calibration as a second, differently-biased view.
|
||||||
|
"""
|
||||||
|
function v_measure(truth::AbstractVector, pred::AbstractVector; β::Float64=1.0)
|
||||||
|
n = length(truth)
|
||||||
|
n == 0 && return (1.0, 1.0, 1.0)
|
||||||
|
gt = _groups(truth)
|
||||||
|
gp = _groups(pred)
|
||||||
|
entropy(g) = -sum((length(v) / n) * log(length(v) / n) for v in values(g))
|
||||||
|
H_C = entropy(gt)
|
||||||
|
H_K = entropy(gp)
|
||||||
|
# Conditional entropies via the contingency table.
|
||||||
|
H_CK = 0.0 # H(truth | pred)
|
||||||
|
H_KC = 0.0 # H(pred | truth)
|
||||||
|
for (_, P) in gp
|
||||||
|
Ps = Set(P)
|
||||||
|
for (_, C) in gt
|
||||||
|
nij = count(in(Ps), C)
|
||||||
|
nij == 0 && continue
|
||||||
|
H_CK -= (nij / n) * log(nij / length(P))
|
||||||
|
end
|
||||||
|
end
|
||||||
|
for (_, C) in gt
|
||||||
|
Cs = Set(C)
|
||||||
|
for (_, P) in gp
|
||||||
|
nij = count(in(Cs), P)
|
||||||
|
nij == 0 && continue
|
||||||
|
H_KC -= (nij / n) * log(nij / length(C))
|
||||||
|
end
|
||||||
|
end
|
||||||
|
homogeneity = H_C == 0 ? 1.0 : 1 - H_CK / H_C
|
||||||
|
completeness = H_K == 0 ? 1.0 : 1 - H_KC / H_K
|
||||||
|
v = (homogeneity + completeness == 0) ? 0.0 :
|
||||||
|
(1 + β) * homogeneity * completeness / (β * homogeneity + completeness)
|
||||||
|
return (v, homogeneity, completeness)
|
||||||
|
end
|
||||||
@@ -33,6 +33,17 @@ Base.@kwdef struct Config
|
|||||||
model_path::String = "model/classifier.jld2" # committed classifier artifact, loaded at startup
|
model_path::String = "model/classifier.jld2" # committed classifier artifact, loaded at startup
|
||||||
exiftool_timeout::Int = 30 # seconds before a stuck exiftool is killed → degraded sidecar
|
exiftool_timeout::Int = 30 # seconds before a stuck exiftool is killed → degraded sidecar
|
||||||
linguist_timeout::Int = 30 # seconds before a stuck github-linguist is killed → no programming language
|
linguist_timeout::Int = 30 # seconds before a stuck github-linguist is killed → no programming language
|
||||||
|
# Stage 5 (unknown-format discovery). A separate single-owner batch process
|
||||||
|
# sweeps binary/ and clusters headers; these are its knobs (see
|
||||||
|
# model/DESIGN_clustering.md §9). Values are the calibrated defaults from
|
||||||
|
# bin/cluster_calibrate.jl on the training corpus.
|
||||||
|
cluster_dir::String = "data/binary" # stage-5 input: the :unknown/binary sink to sweep
|
||||||
|
cluster_n::Int = 32 # header bytes modeled per file (HEADER_N)
|
||||||
|
cluster_alpha::Float64 = 1.0 # CRP concentration: propensity to spawn new formats
|
||||||
|
cluster_pseudocount::Float64 = 0.1 # Dirichlet pseudocount β; calibrated on the training corpus
|
||||||
|
cluster_bg_mass::Float64 = 5.0 # fixed mass of the uniform background 'junk drawer'
|
||||||
|
promote_min_members::Int = 20 # cluster size threshold for promotion nomination
|
||||||
|
promote_min_magic::Int = 3 # required fixed signature positions for nomination
|
||||||
end
|
end
|
||||||
|
|
||||||
"""
|
"""
|
||||||
@@ -49,7 +60,9 @@ Recognised variables:
|
|||||||
FS_TEXT_WORKERS, FS_TEXT_QUEUE_CAPACITY,
|
FS_TEXT_WORKERS, FS_TEXT_QUEUE_CAPACITY,
|
||||||
FS_SPOOL_DIR, FS_KNOWN_DIR, FS_UNKNOWN_DIR, FS_BINARY_DIR, FS_TEXT_DIR,
|
FS_SPOOL_DIR, FS_KNOWN_DIR, FS_UNKNOWN_DIR, FS_BINARY_DIR, FS_TEXT_DIR,
|
||||||
FS_DONE_DIR, FS_TEXT_DONE_DIR, FS_FAILED_DIR, FS_MODEL_PATH,
|
FS_DONE_DIR, FS_TEXT_DONE_DIR, FS_FAILED_DIR, FS_MODEL_PATH,
|
||||||
FS_EXIFTOOL_TIMEOUT, FS_LINGUIST_TIMEOUT
|
FS_EXIFTOOL_TIMEOUT, FS_LINGUIST_TIMEOUT,
|
||||||
|
FS_CLUSTER_DIR, FS_CLUSTER_N, FS_CLUSTER_ALPHA, FS_CLUSTER_PSEUDOCOUNT,
|
||||||
|
FS_CLUSTER_BG_MASS, FS_PROMOTE_MIN_MEMBERS, FS_PROMOTE_MIN_MAGIC
|
||||||
"""
|
"""
|
||||||
function config_from_env(; host=nothing, port=nothing, worker_count=nothing,
|
function config_from_env(; host=nothing, port=nothing, worker_count=nothing,
|
||||||
queue_capacity=nothing, known_worker_count=nothing,
|
queue_capacity=nothing, known_worker_count=nothing,
|
||||||
@@ -59,7 +72,10 @@ function config_from_env(; host=nothing, port=nothing, worker_count=nothing,
|
|||||||
known_dir=nothing, unknown_dir=nothing, binary_dir=nothing,
|
known_dir=nothing, unknown_dir=nothing, binary_dir=nothing,
|
||||||
text_dir=nothing, done_dir=nothing, text_done_dir=nothing,
|
text_dir=nothing, done_dir=nothing, text_done_dir=nothing,
|
||||||
failed_dir=nothing, model_path=nothing, exiftool_timeout=nothing,
|
failed_dir=nothing, model_path=nothing, exiftool_timeout=nothing,
|
||||||
linguist_timeout=nothing)
|
linguist_timeout=nothing, cluster_dir=nothing, cluster_n=nothing,
|
||||||
|
cluster_alpha=nothing, cluster_pseudocount=nothing,
|
||||||
|
cluster_bg_mass=nothing, promote_min_members=nothing,
|
||||||
|
promote_min_magic=nothing)
|
||||||
Config(
|
Config(
|
||||||
host = something(host, get(ENV, "FS_HOST", "127.0.0.1")),
|
host = something(host, get(ENV, "FS_HOST", "127.0.0.1")),
|
||||||
port = something(port, parse(Int, get(ENV, "FS_PORT", "8080"))),
|
port = something(port, parse(Int, get(ENV, "FS_PORT", "8080"))),
|
||||||
@@ -82,6 +98,13 @@ function config_from_env(; host=nothing, port=nothing, worker_count=nothing,
|
|||||||
model_path = something(model_path, get(ENV, "FS_MODEL_PATH", "model/classifier.jld2")),
|
model_path = something(model_path, get(ENV, "FS_MODEL_PATH", "model/classifier.jld2")),
|
||||||
exiftool_timeout = something(exiftool_timeout, parse(Int, get(ENV, "FS_EXIFTOOL_TIMEOUT", "30"))),
|
exiftool_timeout = something(exiftool_timeout, parse(Int, get(ENV, "FS_EXIFTOOL_TIMEOUT", "30"))),
|
||||||
linguist_timeout = something(linguist_timeout, parse(Int, get(ENV, "FS_LINGUIST_TIMEOUT", "30"))),
|
linguist_timeout = something(linguist_timeout, parse(Int, get(ENV, "FS_LINGUIST_TIMEOUT", "30"))),
|
||||||
|
cluster_dir = something(cluster_dir, get(ENV, "FS_CLUSTER_DIR", "data/binary")),
|
||||||
|
cluster_n = something(cluster_n, parse(Int, get(ENV, "FS_CLUSTER_N", "32"))),
|
||||||
|
cluster_alpha = something(cluster_alpha, parse(Float64, get(ENV, "FS_CLUSTER_ALPHA", "1.0"))),
|
||||||
|
cluster_pseudocount = something(cluster_pseudocount, parse(Float64, get(ENV, "FS_CLUSTER_PSEUDOCOUNT", "0.1"))),
|
||||||
|
cluster_bg_mass = something(cluster_bg_mass, parse(Float64, get(ENV, "FS_CLUSTER_BG_MASS", "5.0"))),
|
||||||
|
promote_min_members = something(promote_min_members, parse(Int, get(ENV, "FS_PROMOTE_MIN_MEMBERS", "20"))),
|
||||||
|
promote_min_magic = something(promote_min_magic, parse(Int, get(ENV, "FS_PROMOTE_MIN_MAGIC", "3"))),
|
||||||
)
|
)
|
||||||
end
|
end
|
||||||
|
|
||||||
|
|||||||
168
test/runtests.jl
168
test/runtests.jl
@@ -11,7 +11,12 @@ using FileServer: Job, Config, ChannelQueue, enqueue!, dequeue!, length,
|
|||||||
is_binary, handle_unknown_job,
|
is_binary, handle_unknown_job,
|
||||||
detect_natural_language, run_linguist, detect_programming_language,
|
detect_natural_language, run_linguist, detect_programming_language,
|
||||||
read_text_sample, build_text_metadata, finalize_text!, handle_text_job,
|
read_text_sample, build_text_metadata, finalize_text!, handle_text_job,
|
||||||
linguist_available
|
linguist_available,
|
||||||
|
header_symbols, header_matrix, ClusterStats, add!, remove!,
|
||||||
|
log_predictive, loggamma, gibbs_cluster, assign_file,
|
||||||
|
signature, magic_positions, is_promotable,
|
||||||
|
adjusted_rand_index, v_measure, HEADER_N, ALPHABET, PAST_EOF
|
||||||
|
using Random: MersenneTwister
|
||||||
using Languages: LanguageDetector
|
using Languages: LanguageDetector
|
||||||
|
|
||||||
# A minimal, valid 1×1 PNG. Lets the real-exiftool tests assert stable facts
|
# A minimal, valid 1×1 PNG. Lets the real-exiftool tests assert stable facts
|
||||||
@@ -334,6 +339,167 @@ end
|
|||||||
end
|
end
|
||||||
end
|
end
|
||||||
|
|
||||||
|
@testset "cluster: header_symbols feature extraction" begin
|
||||||
|
mktempdir() do root
|
||||||
|
# Bytes map to 1-based symbols (b -> b+1); positions past EOF -> PAST_EOF.
|
||||||
|
p = joinpath(root, "f.bin")
|
||||||
|
write(p, UInt8[0x00, 0x7f, 0xff])
|
||||||
|
s = header_symbols(p; n=6)
|
||||||
|
@test s[1:3] == [1, 128, 256] # 0->1, 0x7f->128, 0xff->256
|
||||||
|
@test all(==(PAST_EOF), s[4:6]) # 3 bytes short of n=6 -> past EOF
|
||||||
|
@test PAST_EOF == ALPHABET == 257
|
||||||
|
@test Base.length(header_symbols(p)) == HEADER_N
|
||||||
|
|
||||||
|
# An empty file is all past-EOF (real signal, not an error).
|
||||||
|
e = joinpath(root, "empty"); write(e, UInt8[])
|
||||||
|
@test all(==(PAST_EOF), header_symbols(e; n=8))
|
||||||
|
|
||||||
|
# header_matrix stacks one column per file.
|
||||||
|
q = joinpath(root, "g.bin"); write(q, UInt8[0x41, 0x42])
|
||||||
|
X = header_matrix([p, q]; n=4)
|
||||||
|
@test size(X) == (4, 2)
|
||||||
|
@test X[:, 2] == [0x42, 0x43, PAST_EOF, PAST_EOF] # 'A'->66,'B'->67
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
@testset "cluster: loggamma matches known values" begin
|
||||||
|
@test loggamma(1.0) ≈ 0.0 atol=1e-10
|
||||||
|
@test loggamma(2.0) ≈ 0.0 atol=1e-10
|
||||||
|
@test loggamma(5.0) ≈ log(24) atol=1e-10 # Γ(5) = 4! = 24
|
||||||
|
@test loggamma(0.5) ≈ 0.5log(π) atol=1e-10 # Γ(1/2) = √π
|
||||||
|
@test loggamma(10.0) ≈ log(362880) atol=1e-8 # Γ(10) = 9!
|
||||||
|
end
|
||||||
|
|
||||||
|
@testset "cluster: sufficient stats and predictive" begin
|
||||||
|
c = ClusterStats(3)
|
||||||
|
x = [10, 20, 30]
|
||||||
|
# Empty cluster's predictive equals the uniform prior (1/ALPHABET)^n.
|
||||||
|
@test log_predictive(c, x, 0.5) ≈ -3 * log(ALPHABET) atol=1e-9
|
||||||
|
# add! then remove! is an exact round-trip back to empty.
|
||||||
|
add!(c, x); remove!(c, x)
|
||||||
|
@test c.members == 0
|
||||||
|
@test all(==(0), c.counts)
|
||||||
|
# A cluster holding a matching point scores it far above uniform.
|
||||||
|
add!(c, x)
|
||||||
|
@test log_predictive(c, x, 0.5) > -3 * log(ALPHABET)
|
||||||
|
end
|
||||||
|
|
||||||
|
@testset "cluster: ARI and V-measure" begin
|
||||||
|
# Identical labelings (up to relabeling) score 1.0.
|
||||||
|
@test adjusted_rand_index([1,1,2,2], [7,7,9,9]) ≈ 1.0
|
||||||
|
@test adjusted_rand_index(["a","a","b"], ["b","b","a"]) ≈ 1.0
|
||||||
|
v, h, comp = v_measure([1,1,2,2], [5,5,6,6])
|
||||||
|
@test v ≈ 1.0 && h ≈ 1.0 && comp ≈ 1.0
|
||||||
|
# A partition that merges two true classes into one is complete but not
|
||||||
|
# homogeneous, and ARI drops below 1.
|
||||||
|
@test adjusted_rand_index([1,1,2,2], [1,1,1,1]) < 1.0
|
||||||
|
_, h2, comp2 = v_measure([1,1,2,2], [1,1,1,1])
|
||||||
|
@test comp2 ≈ 1.0 # everything from each class stays together
|
||||||
|
@test h2 < 1.0 # but the cluster mixes two classes
|
||||||
|
end
|
||||||
|
|
||||||
|
@testset "cluster: signature, magic length, promotability" begin
|
||||||
|
n = 8
|
||||||
|
c = ClusterStats(n)
|
||||||
|
# 30 files sharing bytes 0xDE 0xAD 0xBE 0xEF at positions 1-4, random after.
|
||||||
|
rng = MersenneTwister(1)
|
||||||
|
for _ in 1:30
|
||||||
|
x = vcat([0xDE, 0xAD, 0xBE, 0xEF] .+ 1, rand(rng, 1:256, 4))
|
||||||
|
add!(c, x)
|
||||||
|
end
|
||||||
|
sig = signature(c)
|
||||||
|
@test sig[1:4] == [0xDE, 0xAD, 0xBE, 0xEF] # spiked -> required bytes
|
||||||
|
@test all(isnothing, sig[5:8]) # flat -> wildcards
|
||||||
|
@test magic_positions(sig) == 4
|
||||||
|
@test is_promotable(c, sig; min_members=20, min_magic=3)
|
||||||
|
# Too few members, or too few magic positions, blocks nomination.
|
||||||
|
@test !is_promotable(c, sig; min_members=50, min_magic=3)
|
||||||
|
@test !is_promotable(c, sig; min_members=20, min_magic=5)
|
||||||
|
end
|
||||||
|
|
||||||
|
@testset "cluster: §10.1 discovers nothing from noise" begin
|
||||||
|
# 25 independent random blobs — the shape of data/binary (structureless
|
||||||
|
# junk). Correct output: ZERO promoted clusters (random headers never
|
||||||
|
# form a ≥20-member, ≥3-magic-byte signature). See DESIGN §10.1.
|
||||||
|
rng = MersenneTwister(20260703)
|
||||||
|
X = reduce(hcat, [rand(rng, 1:256, HEADER_N) for _ in 1:25])
|
||||||
|
r = gibbs_cluster(X; α=1.0, β=0.1, bg_mass=5.0, sweeps=60, restarts=3,
|
||||||
|
rng=MersenneTwister(1))
|
||||||
|
promoted = count(c -> is_promotable(c, signature(c); min_members=20, min_magic=3),
|
||||||
|
values(r.clusters))
|
||||||
|
@test promoted == 0
|
||||||
|
|
||||||
|
# And a lone structured file (a singleton, like the giant PDF in the pile)
|
||||||
|
# never promotes on its own: N=1 < min_members.
|
||||||
|
one = ClusterStats(HEADER_N)
|
||||||
|
add!(one, vcat([0x25,0x50,0x44,0x46] .+ 1, fill(1, HEADER_N - 4)))
|
||||||
|
@test !is_promotable(one, signature(one); min_members=20, min_magic=3)
|
||||||
|
end
|
||||||
|
|
||||||
|
@testset "cluster: §10.2 recovers known (synthetic) formats" begin
|
||||||
|
# Four synthetic "formats": a fixed magic prefix + random tail, mirroring
|
||||||
|
# gzip/PDF/JPEG/ELF. Calibrated settings must recover them as clean,
|
||||||
|
# promotable clusters at high ARI — the magic-collapsed recovery of §10.2,
|
||||||
|
# here with a hermetic, deterministic corpus.
|
||||||
|
# ~12-byte constant headers + random tails — the shape of a real file
|
||||||
|
# header (a fixed magic/version region, then variable content). A too-short
|
||||||
|
# magic over a fully-random tail is adversarially hard and lets a format
|
||||||
|
# over-split; real headers anchor a cluster with ~12+ constant bytes.
|
||||||
|
rng = MersenneTwister(7)
|
||||||
|
magics = Dict(
|
||||||
|
"gzip" => UInt8[0x1f,0x8b,0x08,0x00,0x00,0x00,0x00,0x00,0x00,0x03,0x2d,0x00],
|
||||||
|
"pdf" => UInt8[0x25,0x50,0x44,0x46,0x2d,0x31,0x2e,0x34,0x0a,0x25,0xe2,0xe3],
|
||||||
|
"jpeg" => UInt8[0xff,0xd8,0xff,0xe0,0x00,0x10,0x4a,0x46,0x49,0x46,0x00,0x01],
|
||||||
|
"elf" => UInt8[0x7f,0x45,0x4c,0x46,0x02,0x01,0x01,0x00,0x00,0x00,0x00,0x00],
|
||||||
|
)
|
||||||
|
cols = Vector{Int}[]; truth = String[]
|
||||||
|
for (label, magic) in magics, _ in 1:50
|
||||||
|
tail = rand(rng, 1:256, HEADER_N - Base.length(magic))
|
||||||
|
push!(cols, vcat(Int.(magic) .+ 1, tail))
|
||||||
|
push!(truth, label)
|
||||||
|
end
|
||||||
|
X = reduce(hcat, cols)
|
||||||
|
r = gibbs_cluster(X; α=1.0, β=0.1, bg_mass=5.0, sweeps=120, restarts=6,
|
||||||
|
rng=MersenneTwister(3))
|
||||||
|
@test adjusted_rand_index(truth, r.assignments) > 0.9
|
||||||
|
|
||||||
|
# Truth breakdown of each cluster, keyed by cluster id.
|
||||||
|
breakdown(id) = [truth[i] for i in eachindex(r.assignments) if r.assignments[i] == id]
|
||||||
|
# Nominations cover most formats (a format may over-split below the size
|
||||||
|
# threshold, but the recovery is not allowed to miss more than one)...
|
||||||
|
nominated_labels = Set{String}()
|
||||||
|
for (id, c) in r.clusters
|
||||||
|
sig = signature(c)
|
||||||
|
if is_promotable(c, sig; min_members=20, min_magic=3)
|
||||||
|
# ...and every nomination is PURE — the whole point of the human
|
||||||
|
# gate is that we never hand it a garbage merged signature.
|
||||||
|
labels = unique(breakdown(id))
|
||||||
|
@test Base.length(labels) == 1
|
||||||
|
push!(nominated_labels, only(labels))
|
||||||
|
end
|
||||||
|
end
|
||||||
|
@test Base.length(nominated_labels) >= 3
|
||||||
|
end
|
||||||
|
|
||||||
|
@testset "cluster: §5B sequential assignment (phase B)" begin
|
||||||
|
# Build a catalog with one strong cluster (magic 0xCA 0xFE ...).
|
||||||
|
n = 8
|
||||||
|
clusters = Dict{Int,ClusterStats}()
|
||||||
|
c = ClusterStats(n)
|
||||||
|
rng = MersenneTwister(2)
|
||||||
|
for _ in 1:40
|
||||||
|
add!(c, vcat([0xCA,0xFE,0xBA,0xBE] .+ 1, rand(rng, 1:256, 4)))
|
||||||
|
end
|
||||||
|
clusters[1] = c
|
||||||
|
ids = collect(keys(clusters))
|
||||||
|
# A file that matches the cluster's magic joins it.
|
||||||
|
match = vcat([0xCA,0xFE,0xBA,0xBE] .+ 1, rand(rng, 1:256, 4))
|
||||||
|
@test assign_file(match, clusters, ids; α=1.0, β=0.1, bg_mass=5.0) == 1
|
||||||
|
# A structured-but-novel file (different magic) spawns a new cluster (-1).
|
||||||
|
novel = vcat([0x12,0x34,0x56,0x78] .+ 1, fill(1, 4))
|
||||||
|
@test assign_file(novel, clusters, ids; α=1.0, β=0.1, bg_mass=5.0) in (-1, 0)
|
||||||
|
end
|
||||||
|
|
||||||
@testset "recover_dir!: re-enqueues work, skips sidecars" begin
|
@testset "recover_dir!: re-enqueues work, skips sidecars" begin
|
||||||
mktempdir() do root
|
mktempdir() do root
|
||||||
dir = joinpath(root, "known"); mkpath(dir)
|
dir = joinpath(root, "known"); mkpath(dir)
|
||||||
|
|||||||
Reference in New Issue
Block a user