using Test using FileServer using JSON3 # Pull internals into scope. These aren't exported (only `run` is), but the # whole risk profile of this pipeline lives in these functions, so we test them # directly rather than only through the HTTP surface. using FileServer: Job, Config, ChannelQueue, enqueue!, dequeue!, length, sanitize_filename, recover_dir!, normalize_metadata, build_metadata, finalize_known!, run_exiftool, is_binary, handle_unknown_job, detect_natural_language, run_linguist, detect_programming_language, read_text_sample, build_text_metadata, finalize_text!, handle_text_job, linguist_available using Languages: LanguageDetector # A minimal, valid 1×1 PNG. Lets the real-exiftool tests assert stable facts # (FileType == "PNG", 1×1 dimensions) that don't drift across exiftool versions. const PNG_1x1 = UInt8[137,80,78,71,13,10,26,10,0,0,0,13,73,72,68,82,0,0,0,1,0, 0,0,1,8,6,0,0,0,31,21,196,137,0,0,0,11,73,68,65,84,120,218,99,100,248,255, 191,30,0,5,132,2,127,194,91,30,42,0,0,0,0,73,69,78,68,174,66,96,130] "Build a Config whose data dirs all live under a fresh temp directory." function tmp_config(root; kwargs...) cfg = Config(; spool_dir = joinpath(root, "spool"), known_dir = joinpath(root, "known"), unknown_dir = joinpath(root, "unknown"), binary_dir = joinpath(root, "binary"), text_dir = joinpath(root, "text"), done_dir = joinpath(root, "done"), text_done_dir = joinpath(root, "text_done"), failed_dir = joinpath(root, "failed"), kwargs..., ) FileServer.ensure_dirs(cfg) return cfg end @testset "FileServer" begin @testset "sanitize_filename" begin @test sanitize_filename("report.pdf") == "report.pdf" # Directory components and traversal are stripped, not preserved. @test sanitize_filename("../../etc/passwd") == "passwd" @test sanitize_filename("/abs/path/x.txt") == "x.txt" # Leading dots removed so "..", ".hidden" can't sneak through. @test sanitize_filename("..") == "unnamed" @test sanitize_filename(".hidden") == "hidden" # Unsafe chars collapse to underscores; empty falls back to "unnamed". @test sanitize_filename("a b&c*.d") == "a_b_c_.d" @test sanitize_filename("") == "unnamed" # Length is capped. @test Base.length(sanitize_filename("a"^500)) == FileServer.MAX_NAME_LEN end @testset "normalize_metadata" begin job = Job("id-1", "photo.jpg", "/data/known/id-1-photo.jpg", 4242, 0.0) # Group-prefixed tags as exiftool -G emits them are already group-stripped # by run_exiftool before reaching normalize_metadata, so keys are bare. bytag = Dict{String,Any}( "FileType" => "JPEG", "MIMEType" => "image/jpeg", "ImageWidth" => 800, "ImageHeight"=> 600, "Author" => "Ada Lovelace", "Creator" => "Acrobat", # feeds created_by, not author "CreateDate" => "2020:01:02 03:04:05", "ModifyDate" => "2020:01:02 03:04:06", "PageCount" => 12, ) m = normalize_metadata(job, bytag) @test m.file_type == "JPEG" @test m.mime_type == "image/jpeg" @test m.dimensions == (width = 800, height = 600) @test m.author == "Ada Lovelace" @test m.created_by == "Acrobat" @test m.created_date == "2020:01:02 03:04:05" @test m.page_count == 12 @test m.error === nothing @test m.raw === bytag # file_size is authoritative from the Job, never from exiftool. @test m.file_size == 4242 end @testset "normalize_metadata: missing tags degrade to nothing" begin job = Job("id-2", "blob.bin", "/data/known/id-2-blob.bin", 7, 0.0) m = normalize_metadata(job, Dict{String,Any}()) @test m.file_type === nothing @test m.dimensions === nothing # neither width nor height present @test m.author === nothing @test m.file_size == 7 @test m.error === nothing # empty-but-present dict is still "success" end @testset "build_metadata: degraded on extraction failure" begin mktempdir() do root cfg = tmp_config(root; exiftool_timeout=5) # Point at a nonexistent file → exiftool exits non-zero → degraded. job = Job("id-3", "gone.dat", joinpath(cfg.known_dir, "id-3-gone.dat"), 99, 0.0) m = build_metadata(job, cfg) @test m.error !== nothing @test m.file_type === nothing @test m.raw === nothing @test m.file_size == 99 # still authoritative from the Job @test m.id == "id-3" end end @testset "run_exiftool: real extraction on a PNG" begin mktempdir() do root p = joinpath(root, "pixel.png") write(p, PNG_1x1) bytag = run_exiftool(p, 30) @test bytag !== nothing @test bytag["FileType"] == "PNG" @test bytag["ImageWidth"] == 1 @test bytag["ImageHeight"] == 1 end end @testset "finalize_known!: sidecar-first commit, end to end" begin mktempdir() do root cfg = tmp_config(root) # A real known-stage file to enrich. src = joinpath(cfg.known_dir, "id-9-pixel.png") write(src, PNG_1x1) job = Job("id-9", "pixel.png", src, Base.length(PNG_1x1), 0.0) meta = build_metadata(job, cfg) file_dest, sidecar = finalize_known!(cfg, job, meta) # File moved into done/, original gone from known/. @test isfile(file_dest) @test dirname(file_dest) == cfg.done_dir @test !isfile(src) # Sidecar committed alongside it, valid JSON, no leftover .tmp. @test isfile(sidecar) @test endswith(sidecar, ".meta.json") @test !isfile(string(sidecar, ".tmp")) parsed = JSON3.read(read(sidecar, String)) @test parsed.file_type == "PNG" @test parsed.file_size == Base.length(PNG_1x1) @test parsed.error === nothing end end @testset "is_binary: UTF-8 sniff" begin mktempdir() do root # Plain ASCII text → text. txt = joinpath(root, "notes.txt") write(txt, "hello, world\nsecond line\n") @test is_binary(txt) == false # Non-ASCII UTF-8 (accents, CJK, emoji) is valid text — the whole # point of moving off the printable-ASCII/NUL heuristic. uni = joinpath(root, "unicode.txt") write(uni, "café — 日本語 — 🚀\n") @test is_binary(uni) == false # ANSI-colored log: ESC + other text control bytes are text-safe. ansi = joinpath(root, "colored.log") write(ansi, "\e[31merror\e[0m: tab\there\r\nnext\n") @test is_binary(ansi) == false # A NUL byte anywhere in the sniff window → binary (it's a control # byte outside the text-safe set, even though it's valid UTF-8). bin = joinpath(root, "blob.dat") write(bin, UInt8[0x01, 0x02, 0x00, 0x03]) @test is_binary(bin) == true # A non-NUL, non-text control byte (e.g. 0x07 BEL) → binary. ctrl = joinpath(root, "ctrl.dat") write(ctrl, UInt8[UInt8('h'), UInt8('i'), 0x07]) @test is_binary(ctrl) == true # Malformed UTF-8 (lone continuation / bad lead byte) → binary. bad = joinpath(root, "bad.dat") write(bad, UInt8[UInt8('a'), 0xff, 0xfe, 0xc3, 0x28]) @test is_binary(bad) == true # A multi-byte char split by the sniff boundary must NOT read as # binary: pad to one byte short of the window, then a 2-byte 'é' # (0xc3 0xa9) so only its lead byte lands inside the window. split = joinpath(root, "split.txt") write(split, vcat(fill(UInt8('a'), FileServer.CONTENT_SNIFF_BYTES - 1), UInt8[0xc3, 0xa9])) @test is_binary(split) == false # Empty file → treated as text. empty = joinpath(root, "empty") write(empty, UInt8[]) @test is_binary(empty) == false # Binary garbage past the sniff window is not seen → still text. far = joinpath(root, "far.txt") write(far, vcat(fill(UInt8('a'), FileServer.CONTENT_SNIFF_BYTES), UInt8[0x00])) @test is_binary(far) == false end end @testset "handle_unknown_job: binary terminal, text routed to stage 4" begin mktempdir() do root cfg = tmp_config(root) text_queue = ChannelQueue(10) # A binary file (embedded NUL) lands in binary/ and is NOT enqueued. bpath = joinpath(cfg.unknown_dir, "id-b-blob.dat") write(bpath, UInt8[0x00, 0xFF, 0x10]) bjob = Job("id-b", "blob.dat", bpath, filesize(bpath), 0.0) handle_unknown_job(bjob, cfg, 1, text_queue) @test isfile(joinpath(cfg.binary_dir, "id-b-blob.dat")) @test !isfile(bpath) @test length(text_queue) == 0 # A text file lands in text/ AND is routed onto the stage-4 queue, # with its path updated to the new text/ location. tpath = joinpath(cfg.unknown_dir, "id-t-notes.log") write(tpath, "just some log text\n") tjob = Job("id-t", "notes.log", tpath, filesize(tpath), 0.0) handle_unknown_job(tjob, cfg, 1, text_queue) moved = joinpath(cfg.text_dir, "id-t-notes.log") @test isfile(moved) @test !isfile(tpath) @test length(text_queue) == 1 routed = dequeue!(text_queue) @test routed.id == "id-t" @test routed.path == moved end end @testset "detect_natural_language" begin d = LanguageDetector() name, code, conf = detect_natural_language(d, "The quick brown fox jumps over the lazy dog and then runs away quickly today.") @test name == "English" @test code == "eng" @test conf isa Real && 0.0 <= conf <= 1.0 # Empty / whitespace-only text yields no result rather than throwing # (the detector itself errors on empty input). @test detect_natural_language(d, "") == (nothing, nothing, nothing) @test detect_natural_language(d, " \n\t ") == (nothing, nothing, nothing) end @testset "read_text_sample: bounded, UTF-8 safe" begin mktempdir() do root p = joinpath(root, "notes.txt") write(p, "café — 日本語 — hello\n") @test read_text_sample(p) == "café — 日本語 — hello\n" # Reads at most LANG_SAMPLE_BYTES, and doesn't choke on a multi-byte # char straddling that boundary (trailing 'é' half-in the window). big = joinpath(root, "big.txt") write(big, vcat(fill(UInt8('a'), FileServer.LANG_SAMPLE_BYTES - 1), UInt8[0xc3, 0xa9])) # 'é' split by the edge s = read_text_sample(big) @test Base.length(s) == FileServer.LANG_SAMPLE_BYTES - 1 # trailing half-char trimmed @test all(==('a'), s) end end @testset "run_linguist: real detection on source vs. prose" begin if !linguist_available() @info "github-linguist not on PATH; skipping run_linguist tests" else mktempdir() do root # A Python source file → linguist names the language. py = joinpath(root, "script.py") write(py, "import sys\ndef main():\n print('hi')\nmain()\n") @test run_linguist(py, 30) == "Python" # Plain prose reports as "Text", which collapses to nothing. prose = joinpath(root, "notes.txt") write(prose, "The quarterly report shows steady growth this year.\n") @test run_linguist(prose, 30) === nothing end end end @testset "build_text_metadata + finalize_text!: end to end" begin mktempdir() do root cfg = tmp_config(root) d = LanguageDetector() src = joinpath(cfg.text_dir, "id-x-script.py") write(src, join(["# a short program in English prose comment", "import sys", "def greet(name):", " print('hello ' + name + ' welcome to the show today')", "greet('world')", ""], "\n")) job = Job("id-x", "script.py", src, filesize(src), 0.0) meta = build_text_metadata(d, job, cfg) @test meta.id == "id-x" @test meta.content_type == "text" @test meta.file_size == filesize(src) @test meta.language !== nothing # some natural language detected @test meta.error === nothing # programming_language is best-effort; present only when linguist is. if linguist_available() @test meta.programming_language == "Python" end file_dest, sidecar = finalize_text!(cfg, job, meta) # File moved into text_done/, original gone from text/. @test isfile(file_dest) @test dirname(file_dest) == cfg.text_done_dir @test !isfile(src) # Sidecar committed alongside it, valid JSON, no leftover .tmp. @test isfile(sidecar) @test endswith(sidecar, ".meta.json") @test !isfile(string(sidecar, ".tmp")) parsed = JSON3.read(read(sidecar, String)) @test parsed.content_type == "text" @test parsed.file_size == filesize(file_dest) end end @testset "handle_text_job: enriches and commits to text_done/" begin mktempdir() do root cfg = tmp_config(root) d = LanguageDetector() src = joinpath(cfg.text_dir, "id-h-readme.md") write(src, "# Project\n\nThis project does something useful and interesting for everyone.\n") job = Job("id-h", "readme.md", src, filesize(src), 0.0) handle_text_job(job, cfg, 1, d) @test isfile(joinpath(cfg.text_done_dir, "id-h-readme.md")) @test isfile(joinpath(cfg.text_done_dir, "id-h-readme.md.meta.json")) @test !isfile(src) end end @testset "recover_dir!: re-enqueues work, skips sidecars" begin mktempdir() do root dir = joinpath(root, "known"); mkpath(dir) uuid = "0123456789abcdef0123456789abcdef0123" # 36 chars work = joinpath(dir, string(uuid, "-report.pdf")) write(work, "x") write(joinpath(dir, string(uuid, "-report.pdf.meta.json")), "{}") # sidecar write(joinpath(dir, "shortname"), "y") # no uuid prefix q = ChannelQueue(10) n = recover_dir!(dir, q) @test n == 2 # the two real files, not the sidecar @test length(q) == 2 jobs = [dequeue!(q), dequeue!(q)] # sorted by filename on recovery # "0123...-report.pdf" sorts before "shortname". @test jobs[1].id == uuid @test jobs[1].original_name == "report.pdf" @test jobs[1].path == work # File with no uuid prefix keeps its whole name; gets a minted id. @test jobs[2].original_name == "shortname" @test !isempty(jobs[2].id) end end end