From 9f5ede9e5b36a120cde7f1582e0c96c8ac823ada Mon Sep 17 00:00:00 2001 From: Divyam Talwar Date: Sun, 20 Sep 2026 03:18:48 +0530 Subject: [PATCH] fix(fetch): validate remote metadata before resumed publication Address #23 with focused regression coverage. AI-assisted implementation and isolated source review; exact validation and remaining platform limitations are recorded in the draft PR. Signed-off-by: Divyam Talwar --- QUICKSTART.md | 14 + quantprobe/fetch.py | 196 +++++++++- tests/smoke.py | 22 ++ tests/test_fetch_remote_metadata.py | 554 ++++++++++++++++++++++++++++ 4 files changed, 771 insertions(+), 15 deletions(-) create mode 100644 tests/test_fetch_remote_metadata.py diff --git a/QUICKSTART.md b/QUICKSTART.md index 69b20a4..24654b8 100644 --- a/QUICKSTART.md +++ b/QUICKSTART.md @@ -166,6 +166,20 @@ quantprobe run --gguf ./models/Qwen3-30B-A3B-Q2_K.gguf quantprobe bench --gguf ./models/Qwen3-30B-A3B-Q2_K.gguf ``` +**What `fetch` does when Hugging Face won't say how big the file is.** Completeness here is a +byte count against the remote's `Content-Length`, and that number is only read off a *successful* +`HEAD` — an error page carries a length too, of its own body, and it is never compared against +your file in either direction. So the two cases part ways. A **new download** refuses to start +(`CANNOT FETCH`, non-zero exit) and writes nothing: it would have no size to certify itself +against. A **file already on disk** is still reused, exactly as before, but it is reported as +`already present, NOT VERIFIED` rather than "complete" — presence and a matching name, nothing +more. That keeps `quantprobe auto` working offline against a model you already have, at the price +of saying plainly that it was not checked: `fetch` does no checksum, ETag or GGUF-header +validation, so even a size that *does* match is a length agreeing with a length, not proof of the +contents. Re-run with the remote reachable to have the size confirmed, or `--force` to download it +again. A same-named file of a *different* size, confirmed against a good `HEAD`, is still refused +outright — that one is a real disagreement. + ### Make your own compressed model The one-command version — picks a requantizable source from the repo, fetches the eval corpus, diff --git a/quantprobe/fetch.py b/quantprobe/fetch.py index 6d310b1..c85f78e 100644 --- a/quantprobe/fetch.py +++ b/quantprobe/fetch.py @@ -1,11 +1,16 @@ -"""hf_fetch.py -- robust multi-file HF downloader (manual HTTP Range, retry-on-break), bypassing the -hf CLI's Xet-backend stalls. Usage: python -m weights.hf_fetch [file2 ...] +"""quantprobe fetch -- robust multi-file HF downloader (manual HTTP Range, retry-on-break), +bypassing the hf CLI's Xet-backend stalls. Grown from weights/hf_fetch.py, whose usage line this +docstring still carried: the entry points are `quantprobe fetch [file ...]` and +`python -m quantprobe.fetch [file2 ...]`. Resumes partial .part files; token from HF_TOKEN env or ~/.cache/huggingface/token. +Completeness is a byte count against the remote's Content-Length and nothing else - see `fetch` +for what that does and does not certify. """ from __future__ import annotations import os +import re import sys import time @@ -30,21 +35,116 @@ def token(): return open(p).read().strip() if os.path.exists(p) else None +_CONTENT_RANGE = re.compile(r"^\s*bytes\s+(\d+)-(\d+)/(\d+)\s*$", re.IGNORECASE) + + +def remote_size(url, hdr, timeout=60): + """The remote length from a HEAD, or `(None, why)` when the answer cannot be trusted. + + An error response carries a Content-Length too - of its own JSON or HTML body. Reading that + header without looking at the status is how a 404 hands back a number that may happen to + match a half-written local file, or to differ from a perfectly good one. Status first, then + the header, and anything missing, unparseable or non-positive is a refusal rather than a + silent 0. + + `(None, why)` means "no size", not "wrong size", and the two callers treat it differently: + a NEW download has nothing but the byte count to certify itself with and refuses to start, + while an already-present file keeps its long-standing reuse and is reported as present + rather than verified. Neither path may compare a local length against a number that came + off an error page. + """ + try: + r = requests.head(url, headers=hdr, allow_redirects=True, timeout=timeout) + except requests.exceptions.RequestException as e: + return None, f"HEAD request failed ({str(e)[:60]})" + if r.status_code != 200: + return None, f"HEAD returned status {r.status_code}" + raw = r.headers.get("Content-Length") + if raw is None: + return None, "the response carried no Content-Length" + try: + n = int(str(raw).strip()) + except ValueError: + return None, f"unparseable Content-Length {raw!r}" + if n <= 0: + return None, "Content-Length must be positive for a model download" + return n, None + + +def check_content_range(value, want_start, want_total): + """Bytes a 206 is allowed to deliver, or `(None, why)` if its range is not the one we asked for. + + Completion below is `size == total`, so bytes appended at an offset nobody checked make a + file of exactly the right length and the wrong contents - which then gets promoted over the + real one. Everything the server claims is therefore compared against what was requested + before a byte is written. A server may legitimately answer an open-ended range with less + than the remainder, so a short-but-consistent range is accepted and the loop asks again from + the new offset. + """ + if not value: + return None, "206 carried no Content-Range" + m = _CONTENT_RANGE.match(value) + if not m: + return None, f"unparseable Content-Range {value!r}" + try: + start, end, total = (int(g) for g in m.groups()) + except ValueError: + return None, "Content-Range numbers are not representable" + if total != want_total: + return None, f"remote is now {total:,} B, was {want_total:,} B - the file changed" + if start != want_start: + return None, f"range starts at {start:,}, we asked from {want_start:,}" + if end < start or end >= total: + return None, f"range {start:,}-{end:,} does not fit a {total:,} B file" + return end - start + 1, None + + def fetch(repo, dest, fname, tok, tries=100, force=False): + """Download `fname` into `dest`, resuming a `.part`; True when `dest/fname` is usable. + + The only completeness test here is a byte count against the remote's own Content-Length, + and the two paths ask for it with different stakes: + + NEW DOWNLOAD - the byte count is the whole gate, so an untrusted or non-positive size is + refused before anything is written or deleted (including before `--force` removes a + published file to make room). + ALREADY PRESENT - an unforced skip keeps the file. When the remote answers properly the + size is compared, and a mismatch is still refused (U-18). When it does not answer, the + file is reused and reported as PRESENT, not complete: this is a name-and-presence match + and nothing more. It is NOT a content check - no checksum, ETag or GGUF-header validation + is performed anywhere in this module, and a matching length is not an authentication + either. + """ url = f"https://huggingface.co/{repo}/resolve/main/{fname}" out = os.path.join(dest, fname) part = out + ".part" hdr0 = {"Authorization": f"Bearer {tok}"} if tok else {} if os.path.exists(out) and not force: # name-only skip once handed an incompatible file to llama-speculative (a June-era GGUF - # under the target name): compare against the remote before declaring completeness. + # under the target name): ask the remote first, and call this complete only when a good + # answer says the two sizes agree. U-18 - the mismatch refusal below - is the other half, + # and it needs a real length for the same reason completion does. have = os.path.getsize(out) - try: - r = requests.head(url, headers=hdr0, allow_redirects=True, timeout=60) - remote = int(r.headers.get("Content-Length", 0)) - except requests.exceptions.RequestException: - remote = 0 - if remote and have != remote: + remote, why = remote_size(url, hdr0) + if remote is None: + # No trusted size, so there is nothing to compare and nothing is compared - the + # error-page length that used to decide this is not consulted in either direction. + # The file is REUSED, as it always has been when the remote is unobtainable: + # `auto` hands this path a model the user already downloaded, and refusing it would + # end the command for anyone offline with a perfectly good file. What changes is + # only what is claimed. "already complete"/"size matches remote" asserted a check + # that did not happen; this says presence, names why the size is missing, and + # leaves the bytes alone (a `.part` beside it is not promoted - nothing is renamed + # on this path). + print( + f" {fname}: already present, NOT VERIFIED - {why}. {have:,} B are on disk and " + f"were reused on name and presence alone; this is not a check of the contents. " + f"Re-run when the remote answers to have the size confirmed, or --force to " + f"download it again.", + flush=True, + ) + return True + if have != remote: print( f" {fname}: EXISTING FILE IS NOT THIS FILE - local {have:,} B vs remote " f"{remote:,} B. A same-named file from another source is on disk; it may be an " @@ -53,20 +153,32 @@ def fetch(repo, dest, fname, tok, tries=100, force=False): flush=True, ) return False - note = "size matches remote" if remote else "remote size unavailable, name+presence only" - print(f" {fname}: already complete ({note}; --force re-downloads)", flush=True) + print( + f" {fname}: already complete (size matches remote; --force re-downloads)", flush=True + ) return True + # Asked BEFORE the --force deletion below: that deletion exists to make room for a download, + # so finding out afterwards that the download cannot start costs the user the file for + # nothing. Nothing on disk has been touched at this point. + total, why = remote_size(url, hdr0) + if not total: + print( + f" {fname}: CANNOT FETCH - {why or 'the remote reports a 0 B file'}. There is no " + f"completed {fname} here to fall back on, and completion for a new download is a " + f"byte count, so a download against a size we do not have could never be " + f"certified; nothing was written or removed.", + flush=True, + ) + return False if os.path.exists(out) and force: os.remove(out) if os.path.exists(part): os.remove(part) - r = requests.head(url, headers=hdr0, allow_redirects=True, timeout=60) - total = int(r.headers.get("Content-Length", 0)) print(f" {fname}: {total / 1e9:.2f} GB", flush=True) t = 0 while t < tries: have = os.path.getsize(part) if os.path.exists(part) else 0 - if total and have >= total: + if have >= total: break try: h = dict(hdr0) @@ -78,12 +190,30 @@ def fetch(repo, dest, fname, tok, tries=100, force=False): time.sleep(5) t += 1 continue + if r.status_code == 206: + expect, why = check_content_range(r.headers.get("Content-Range"), have, total) + if expect is None: + # Rejected before the file is opened, so the prefix already on disk is + # untouched: it is this response that is wrong, not those bytes. + r.close() + print(f" bad range - {why}, retry {t + 1}", flush=True) + time.sleep(5) + t += 1 + continue + else: + expect = total # 200 to a Range request is the whole file: restart, not append mode = "ab" if (have and r.status_code == 206) else "wb" t0 = last = time.time() base = have if mode == "ab" else 0 + got = 0 + oversized = False with open(part, mode) as f: for chunk in r.iter_content(1 << 22): if chunk: + got += len(chunk) + if got > expect: + oversized = True + break f.write(chunk) if time.time() - last > 20: sz = os.path.getsize(part) @@ -92,22 +222,58 @@ def fetch(repo, dest, fname, tok, tries=100, force=False): flush=True, ) last = time.time() + if oversized or got != expect: + # A body that ENDED CLEANLY at the wrong length is the server contradicting + # itself - a truncated CDN object, or an error page under a 200 - not a dropped + # connection. Those arrive as exceptions below and keep their progress, because + # what they delivered was real bytes at the right offset. This is not, so the + # file goes back to the length it had before the request. + r.close() + os.truncate(part, base) + print( + f" {got:,} B for a {expect:,} B range - rolled back to {base:,}, " + f"retry {t + 1}", + flush=True, + ) + time.sleep(3) + t += 1 + continue except ( requests.exceptions.ChunkedEncodingError, requests.exceptions.ConnectionError, requests.exceptions.ReadTimeout, requests.exceptions.Timeout, ) as e: + # Interrupted, not contradicted: keep the progress and resume from it next round. print( f" break at {os.path.getsize(part) if os.path.exists(part) else 0:,}, retry {t + 1}: {str(e)[:60]}", flush=True, ) time.sleep(3) t += 1 - if total and os.path.exists(part) and os.path.getsize(part) == total: + size = os.path.getsize(part) if os.path.exists(part) else 0 + if size == total: os.replace(part, out) print(f" {fname}: DONE", flush=True) return True + if size > total: + # `have >= total` ends the loop before any request, so this is where an already-oversized + # `.part` lands - and "INCOMPLETE" was the one thing it is not. It cannot be a prefix of + # this file, so no retry can shrink it and it is never promoted; say which file is in the + # way and what to do with it. Not --force: that only clears a `.part` when a completed + # file sits beside it, which is exactly the case that does not apply here. + note = ( + "" + if os.path.exists(out) + else f" `--force` does not clear it while no completed {fname} sits next to it." + ) + print( + f" {fname}: PARTIAL FILE IS OVERSIZED - {part} holds {size:,} B, more than the " + f"{total:,} B the remote reports for {fname}, so it cannot be a prefix of this " + f"file and nothing was promoted. Move or delete that .part file, then re-run." + note, + flush=True, + ) + return False print(f" {fname}: INCOMPLETE", flush=True) return False diff --git a/tests/smoke.py b/tests/smoke.py index fce0397..aca63ee 100644 --- a/tests/smoke.py +++ b/tests/smoke.py @@ -1766,6 +1766,10 @@ def t_fetch_force_and_collision(): import os, tempfile from quantprobe import fetch as fmod class _R: + # A real requests.Response always carries a status; a mock with only `headers` is + # indistinguishable from an error page, and fetch no longer reads a length off one. + # The two assertions below are unchanged - see tests/test_fetch_remote_metadata.py. + status_code = 200 headers = {"Content-Length": "1000"} real_head = fmod.requests.head fmod.requests.head = lambda *a, **k: _R() @@ -1782,6 +1786,24 @@ class _R: assert rc == 0 and "--force" in out +def t_fetch_remote_metadata_and_range_integrity(): + """An error page's Content-Length decides nothing, and a 206 proves its offset before it lands. + + Both halves of the same defect - a byte count standing in for "this is the right file". The + skip path no longer reads a length off an unsuccessful HEAD in either direction (it neither + certifies nor accuses); an already-present file is still REUSED when the size is + unobtainable, but reported as present, not complete. The resume path validates + Content-Range before opening the file, and counts what it wrote. + + The cases live in tests/test_fetch_remote_metadata.py as unittest (they need a per-case + temp dir and a fake transport), and include the `auto` caller regression that pays for the + skip path; this runs every one of them inside the pytest-free suite too, so + `python tests/smoke.py` stays the single gate CONTRIBUTING.md points contributors at. + """ + from tests.test_fetch_remote_metadata import run_smoke + return run_smoke() + + def t_c11_depth_aware_dense_split(): # C-11 (prereg #66): the dense split must budget for the desktop reserve + compute buffer and # shrink its GPU layer count as context deepens - the old flat vc*0.9 emitted a 16k config diff --git a/tests/test_fetch_remote_metadata.py b/tests/test_fetch_remote_metadata.py new file mode 100644 index 0000000..215ed02 --- /dev/null +++ b/tests/test_fetch_remote_metadata.py @@ -0,0 +1,554 @@ +"""`quantprobe fetch` must not publish bytes it did not check, or rename what it did not measure. + +Two halves of one defect, because both end at the same place - a size comparison standing in for +"this is the right file": + + * the skip path read Content-Length off a HEAD without looking at the status, so an ERROR + PAGE's body length decided, by byte count, whether the file on disk was announced complete + or denounced as "NOT THIS FILE" (and the user sent to `--force`, which deletes it); + * the resume path appended a 206's body without reading its Content-Range, so bytes delivered + at the wrong offset landed on top of a good prefix and produced a file of exactly the right + length and the wrong contents, which the final size check then promoted. + +What is deliberately NOT changed: an already-present file whose size cannot be confirmed is +still reused. That is the module's long-standing offline behaviour and `auto` depends on it - +it is just no longer allowed to call itself "complete" or claim the size "matches". Presence is +reported as presence. There is no content authentication here, and these tests pin the wording +as much as the return value, because the wording is the whole difference. + +Every fixture is finite: the fake transport answers an exhausted queue with a retryable 503, and +`tries` is small, so a regression cannot turn these into a hang. Nothing touches the network, a +model file, or a real inference run. +""" + +from __future__ import annotations + +import io +import os +import shutil +import sys +import tempfile +import time +import unittest +from contextlib import redirect_stdout +from unittest import mock + +import requests + +from quantprobe import fetch as fmod + +REPO, NAME = "org/repo", "model.gguf" +BODY = b"0123456789" # 10 B "model"; short enough that a wrong offset is visible by eye + + +class FakeResponse: + """The part of ``requests.Response`` that `fetch` touches: status, headers, body, close. + + ``iter_content`` ignores the caller's chunk size deliberately. `fetch` asks for 4 MiB and + these bodies are ten bytes, so a fixed small step is the only way to model a connection + that breaks PART WAY through a response. + """ + + def __init__(self, status_code=200, headers=None, body=b"", break_after=None, step=3): + self.status_code = status_code + self.headers = requests.structures.CaseInsensitiveDict(headers or {}) + self.body = body + self.break_after = break_after + self.step = step + self.closed = False + + def iter_content(self, chunk_size=1): + sent = 0 + while sent < len(self.body): + if self.break_after is not None and sent >= self.break_after: + raise requests.exceptions.ChunkedEncodingError("connection broken") + chunk = self.body[sent : sent + self.step] + sent += len(chunk) + yield chunk + + def close(self): + self.closed = True + + +class FakeRequests: + """Stands in for the ``requests`` module inside quantprobe.fetch.""" + + exceptions = requests.exceptions + + def __init__(self, head, gets=()): + self._head = head + self._gets = list(gets) + self.head_calls = 0 + self.get_calls = 0 + self.ranges = [] + + def head(self, url, headers=None, allow_redirects=False, timeout=None): + self.head_calls += 1 + if isinstance(self._head, Exception): + raise self._head + return self._head + + def get(self, url, headers=None, stream=False, timeout=None, allow_redirects=False): + self.get_calls += 1 + self.ranges.append((headers or {}).get("Range")) + if self._gets: + return self._gets.pop(0) + # Queue exhausted: a retryable status, so the bound on the loop stays `tries` and no + # fixture can spin. + return FakeResponse(status_code=503) + + +class FakeClock: + """Real ``time()`` (the progress printer does arithmetic on it), recorded ``sleep()``.""" + + def __init__(self): + self.slept = [] + + def time(self): + return time.time() + + def sleep(self, seconds): + self.slept.append(seconds) + + +class FetchCase(unittest.TestCase): + def setUp(self): + self.dir = tempfile.mkdtemp(prefix="qp-fetch-") + self.addCleanup(shutil.rmtree, self.dir, ignore_errors=True) + self.out = os.path.join(self.dir, NAME) + self.part = self.out + ".part" + real_requests, real_time = fmod.requests, fmod.time + self.clock = FakeClock() + fmod.time = self.clock + + def restore(): + fmod.requests, fmod.time = real_requests, real_time + + self.addCleanup(restore) + + def net(self, head, gets=()): + self.fake = FakeRequests(head, gets) + fmod.requests = self.fake + return self.fake + + def run_fetch(self, tries=2, force=False): + buf = io.StringIO() + with redirect_stdout(buf): + ok = fmod.fetch(REPO, self.dir, NAME, None, tries=tries, force=force) + self.printed = buf.getvalue() + return ok + + def write(self, path, data): + with open(path, "wb") as f: + f.write(data) + + def read(self, path): + with open(path, "rb") as f: + return f.read() + + def assert_reported_as_present_not_verified(self): + """The reuse is allowed to say presence and nothing more.""" + low = self.printed.lower() + self.assertIn("already present", low) + self.assertIn("not verified", low) + self.assertNotIn("already complete", low) + self.assertNotIn("size matches", low) + + +class TestUnverifiableMetadataNeverDecidesByByteCount(FetchCase): + """An error page's Content-Length must not certify a file, and must not condemn one either. + + The refusal these cases pin is narrow and specific: a length that did not come from a + successful response is not a length at all, so it is neither compared nor believed. What + happens next for an already-present file is the unchanged offline behaviour - reuse it, and + say plainly that this is presence, not verification. + """ + + def test_error_response_length_is_neither_completion_nor_an_accusation(self): + # The sharp case: a 404's JSON body is 27 B and so is the half-written file on disk. + # Either verdict reached through that number is reached through an error page. + self.write(self.out, b"x" * 27) + fake = self.net(FakeResponse(404, {"Content-Length": "27"}, b"x" * 27)) + self.assertIs(self.run_fetch(), True) + self.assert_reported_as_present_not_verified() + self.assertNotIn("NOT THIS FILE", self.printed) + self.assertEqual(fake.get_calls, 0, "an unverified skip must not open a download") + self.assertEqual(self.read(self.out), b"x" * 27, "the local file must be left alone") + + def test_non_200_head_is_not_reported_as_a_foreign_file(self): + # Accusing the user's file of being "not this file" on the strength of an error page's + # Content-Length is a wrong diagnosis that sends them to --force, which deletes it. + self.write(self.out, BODY) + self.net(FakeResponse(403, {"Content-Length": "12"}, b"forbidden!!!")) + self.assertIs(self.run_fetch(), True) + self.assertNotIn("NOT THIS FILE", self.printed) + self.assert_reported_as_present_not_verified() + + def test_missing_content_length_is_presence_only(self): + self.write(self.out, BODY) + self.net(FakeResponse(200, {})) + self.assertIs(self.run_fetch(), True) + self.assert_reported_as_present_not_verified() + + def test_malformed_content_length_is_refused_not_raised(self): + # ValueError is not a RequestException: unguarded, "1,024" left fetch() as a traceback. + self.write(self.out, BODY) + self.net(FakeResponse(200, {"Content-Length": "not-a-number"})) + self.assertIs(self.run_fetch(), True) + self.assert_reported_as_present_not_verified() + + def test_negative_content_length_is_refused(self): + self.write(self.out, BODY) + self.net(FakeResponse(200, {"Content-Length": "-1"})) + self.assertIs(self.run_fetch(), True) + self.assertNotIn("NOT THIS FILE", self.printed) + self.assert_reported_as_present_not_verified() + + def test_zero_content_length_is_refused(self): + # A 0 B "model" is not a size this module can complete against, so it is not one it + # will quote either - but the file on disk keeps its long-standing reuse. + self.write(self.out, BODY) + self.net(FakeResponse(200, {"Content-Length": "0"})) + self.assertIs(self.run_fetch(), True) + self.assert_reported_as_present_not_verified() + + def test_failed_head_keeps_the_cached_file_usable(self): + # The offline case, and the reason the fallback survives: a user with the right file and + # no network must still be able to use it. + self.write(self.out, BODY) + fake = self.net(requests.exceptions.ConnectionError("no route to host")) + self.assertIs(self.run_fetch(), True) + self.assert_reported_as_present_not_verified() + self.assertEqual(fake.get_calls, 0) + self.assertEqual(self.read(self.out), BODY, "an unreachable remote must not cost the file") + + def test_unverified_reuse_never_promotes_a_partial(self): + # Presence of `out` is what is being reused. A `.part` next to it is someone else's + # interrupted download and must stay exactly where it is, under its own name. + self.write(self.out, BODY) + self.write(self.part, b"ZZZZZZZZZZZZ") + self.net(requests.exceptions.ConnectionError("no route to host")) + self.assertIs(self.run_fetch(), True) + self.assert_reported_as_present_not_verified() + self.assertEqual(self.read(self.out), BODY) + self.assertEqual(self.read(self.part), b"ZZZZZZZZZZZZ", "the partial must not be renamed") + + def test_matching_size_on_a_good_head_is_completion(self): + self.write(self.out, BODY) + self.net(FakeResponse(200, {"Content-Length": str(len(BODY))})) + self.assertIs(self.run_fetch(), True) + self.assertIn("already complete", self.printed) + self.assertIn("size matches remote", self.printed) + + def test_size_mismatch_on_a_good_head_still_refuses(self): + # U-18 stays intact: a same-named file of a different size is not this file. + self.write(self.out, BODY[:9]) + self.net(FakeResponse(200, {"Content-Length": str(len(BODY))})) + self.assertIs(self.run_fetch(), False) + self.assertIn("NOT THIS FILE", self.printed) + + def test_untrusted_head_does_not_start_a_download(self): + # Nothing is published here yet, so there is no presence to fall back on. Completion for + # a NEW download IS the byte count, so a download against a bogus total could never be + # certified - and a `.part` sized to an error page must never be promoted. + fake = self.net(FakeResponse(500, {"Content-Length": "31"}, b"x" * 31)) + self.assertIs(self.run_fetch(), False) + self.assertEqual(fake.get_calls, 0, "no bytes should be requested against a bogus size") + self.assertFalse(os.path.exists(self.part)) + self.assertFalse(os.path.exists(self.out)) + + +class TestRangeIntegrity(FetchCase): + """What a 206 claims is checked against what we asked for, before anything is written.""" + + def head_ok(self, gets): + return self.net(FakeResponse(200, {"Content-Length": str(len(BODY))}), gets) + + def test_plain_200_download_completes(self): + self.head_ok([FakeResponse(200, {"Content-Length": "10"}, BODY)]) + self.assertIs(self.run_fetch(), True) + self.assertEqual(self.read(self.out), BODY) + + def test_valid_206_resume_appends_exactly(self): + self.write(self.part, BODY[:4]) + fake = self.head_ok([FakeResponse(206, {"Content-Range": "bytes 4-9/10"}, BODY[4:])]) + self.assertIs(self.run_fetch(), True) + self.assertEqual(self.read(self.out), BODY) + self.assertEqual(fake.ranges[0], "bytes=4-") + + def test_short_but_consistent_206_keeps_its_progress_and_asks_again(self): + # A server may answer an open-ended range with LESS than the remainder. That is legal, + # the bytes are at the offset we asked for, and rolling them back would make some CDNs + # unresumable. It must be accepted and the loop must re-ask from the new offset. + self.write(self.part, BODY[:4]) + fake = self.head_ok( + [ + FakeResponse(206, {"Content-Range": "bytes 4-6/10"}, BODY[4:7]), + FakeResponse(206, {"Content-Range": "bytes 7-9/10"}, BODY[7:]), + ] + ) + self.assertIs(self.run_fetch(tries=3), True) + self.assertEqual(self.read(self.out), BODY) + self.assertEqual(fake.ranges, ["bytes=4-", "bytes=7-"], "must resume from 7, not 4") + self.assertNotIn("rolled back", self.printed, "a legal short range is not a rollback") + + def test_206_at_the_wrong_offset_is_rejected_and_the_prefix_survives(self): + # Server restarts from 0 while we hold 4 B. Appending gives 10 B - the exact size the + # completion check wants - of a file that is four bytes of nonsense at the front. + self.write(self.part, BODY[:4]) + self.head_ok([FakeResponse(206, {"Content-Range": "bytes 0-5/10"}, BODY[:6])]) + self.assertIs(self.run_fetch(), False) + self.assertFalse(os.path.exists(self.out), "a mis-offset resume must never be promoted") + self.assertEqual(self.read(self.part), BODY[:4], "the good prefix must be kept as-is") + + def test_206_for_a_different_total_is_rejected(self): + # The upload changed under us: the remainder is from a file we never measured. + self.write(self.part, BODY[:4]) + self.head_ok([FakeResponse(206, {"Content-Range": "bytes 4-9/99"}, BODY[4:])]) + self.assertIs(self.run_fetch(), False) + self.assertFalse(os.path.exists(self.out)) + self.assertEqual(self.read(self.part), BODY[:4]) + + def test_206_ending_past_the_end_of_the_file_is_rejected(self): + # `bytes 4-10/10` addresses an eleventh byte of a ten-byte file. The range is + # self-contradictory, so nothing it carries is trustworthy - and believing `end` would + # size the write past the total the completion check uses. + self.write(self.part, BODY[:4]) + self.head_ok([FakeResponse(206, {"Content-Range": "bytes 4-10/10"}, BODY[4:] + b"X")]) + self.assertIs(self.run_fetch(), False) + self.assertFalse(os.path.exists(self.out)) + self.assertEqual(self.read(self.part), BODY[:4]) + + def test_206_ending_before_its_own_start_is_rejected(self): + # `bytes 4-3/10` is a negative-length range: it would make the allowed payload -1 B. + self.write(self.part, BODY[:4]) + self.head_ok([FakeResponse(206, {"Content-Range": "bytes 4-3/10"}, BODY[4:])]) + self.assertIs(self.run_fetch(), False) + self.assertFalse(os.path.exists(self.out)) + self.assertEqual(self.read(self.part), BODY[:4]) + + def test_206_without_a_content_range_is_rejected(self): + self.write(self.part, BODY[:4]) + self.head_ok([FakeResponse(206, {}, BODY[4:])]) + self.assertIs(self.run_fetch(), False) + self.assertEqual(self.read(self.part), BODY[:4]) + + def test_206_longer_than_its_declared_range_rolls_back(self): + self.write(self.part, BODY[:4]) + self.head_ok([FakeResponse(206, {"Content-Range": "bytes 4-9/10"}, BODY[4:] + b"XXX")]) + self.assertIs(self.run_fetch(), False) + self.assertEqual(self.read(self.part), BODY[:4], "excess must be rolled back, not kept") + + def test_206_shorter_than_its_declared_range_rolls_back(self): + # Ended cleanly at the wrong length: the server contradicting itself, not a broken + # connection, so the bytes are suspect and the known-good prefix is restored. + self.write(self.part, BODY[:4]) + self.head_ok([FakeResponse(206, {"Content-Range": "bytes 4-9/10"}, BODY[4:7])]) + self.assertIs(self.run_fetch(), False) + self.assertEqual(self.read(self.part), BODY[:4]) + + def test_interrupted_206_keeps_its_progress_and_resumes(self): + # The opposite case, and the reason rollback is scoped to clean responses: a transfer + # that dies mid-body delivered real bytes at the right offset. Keep them. + self.write(self.part, BODY[:4]) + fake = self.head_ok( + [ + FakeResponse(206, {"Content-Range": "bytes 4-9/10"}, BODY[4:], break_after=3), + FakeResponse(206, {"Content-Range": "bytes 7-9/10"}, BODY[7:]), + ] + ) + self.assertIs(self.run_fetch(tries=3), True) + self.assertEqual(self.read(self.out), BODY) + self.assertEqual(fake.ranges, ["bytes=4-", "bytes=7-"], "must resume from 7, not 4") + + def test_200_answer_to_a_range_request_restarts_instead_of_appending(self): + # A server that ignores Range sends the whole file; treating it as a continuation + # doubles the prefix. + self.write(self.part, b"ZZZZ") + self.head_ok([FakeResponse(200, {"Content-Length": "10"}, BODY)]) + self.assertIs(self.run_fetch(), True) + self.assertEqual(self.read(self.out), BODY) + + +class TestResponseBoundaries(FetchCase): + def test_excess_bytes_before_transport_failure_cannot_be_published(self): + class ExcessThenFailure(FakeResponse): + def iter_content(self, chunk_size=1): + yield BODY[4:] + raise requests.exceptions.ReadTimeout("after excess bytes") + + self.write(self.part, BODY[:4]) + response = ExcessThenFailure(206, {"Content-Range": "bytes 4-5/10"}) + self.net(FakeResponse(200, {"Content-Length": "10"}), [response]) + self.assertFalse(self.run_fetch(tries=1)) + self.assertFalse(os.path.exists(self.out)) + self.assertEqual(self.read(self.part), BODY[:4]) + + def test_unrepresentable_range_numbers_reject_without_escaping(self): + self.write(self.part, BODY[:4]) + response = FakeResponse(206, {"Content-Range": "bytes 4-" + "9" * 5000 + "/10"}) + self.net(FakeResponse(200, {"Content-Length": "10"}), [response]) + self.assertFalse(self.run_fetch(tries=1)) + self.assertEqual(self.read(self.part), BODY[:4]) + + def test_oversized_partial_is_named_as_such_and_never_promoted(self): + # `have >= total` ends the loop with no request at all, so this is fully deterministic. + # The old report was a bare "INCOMPLETE", which is the one thing a 15 B partial of a + # 10 B file is not - and it named no file and no way out. `--force` is no help either: + # it only clears a `.part` when a completed file sits beside it, and none does here. + self.write(self.part, BODY + b"XXXXX") + fake = self.net(FakeResponse(200, {"Content-Length": "10"})) + self.assertIs(self.run_fetch(tries=2), False) + self.assertEqual(fake.get_calls, 0, "an oversized partial needs no request to diagnose") + self.assertFalse(os.path.exists(self.out), "nothing may be published from it") + self.assertEqual(self.read(self.part), BODY + b"XXXXX", "its bytes are the user's") + low = self.printed.lower() + self.assertIn("oversized", low) + self.assertNotIn("incomplete", low, "15 B of a 10 B file is not an incomplete download") + self.assertIn(self.part, self.printed, "the message must name THIS partial, by path") + self.assertIn("move or delete", low) + self.assertIn("`--force` does not clear it", self.printed) + + +class TestMetadataReviewControls(FetchCase): + def test_force_rejects_untrusted_head_before_deleting_published_file(self): + # Ordering matters more than the verdict: --force deletes `out` to make room for the + # download, so the size must be known to be fetchable BEFORE the file is gone. + self.write(self.out, BODY) + self.net(FakeResponse(404, {"Content-Length": "17"})) + self.assertFalse(self.run_fetch(tries=1, force=True)) + self.assertEqual(self.read(self.out), BODY) + + def test_range_unit_and_header_names_are_case_insensitive(self): + self.write(self.part, BODY[:4]) + self.net( + FakeResponse(200, {"cOnTeNt-LeNgTh": "10"}), + [FakeResponse(206, {"content-range": "ByTeS 4-9/10"}, BODY[4:])], + ) + self.assertTrue(self.run_fetch(tries=1)) + self.assertEqual(self.read(self.out), BODY) + + def test_error_head_cannot_promote_matching_partial(self): + # The complement of the presence fallback: there is no published file to reuse, only a + # `.part` whose length happens to equal an error page's. It stays a `.part`. + self.write(self.part, BODY) + self.net(FakeResponse(500, {"Content-Length": "10"})) + self.assertFalse(self.run_fetch(tries=1)) + self.assertFalse(os.path.exists(self.out)) + self.assertEqual(self.read(self.part), BODY) + + +class TestUpstreamSmokeGuardStillHolds(unittest.TestCase): + """The one pre-existing test this work touched, run here so the edit is checkable. + + `tests/smoke.py:t_fetch_force_and_collision` (U-18) mocks a HEAD response with nothing but + `headers`, which no real ``requests.Response`` ever is. Reading a length off that object + means reading one off something indistinguishable from an error page, so the mock gained a + `status_code = 200` - the state it was always standing in for. Its two assertions were not + touched, and this pins that: if the mock edit had relaxed the U-18 refusal instead of + describing a successful response, the upstream test would fail right here. + """ + + def test_u18_force_and_collision_smoke_test_passes_unchanged(self): + from tests.smoke import t_fetch_force_and_collision + + buf = io.StringIO() + with redirect_stdout(buf): + t_fetch_force_and_collision() + + +class TestAutoReusesAnAlreadyPresentModel(unittest.TestCase): + """`quantprobe auto` is the caller that pays for this, so it is the caller that is tested. + + `auto.run` treats a falsy `fetch` as fatal (`SystemExit("download failed...")`), so a skip + path that refuses an unverifiable HEAD does not degrade gracefully - it ends the command for + a user who already has the model on disk and no working network. This drives the real + `quantprobe.fetch.fetch` through the real `auto.run`; only the three external-I/O edges are + replaced (the HF tree listing, the GGUF header parse, and hardware detection), and no + provider, model download or inference is involved. + """ + + LISTED = ("model-Q4_K_M.gguf", 4_030_000_000) + + def setUp(self): + self.dir = tempfile.mkdtemp(prefix="qp-auto-") + self.addCleanup(shutil.rmtree, self.dir, ignore_errors=True) + self.local = os.path.join(self.dir, self.LISTED[0]) + with open(self.local, "wb") as f: + f.write(b"GGUF-already-on-disk") + real_requests, real_time = fmod.requests, fmod.time + fmod.time = FakeClock() + self.net = FakeRequests(requests.exceptions.ConnectionError("no route to host")) + fmod.requests = self.net + + def restore(): + fmod.requests, fmod.time = real_requests, real_time + + self.addCleanup(restore) + + def run_auto(self): + from quantprobe import auto as automod + from quantprobe import cli as climod + from quantprobe import detect as detmod + + argv = [ + "quantprobe", + "auto", + REPO, + "--total", + "7.2", + "--dir", + self.dir, + # explicit hardware: `optimize.resolve` only probes the host when every one of + # these is None, so detection is never reached. Stubbed as well, so a change to + # that rule shows up as a wrong number here rather than as a real hardware read. + "--vram", + "24", + "--vram-bw", + "936", + "--ram", + "64", + "--ram-bw", + "86", + "--disk-bw", + "3", + ] + buf = io.StringIO() + with ( + mock.patch.object(sys, "argv", argv), + mock.patch.object(automod, "list_ggufs", lambda repo: [self.LISTED]), + mock.patch.object(automod, "local_spec_or_none", lambda *a: (None, None)), + mock.patch.object( + detmod, + "detect", + lambda *a, **k: (_ for _ in ()).throw(AssertionError("host hardware probed")), + ), + redirect_stdout(buf), + ): + climod.main() + return buf.getvalue() + + def test_auto_still_runs_a_present_model_when_the_head_fails(self): + printed = self.run_auto() + self.assertIn("ready. Run it:", printed) + self.assertIn(self.local, printed) + low = printed.lower() + self.assertIn("already present", low) + self.assertIn("not verified", low) + self.assertNotIn("already complete", low) + self.assertNotIn("size matches", low) + self.assertEqual(self.net.head_calls, 1, "the real fetch path must have been taken") + self.assertEqual(self.net.get_calls, 0, "a present model must not be re-downloaded") + with open(self.local, "rb") as f: + self.assertEqual(f.read(), b"GGUF-already-on-disk", "the cached file is untouched") + + +def run_smoke(): + """Hook for tests/smoke.py, which is pytest-free and collects plain callables.""" + suite = unittest.defaultTestLoader.loadTestsFromModule(sys.modules[__name__]) + buf = io.StringIO() + result = unittest.TextTestRunner(stream=buf, verbosity=0).run(suite) + if not result.wasSuccessful(): + raise AssertionError( + f"{len(result.failures)} failed, {len(result.errors)} errored:\n{buf.getvalue()}" + )