Files
ukmesh/viewshed-worker/link_queue_v3.py
T
gadgethd 1ebc496965 Map UI redesign, live-path visibility, feed latency, and security hardening (#19)
* Fix map node freshness consistency

* Harden output, ingest, caches, and WebSocket limits

* Enforce public visibility across derived data

* Harden proxy and operator deployment boundary

* Make owner grants authoritative and reconcile ACLs safely

* Bound path, spam, and statistics analysis

* Make link and coverage jobs crash-safe

* Implement strategic security remediation

* Fix production cutover configuration

* Fix disabled viewshed worker health signal

* Serve stale stats during background refresh

* Retain stale stats through refresh windows

* Bound analytics work to protect ingestion

* Prioritize summary warmup over chart scans

* Throttle path history rebuilds

* Bound path history result memory

* Stream path history aggregation

* Give bounded path rebuild one CPU

* Serve stale charts during bounded refresh

* Prioritize startup stats before chart scans

* Bound path history segment cardinality

* Pin path rebuild context to privacy generation

* Self-host original frontend fonts

* Allow bounded path rebuild to complete

* Improve live map UI and low-latency group feed

- Dock node details on the right with selection highlight and collapsible layers
- Add node legend, 24h activity sparkline, copy-link, and layout/overlap fixes
- Keep all repeaters visible during Live Path focus
- Send GroupText feed packets immediately over WebSocket (no batch delay)
- Cache expensive stats/observer activity more aggressively to protect ingest
- Remove stale local planning/audit markdown from the tree

* fix(ci): supply OPERATOR_SITE_TOKEN for compose validation

Workers/Compose CI failed because docker-compose requires
OPERATOR_SITE_TOKEN. Add CI placeholders for that and MQTT_PASSWORD.
2026-07-27 02:39:12 +01:00

306 lines
10 KiB
Python

"""Crash-safe Redis protocol for MeshCore link jobs."""
import hashlib
import json
import os
import secrets
import threading
import time
import uuid
READY = 'meshcore:link:v3:ready'
DEFERRED = 'meshcore:link:v3:deferred'
PAYLOADS = 'meshcore:link:v3:payloads'
STATES = 'meshcore:link:v3:states'
ATTEMPTS = 'meshcore:link:v3:attempts'
BYTES = 'meshcore:link:v3:bytes'
DEDUPE = 'meshcore:link:v3:dedupe'
DEDUPE_BY_JOB = 'meshcore:link:v3:dedupe_by_job'
LEASES = 'meshcore:link:v3:leases'
TOKENS = 'meshcore:link:v3:tokens'
DEAD = 'meshcore:link:v3:dead'
COMPLETED = 'meshcore:link:v3:completed'
COUNTERS = 'meshcore:link:v3:counters'
REBUILD = 'meshcore:link:v3:rebuild'
WORKER_HEARTBEAT = 'meshcore:link:v3:worker_heartbeat'
MAX_JOBS = max(1, min(100_000, int(os.environ.get('LINK_QUEUE_V3_MAX_JOBS', '5000'))))
MAX_BYTES = max(1, min(1024 * 1024 * 1024, int(os.environ.get('LINK_QUEUE_V3_MAX_BYTES', str(64 * 1024 * 1024)))))
MAX_PAYLOAD_BYTES = max(1, min(1024 * 1024, int(os.environ.get('LINK_QUEUE_V3_MAX_PAYLOAD_BYTES', str(32 * 1024)))))
MAX_ATTEMPTS = max(1, min(20, int(os.environ.get('LINK_QUEUE_V3_MAX_ATTEMPTS', '5'))))
LEASE_MS = max(10_000, min(30 * 60_000, int(os.environ.get('LINK_QUEUE_V3_LEASE_MS', '120000'))))
COMPLETED_RETENTION_MS = max(60_000, int(os.environ.get('LINK_QUEUE_V3_COMPLETED_RETENTION_MS', str(7 * 24 * 60 * 60_000))))
ADMIT_SCRIPT = """
local existing = redis.call('HGET', KEYS[6], ARGV[2])
if existing then
local existing_state = redis.call('HGET', KEYS[3], existing)
if existing == ARGV[1] and existing_state == 'complete' then
return {'duplicate', existing}
end
if existing_state == 'queued' or existing_state == 'in_flight' or existing_state == 'dead' then
return {'coalesced', existing}
end
end
local payload_bytes = tonumber(ARGV[4])
if payload_bytes > tonumber(ARGV[7]) then return {'oversized', ''} end
local count = tonumber(redis.call('HGET', KEYS[10], 'count') or '0')
local bytes = tonumber(redis.call('HGET', KEYS[10], 'bytes') or '0')
if count + 1 > tonumber(ARGV[5]) or bytes + payload_bytes > tonumber(ARGV[6]) then
return {'full', ''}
end
redis.call('HSET', KEYS[2], ARGV[1], ARGV[3])
redis.call('HSET', KEYS[3], ARGV[1], 'queued')
redis.call('HSET', KEYS[4], ARGV[1], '0')
redis.call('HSET', KEYS[5], ARGV[1], tostring(payload_bytes))
redis.call('HSET', KEYS[6], ARGV[2], ARGV[1])
redis.call('HSET', KEYS[7], ARGV[1], ARGV[2])
redis.call('HINCRBY', KEYS[10], 'count', 1)
redis.call('HINCRBY', KEYS[10], 'bytes', payload_bytes)
if ARGV[8] == '' and redis.call('EXISTS', KEYS[9]) == 1 then
redis.call('LPUSH', KEYS[8], ARGV[1])
else
redis.call('LPUSH', KEYS[1], ARGV[1])
end
return {'accepted', ARGV[1]}
"""
CLAIM_SCRIPT = """
if redis.call('EXISTS', KEYS[8]) == 0 then
local recovered = 0
while recovered < 1000 do
local deferred_id = redis.call('RPOP', KEYS[7])
if not deferred_id then break end
if redis.call('HGET', KEYS[3], deferred_id) == 'queued' then
redis.call('LPUSH', KEYS[1], deferred_id)
recovered = recovered + 1
end
end
end
while true do
local job_id = redis.call('RPOP', KEYS[1])
if not job_id then return nil end
if redis.call('HGET', KEYS[3], job_id) == 'queued' then
redis.call('HSET', KEYS[3], job_id, 'in_flight')
redis.call('HINCRBY', KEYS[4], job_id, 1)
redis.call('HSET', KEYS[6], job_id, ARGV[1])
redis.call('ZADD', KEYS[5], ARGV[2], job_id)
local payload = redis.call('HGET', KEYS[2], job_id)
return {job_id, payload or '', redis.call('HGET', KEYS[4], job_id)}
end
end
"""
ACK_SCRIPT = """
if redis.call('HGET', KEYS[3], ARGV[1]) ~= 'in_flight'
or redis.call('HGET', KEYS[6], ARGV[1]) ~= ARGV[2] then
return 0
end
local payload_bytes = tonumber(redis.call('HGET', KEYS[5], ARGV[1]) or '0')
redis.call('ZREM', KEYS[7], ARGV[1])
redis.call('HDEL', KEYS[6], ARGV[1])
redis.call('HDEL', KEYS[2], ARGV[1])
redis.call('HDEL', KEYS[4], ARGV[1])
redis.call('HDEL', KEYS[5], ARGV[1])
redis.call('HSET', KEYS[3], ARGV[1], 'complete')
redis.call('ZADD', KEYS[10], ARGV[3], ARGV[1])
local count = math.max(0, tonumber(redis.call('HGET', KEYS[9], 'count') or '0') - 1)
local bytes = math.max(0, tonumber(redis.call('HGET', KEYS[9], 'bytes') or '0') - payload_bytes)
redis.call('HSET', KEYS[9], 'count', count, 'bytes', bytes)
return 1
"""
NACK_SCRIPT = """
if redis.call('HGET', KEYS[3], ARGV[1]) ~= 'in_flight'
or redis.call('HGET', KEYS[6], ARGV[1]) ~= ARGV[2] then
return 'invalid'
end
redis.call('ZREM', KEYS[7], ARGV[1])
redis.call('HDEL', KEYS[6], ARGV[1])
local attempts = tonumber(redis.call('HGET', KEYS[4], ARGV[1]) or '0')
if attempts >= tonumber(ARGV[3]) then
redis.call('HSET', KEYS[3], ARGV[1], 'dead')
redis.call('LPUSH', KEYS[8], ARGV[1])
return 'dead'
end
redis.call('HSET', KEYS[3], ARGV[1], 'queued')
redis.call('LPUSH', KEYS[1], ARGV[1])
return 'retry'
"""
RENEW_SCRIPT = """
if redis.call('HGET', KEYS[1], ARGV[1]) ~= 'in_flight'
or redis.call('HGET', KEYS[2], ARGV[1]) ~= ARGV[2] then
return 0
end
redis.call('ZADD', KEYS[3], ARGV[3], ARGV[1])
return 1
"""
REAP_SCRIPT = """
local expired = redis.call('ZRANGEBYSCORE', KEYS[1], '-inf', ARGV[1], 'LIMIT', 0, ARGV[2])
local count = 0
for _, job_id in ipairs(expired) do
redis.call('ZREM', KEYS[1], job_id)
if redis.call('HGET', KEYS[2], job_id) == 'in_flight' then
redis.call('HSET', KEYS[2], job_id, 'queued')
redis.call('HDEL', KEYS[3], job_id)
redis.call('LPUSH', KEYS[4], job_id)
count = count + 1
end
end
return count
"""
CLEAN_COMPLETED_SCRIPT = """
local expired = redis.call('ZRANGEBYSCORE', KEYS[1], '-inf', ARGV[1], 'LIMIT', 0, ARGV[2])
local count = 0
for _, job_id in ipairs(expired) do
redis.call('ZREM', KEYS[1], job_id)
if redis.call('HGET', KEYS[2], job_id) == 'complete' then
local dedupe_key = redis.call('HGET', KEYS[3], job_id)
if dedupe_key and redis.call('HGET', KEYS[4], dedupe_key) == job_id then
redis.call('HDEL', KEYS[4], dedupe_key)
end
redis.call('HDEL', KEYS[3], job_id)
redis.call('HDEL', KEYS[2], job_id)
count = count + 1
end
end
return count
"""
def _payload_bytes(payload: str) -> int:
return len(payload.encode('utf-8'))
def observation_identity(packet_hash: str, rx_node_id: str) -> tuple[str, str]:
digest = hashlib.sha256(f'observe\0{packet_hash.lower()}\0{rx_node_id.lower()}'.encode()).hexdigest()
return f'lo_{digest}', f'observe:{digest}'
def physical_identity(node_a_id: str, node_b_id: str, generation: str | None = None) -> tuple[str, str, str, str]:
a_id, b_id = sorted((node_a_id, node_b_id))
return f'lp_{uuid.uuid4()}', f'physical:{generation or "live"}:{a_id}:{b_id}', a_id, b_id
def admit(client, job: dict) -> tuple[str, str | None]:
payload = json.dumps(job, separators=(',', ':'), sort_keys=True)
result = client.eval(
ADMIT_SCRIPT, 10,
READY, PAYLOADS, STATES, ATTEMPTS, BYTES, DEDUPE, DEDUPE_BY_JOB,
DEFERRED, REBUILD, COUNTERS,
job['job_id'], job['dedupe_key'], payload, _payload_bytes(payload),
MAX_JOBS, MAX_BYTES, MAX_PAYLOAD_BYTES, job.get('generation') or '',
)
status = str(result[0])
job_id = str(result[1]) if result[1] else None
return status, job_id
def admit_physical(client, node_a_id: str, node_b_id: str, generation: str | None = None) -> tuple[str, str | None]:
job_id, dedupe_key, a_id, b_id = physical_identity(node_a_id, node_b_id, generation)
job = {
'version': 3, 'type': 'physical_pair', 'job_id': job_id,
'dedupe_key': dedupe_key, 'node_a_id': a_id, 'node_b_id': b_id,
}
if generation:
job['generation'] = generation
return admit(client, job)
def claim(client) -> tuple[str, str, dict, int] | None:
token = secrets.token_hex(16)
result = client.eval(
CLAIM_SCRIPT, 8, READY, PAYLOADS, STATES, ATTEMPTS, LEASES, TOKENS,
DEFERRED, REBUILD,
token, int(time.time() * 1000) + LEASE_MS,
)
if not result:
return None
return str(result[0]), token, json.loads(result[1]), int(result[2])
def ack(client, job_id: str, token: str) -> bool:
result = client.eval(
ACK_SCRIPT, 10,
READY, PAYLOADS, STATES, ATTEMPTS, BYTES, TOKENS, LEASES, DEAD,
COUNTERS, COMPLETED,
job_id, token, int(time.time() * 1000) + COMPLETED_RETENTION_MS,
)
return int(result) == 1
def nack(client, job_id: str, token: str) -> str:
return str(client.eval(
NACK_SCRIPT, 8,
READY, PAYLOADS, STATES, ATTEMPTS, BYTES, TOKENS, LEASES, DEAD,
job_id, token, MAX_ATTEMPTS,
))
def reap(client, limit: int = 100) -> int:
return int(client.eval(
REAP_SCRIPT, 4, LEASES, STATES, TOKENS, READY,
int(time.time() * 1000), limit,
))
def cleanup_completed(client, limit: int = 500) -> int:
return int(client.eval(
CLEAN_COMPLETED_SCRIPT, 4, COMPLETED, STATES, DEDUPE_BY_JOB, DEDUPE,
int(time.time() * 1000), limit,
))
class LeaseRenewer:
def __init__(self, redis_factory, job_id: str, token: str):
self.redis_factory = redis_factory
self.job_id = job_id
self.token = token
self.stop_event = threading.Event()
self.thread = threading.Thread(target=self._run, name=f'link-lease-{job_id[:8]}', daemon=True)
def _run(self):
client = self.redis_factory()
try:
while not self.stop_event.wait(max(1.0, LEASE_MS / 3000)):
renewed = client.eval(
RENEW_SCRIPT, 3, STATES, TOKENS, LEASES,
self.job_id, self.token, int(time.time() * 1000) + LEASE_MS,
)
if int(renewed) != 1:
return
finally:
client.close()
def __enter__(self):
self.thread.start()
return self
def __exit__(self, exc_type, exc, tb):
self.stop_event.set()
self.thread.join(timeout=5)
def start_worker_heartbeat(redis_factory, stop_event: threading.Event) -> threading.Thread:
def run():
client = None
while not stop_event.is_set():
try:
if client is None:
client = redis_factory()
client.set(WORKER_HEARTBEAT, str(int(time.time())), ex=45)
stop_event.wait(10)
except Exception:
if client is not None:
client.close()
client = None
stop_event.wait(2)
if client is not None:
client.close()
thread = threading.Thread(target=run, name='link-worker-heartbeat', daemon=True)
thread.start()
return thread