dht
This commit is contained in:
1 parent
c6c6276fe6
commit
698d0ca3f7
49 files changed
+1118
-53
No files matched your search
+58
-1
@@ -32,6 +32,63 @@ class ASNResolver:
|
||||
return
|
||||
self.cache[norm] = asn
|
||||
|
||||
async def resolve_async(self, ip: str | None, db_session=None) -> Optional[int]:
|
||||
"""Resolve ASN via persistent cache; fallback to RDAP API; store result.
|
||||
|
||||
- Checks in-memory cache first.
|
||||
- If not found, checks DB table rdap_cache when available.
|
||||
- If still not found, queries a public API and persists.
|
||||
"""
|
||||
norm = self.normalise(ip)
|
||||
if not norm:
|
||||
return None
|
||||
# In-memory cache first
|
||||
if norm in self.cache:
|
||||
return self.cache[norm]
|
||||
# DB lookup if possible
|
||||
try:
|
||||
if db_session is not None:
|
||||
from sqlalchemy import select
|
||||
from app.core.models.rdap import RdapCache
|
||||
row = (await db_session.execute(select(RdapCache).where(RdapCache.ip == norm))).scalars().first()
|
||||
if row and row.asn is not None:
|
||||
self.cache[norm] = int(row.asn)
|
||||
return int(row.asn)
|
||||
except Exception as e:
|
||||
make_log("ASNResolver", f"DB lookup failed for {norm}: {e}", level="warning")
|
||||
|
||||
# Remote lookup (best-effort)
|
||||
asn: Optional[int] = None
|
||||
try:
|
||||
import httpx
|
||||
url = f"https://api.iptoasn.com/v1/as/ip/{norm}"
|
||||
async with httpx.AsyncClient(timeout=5.0) as client:
|
||||
r = await client.get(url)
|
||||
if r.status_code == 200:
|
||||
j = r.json()
|
||||
num = j.get("as_number")
|
||||
if isinstance(num, int) and num > 0:
|
||||
asn = num
|
||||
except Exception as e:
|
||||
make_log("ASNResolver", f"RDAP lookup failed for {norm}: {e}", level="warning")
|
||||
|
||||
if asn is not None:
|
||||
self.cache[norm] = asn
|
||||
# Persist to DB if possible
|
||||
try:
|
||||
if db_session is not None:
|
||||
from app.core.models.rdap import RdapCache
|
||||
row = await db_session.get(RdapCache, norm)
|
||||
if row is None:
|
||||
row = RdapCache(ip=norm, asn=asn, source="iptoasn")
|
||||
db_session.add(row)
|
||||
else:
|
||||
row.asn = asn
|
||||
row.source = "iptoasn"
|
||||
await db_session.commit()
|
||||
except Exception as e:
|
||||
make_log("ASNResolver", f"DB persist failed for {norm}: {e}", level="warning")
|
||||
return asn
|
||||
|
||||
|
||||
resolver = ASNResolver()
|
||||
|
||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -41,6 +41,10 @@ class DHTConfig:
|
||||
window_size: int = _env_int("DHT_METRIC_WINDOW_SEC", 3600)
|
||||
default_q: float = _env_float("DHT_MIN_Q", 0.6)
|
||||
seed_refresh_interval: int = _env_int("DHT_SEED_REFRESH_INTERVAL", 30)
|
||||
# Gossip / backoff tuning
|
||||
gossip_interval_sec: int = _env_int("DHT_GOSSIP_INTERVAL_SEC", 30)
|
||||
gossip_backoff_base_sec: int = _env_int("DHT_GOSSIP_BACKOFF_BASE_SEC", 5)
|
||||
gossip_backoff_cap_sec: int = _env_int("DHT_GOSSIP_BACKOFF_CAP_SEC", 600)
|
||||
|
||||
|
||||
@lru_cache
|
||||
@@ -51,4 +55,3 @@ def load_config() -> DHTConfig:
|
||||
|
||||
|
||||
dht_config = load_config()
|
||||
|
||||
@@ -141,6 +141,23 @@ class MembershipState:
|
||||
return max(max(filtered_reports), local_estimate)
|
||||
return local_estimate
|
||||
|
||||
def n_estimate_trusted(self, allowed_ids: set[str]) -> float:
|
||||
"""Оценка размера сети только по trusted узлам.
|
||||
Берём активных участников, пересекаем с allowed_ids и оцениваем по их числу
|
||||
и по их N_local репортам (если доступны).
|
||||
"""
|
||||
self.report_local_population()
|
||||
active_trusted = {m["node_id"] for m in self.active_members(include_islands=True) if m.get("node_id") in allowed_ids}
|
||||
filtered_reports = [
|
||||
value for node_id, value in self.n_reports.items()
|
||||
if node_id in active_trusted and self.reachability_ratio(node_id) >= dht_config.default_q
|
||||
]
|
||||
# Для доверенных полагаемся на фактическое количество активных Trusted
|
||||
local_estimate = float(len(active_trusted))
|
||||
if filtered_reports:
|
||||
return max(max(filtered_reports), local_estimate)
|
||||
return local_estimate
|
||||
|
||||
def to_dict(self) -> Dict[str, Any]:
|
||||
return {
|
||||
"members": self.members.to_dict(),
|
||||
@@ -216,4 +233,3 @@ class MembershipManager:
|
||||
|
||||
def active_members(self) -> List[Dict[str, Any]]:
|
||||
return self.state.active_members()
|
||||
|
||||
@@ -29,6 +29,10 @@ merge_conflicts = Counter("dht_merge_conflicts_total", "Number of DHT merge conf
|
||||
view_count_total = Gauge("dht_view_count_total", "Total content views per window", ["content_id", "window"])
|
||||
unique_estimate = Gauge("dht_unique_view_estimate", "Estimated unique viewers per window", ["content_id", "window"])
|
||||
watch_time_seconds = Gauge("dht_watch_time_seconds", "Aggregate watch time per window", ["content_id", "window"])
|
||||
gossip_success = Counter("dht_gossip_success_total", "Successful gossip posts", ["peer"])
|
||||
gossip_failure = Counter("dht_gossip_failure_total", "Failed gossip posts", ["peer"])
|
||||
gossip_skipped = Counter("dht_gossip_skipped_total", "Skipped gossip posts due to backoff", ["peer", "reason"])
|
||||
gossip_backoff = Gauge("dht_gossip_backoff_seconds", "Gossip backoff seconds remaining", ["peer"])
|
||||
|
||||
|
||||
def record_replication_under(content_id: str, have: int) -> None:
|
||||
@@ -51,3 +55,17 @@ def update_view_metrics(content_id: str, window_id: str, views: int, unique: flo
|
||||
view_count_total.labels(content_id=content_id, window=window_id).set(views)
|
||||
unique_estimate.labels(content_id=content_id, window=window_id).set(unique)
|
||||
watch_time_seconds.labels(content_id=content_id, window=window_id).set(watch_time)
|
||||
|
||||
|
||||
def record_gossip_success(peer: str) -> None:
|
||||
gossip_success.labels(peer=peer).inc()
|
||||
gossip_backoff.labels(peer=peer).set(0)
|
||||
|
||||
|
||||
def record_gossip_failure(peer: str, backoff_sec: float) -> None:
|
||||
gossip_failure.labels(peer=peer).inc()
|
||||
gossip_backoff.labels(peer=peer).set(backoff_sec)
|
||||
|
||||
|
||||
def record_gossip_skipped(peer: str, reason: str) -> None:
|
||||
gossip_skipped.labels(peer=peer, reason=reason).inc()
|
||||
@@ -166,15 +166,20 @@ class ReplicationManager:
|
||||
.to_dict(),
|
||||
)
|
||||
|
||||
def ensure_replication(self, content_id: str, membership: MembershipState, now: Optional[float] = None) -> ReplicationState:
|
||||
def ensure_replication(self, content_id: str, membership: MembershipState, now: Optional[float] = None, allowed_nodes: Optional[set[str]] = None) -> ReplicationState:
|
||||
now = now or _now()
|
||||
state = self._load_state(content_id)
|
||||
|
||||
n_estimate = max(1.0, membership.n_estimate())
|
||||
if allowed_nodes is not None and len(allowed_nodes) > 0:
|
||||
n_estimate = max(1.0, membership.n_estimate_trusted(allowed_nodes))
|
||||
else:
|
||||
n_estimate = max(1.0, membership.n_estimate())
|
||||
p_value = max(0, round(math.log2(max(n_estimate / dht_config.replication_target, 1.0))))
|
||||
prefix, _ = bits_from_hex(content_id, p_value)
|
||||
|
||||
active = membership.active_members(include_islands=True)
|
||||
if allowed_nodes is not None:
|
||||
active = [m for m in active if m.get("node_id") in allowed_nodes]
|
||||
responsible = []
|
||||
for member in active:
|
||||
node_prefix, _total = bits_from_hex(member["node_id"], p_value)
|
||||
@@ -265,6 +270,44 @@ class ReplicationManager:
|
||||
rest = [m for m in active if m["node_id"] not in {n for _, n, *_ in rank(responsible)}]
|
||||
assign_with_diversity(rank(rest))
|
||||
|
||||
# Финальный добор по ASN, если всё ещё не достигли диверсификации
|
||||
if not state.diversity_satisfied():
|
||||
current_asn = {lease.asn for lease in state.leases.values() if lease.asn is not None}
|
||||
by_asn: Dict[int, List[dict]] = {}
|
||||
for m in active:
|
||||
a = m.get('asn')
|
||||
if a is None:
|
||||
continue
|
||||
by_asn.setdefault(int(a), []).append(m)
|
||||
for a, group in by_asn.items():
|
||||
if a in current_asn:
|
||||
continue
|
||||
# берём лучшего кандидата этой ASN
|
||||
score, node_id, asn, ip_octet = min(
|
||||
(
|
||||
(rendezvous_score(content_id, g["node_id"]), g["node_id"], g.get("asn"), g.get("ip_first_octet"))
|
||||
for g in group
|
||||
),
|
||||
key=lambda item: item[0],
|
||||
)
|
||||
if node_id in leases_by_node:
|
||||
continue
|
||||
lease = ReplicaLease(
|
||||
node_id=node_id,
|
||||
lease_id=f"{content_id}:{node_id}",
|
||||
issued_at=now,
|
||||
expires_at=now + dht_config.lease_ttl,
|
||||
asn=asn,
|
||||
ip_first_octet=ip_octet,
|
||||
heartbeat_at=now,
|
||||
score=score,
|
||||
)
|
||||
state.assign(lease)
|
||||
leases_by_node[node_id] = lease
|
||||
current_asn.add(int(a))
|
||||
if state.diversity_satisfied():
|
||||
break
|
||||
|
||||
# Ensure we do not exceed replication target with duplicates
|
||||
if len(state.leases) > dht_config.replication_target:
|
||||
# Drop lowest scoring leases until target satisfied while preserving diversity criteria
|
||||
|
||||
@@ -55,3 +55,86 @@ class DHTStore:
|
||||
|
||||
def snapshot(self) -> Dict[str, Dict[str, Any]]:
|
||||
return {fp: record.to_payload() | {"signature": record.signature} for fp, record in self._records.items()}
|
||||
|
||||
|
||||
# ---- Persistent adapter (DB-backed) ----
|
||||
try:
|
||||
from sqlalchemy import create_engine
|
||||
from sqlalchemy.orm import sessionmaker
|
||||
from app.core._config import DATABASE_URL
|
||||
from app.core.models.dht import DHTRecordRow
|
||||
|
||||
def _sync_engine_url(url: str) -> str:
|
||||
return url.replace('+asyncpg', '+psycopg2') if '+asyncpg' in url else url
|
||||
|
||||
class PersistentDHTStore(DHTStore):
|
||||
"""DHT хранилище с простейшей синхронной персистенцией в БД."""
|
||||
|
||||
def __init__(self, node_id: str, signer: Signer, db_url: str | None = None):
|
||||
super().__init__(node_id, signer)
|
||||
self._engine = create_engine(_sync_engine_url(db_url or DATABASE_URL), pool_pre_ping=True)
|
||||
self._Session = sessionmaker(bind=self._engine)
|
||||
|
||||
def _db_get(self, fingerprint: str) -> DHTRecord | None:
|
||||
with self._Session() as s:
|
||||
row = s.get(DHTRecordRow, fingerprint)
|
||||
if not row:
|
||||
return None
|
||||
rec = DHTRecord.create(
|
||||
key=row.key,
|
||||
fingerprint=row.fingerprint,
|
||||
value=row.value or {},
|
||||
node_id=row.node_id,
|
||||
logical_counter=row.logical_counter,
|
||||
signature=row.signature,
|
||||
timestamp=row.timestamp,
|
||||
)
|
||||
return rec
|
||||
|
||||
def _db_put(self, record: DHTRecord) -> None:
|
||||
with self._Session() as s:
|
||||
row = s.get(DHTRecordRow, record.fingerprint)
|
||||
if not row:
|
||||
row = DHTRecordRow(
|
||||
fingerprint=record.fingerprint,
|
||||
key=record.key,
|
||||
schema_version=record.schema_version,
|
||||
logical_counter=record.logical_counter,
|
||||
timestamp=record.timestamp,
|
||||
node_id=record.node_id,
|
||||
signature=record.signature,
|
||||
value=record.value,
|
||||
)
|
||||
s.add(row)
|
||||
else:
|
||||
row.key = record.key
|
||||
row.schema_version = record.schema_version
|
||||
row.logical_counter = record.logical_counter
|
||||
row.timestamp = record.timestamp
|
||||
row.node_id = record.node_id
|
||||
row.signature = record.signature
|
||||
row.value = record.value
|
||||
s.commit()
|
||||
|
||||
def get(self, fingerprint: str) -> DHTRecord | None:
|
||||
rec = super().get(fingerprint)
|
||||
if rec:
|
||||
return rec
|
||||
rec = self._db_get(fingerprint)
|
||||
if rec:
|
||||
self._records[fingerprint] = rec
|
||||
return rec
|
||||
|
||||
def put(self, key: str, fingerprint: str, value: Dict[str, Any], logical_counter: int, merge_strategy=latest_wins_merge) -> DHTRecord:
|
||||
rec = super().put(key, fingerprint, value, logical_counter, merge_strategy)
|
||||
self._db_put(rec)
|
||||
return rec
|
||||
|
||||
def merge_record(self, incoming: DHTRecord, merge_strategy=latest_wins_merge) -> DHTRecord:
|
||||
merged = super().merge_record(incoming, merge_strategy)
|
||||
self._db_put(merged)
|
||||
return merged
|
||||
except Exception:
|
||||
# Fallback: без SQLAlchemy используем чисто in-memory (для тестов/минимальных окружений)
|
||||
class PersistentDHTStore(DHTStore):
|
||||
pass
|
||||
@@ -7,6 +7,8 @@ from sqlalchemy import select
|
||||
from app.core.logger import make_log
|
||||
from app.core.models.node_storage import StoredContent
|
||||
from app.core.network.dht import dht_config
|
||||
from app.core.network.nodes import list_known_public_nodes
|
||||
import httpx
|
||||
from app.core.storage import db_session
|
||||
|
||||
|
||||
@@ -23,9 +25,25 @@ async def replication_daemon(app):
|
||||
async with db_session(auto_commit=False) as session:
|
||||
rows = await session.execute(select(StoredContent.hash))
|
||||
content_hashes = [row[0] for row in rows.all()]
|
||||
# Build allowed (trusted) node_ids set
|
||||
from app.core.models.my_network import KnownNode
|
||||
from app.core._utils.b58 import b58decode
|
||||
from app.core.network.dht.crypto import compute_node_id
|
||||
trusted_rows = (await session.execute(select(KnownNode))).scalars().all()
|
||||
allowed_nodes = set()
|
||||
for kn in trusted_rows:
|
||||
role = (kn.meta or {}).get('role') if kn.meta else None
|
||||
if role == 'trusted' and kn.public_key:
|
||||
try:
|
||||
nid = compute_node_id(b58decode(kn.public_key))
|
||||
allowed_nodes.add(nid)
|
||||
except Exception:
|
||||
pass
|
||||
# Always include ourselves
|
||||
allowed_nodes.add(memory.node_id)
|
||||
for content_hash in content_hashes:
|
||||
try:
|
||||
state = memory.replication.ensure_replication(content_hash, membership_state)
|
||||
state = memory.replication.ensure_replication(content_hash, membership_state, allowed_nodes=allowed_nodes)
|
||||
memory.replication.heartbeat(content_hash, memory.node_id)
|
||||
make_log("Replication", f"Replicated {content_hash} leader={state.leader}", level="debug")
|
||||
except Exception as exc:
|
||||
@@ -50,3 +68,74 @@ async def heartbeat_daemon(app):
|
||||
except Exception as exc:
|
||||
make_log("Replication", f"heartbeat failed: {exc}", level="warning")
|
||||
await asyncio.sleep(dht_config.heartbeat_interval)
|
||||
|
||||
|
||||
async def dht_gossip_daemon(app):
|
||||
# Периодически публикуем снимок DHT на известные публичные ноды
|
||||
await asyncio.sleep(7)
|
||||
memory = getattr(app.ctx, "memory", None)
|
||||
if not memory:
|
||||
return
|
||||
while True:
|
||||
try:
|
||||
# собираем список публичных хостов доверенных узлов
|
||||
async with db_session(auto_commit=True) as session:
|
||||
from sqlalchemy import select
|
||||
from app.core.models.my_network import KnownNode
|
||||
nodes = (await session.execute(select(KnownNode))).scalars().all()
|
||||
urls = []
|
||||
peers = []
|
||||
for n in nodes:
|
||||
role = (n.meta or {}).get('role') if n.meta else None
|
||||
if role != 'trusted':
|
||||
continue
|
||||
pub = (n.meta or {}).get('public_host') or n.ip
|
||||
if not pub:
|
||||
continue
|
||||
base = pub.rstrip('/')
|
||||
if not base.startswith('http'):
|
||||
base = f"http://{base}:{n.port or 80}"
|
||||
url = base + "/api/v1/dht.put"
|
||||
urls.append(url)
|
||||
peers.append(base)
|
||||
if urls:
|
||||
snapshot = memory.dht_store.snapshot()
|
||||
records = list(snapshot.values())
|
||||
payload = {
|
||||
'public_key': None, # получатель обязует себя проверять подпись записи, не отправителя тут
|
||||
'records': records,
|
||||
}
|
||||
# public_key не обязателен в пакетном режиме, записи содержат собственные подписи
|
||||
timeout = httpx.Timeout(5.0, read=10.0)
|
||||
async with httpx.AsyncClient(timeout=timeout) as client:
|
||||
# Throttle/backoff per peer
|
||||
from time import time as _t
|
||||
from app.core.network.dht.prometheus import record_gossip_success, record_gossip_failure, record_gossip_skipped
|
||||
if not hasattr(dht_gossip_daemon, '_backoff'):
|
||||
dht_gossip_daemon._backoff = {}
|
||||
backoff = dht_gossip_daemon._backoff # type: ignore
|
||||
for url, peer in zip(urls, peers):
|
||||
now = _t()
|
||||
st = backoff.get(peer) or { 'fail': 0, 'next': 0.0 }
|
||||
if now < st['next']:
|
||||
record_gossip_skipped(peer, 'backoff')
|
||||
continue
|
||||
try:
|
||||
r = await client.post(url, json=payload)
|
||||
if 200 <= r.status_code < 300:
|
||||
backoff[peer] = { 'fail': 0, 'next': 0.0 }
|
||||
record_gossip_success(peer)
|
||||
else:
|
||||
raise Exception(f"HTTP {r.status_code}")
|
||||
except Exception as exc:
|
||||
st['fail'] = st.get('fail', 0) + 1
|
||||
base = max(1, int(dht_config.gossip_backoff_base_sec))
|
||||
cap = max(base, int(dht_config.gossip_backoff_cap_sec))
|
||||
wait = min(cap, base * (2 ** max(0, st['fail'] - 1)))
|
||||
st['next'] = now + wait
|
||||
backoff[peer] = st
|
||||
record_gossip_failure(peer, wait)
|
||||
make_log('DHT.gossip', f'gossip failed {url}: {exc}; backoff {wait}s', level='debug')
|
||||
except Exception as exc:
|
||||
make_log('DHT.gossip', f'iteration error: {exc}', level='warning')
|
||||
await asyncio.sleep(dht_config.gossip_interval_sec)
|
||||
Reference in new issue
Block a user