commit d8208685a297c218c57b3f993d2073bb1a95751b Author: ookami125 Date: Sun Jun 21 23:12:32 2026 -0400 Initial commit: multi-peer torrent download engine Reactor/loop-pool engine with TCP/µTP/MSE transports, per-connection pipelining, priority-driven piece selection with endgame, and the Python FFI test harness. Co-Authored-By: Claude Opus 4.8 diff --git a/.dockerignore b/.dockerignore new file mode 100644 index 0000000..676fbc3 --- /dev/null +++ b/.dockerignore @@ -0,0 +1,9 @@ +build/ +build-asan/ +__pycache__/ +*.pyc +*.torrent +.git/ +.agents/ +.codex/ +interop/results/ diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..9cf9857 --- /dev/null +++ b/.gitignore @@ -0,0 +1,7 @@ +build/ +build-asan/ +__pycache__/ +*.pyc +*.torrent +interop/results/ +*.mkv diff --git a/CMakeLists.txt b/CMakeLists.txt new file mode 100644 index 0000000..c572943 --- /dev/null +++ b/CMakeLists.txt @@ -0,0 +1,58 @@ +cmake_minimum_required(VERSION 3.16) +project(torrent_peer C) + +set(CMAKE_C_STANDARD 11) +set(CMAKE_C_STANDARD_REQUIRED ON) + +if(NOT CMAKE_BUILD_TYPE) + set(CMAKE_BUILD_TYPE Release) +endif() + +# Fast by default; -march=native is great for local benchmarking. Disable with +# -DPEER_NATIVE=OFF for portable / CI builds. +option(PEER_NATIVE "Optimize for the build host (-march=native)" ON) +option(PEER_ASAN "Build with AddressSanitizer/UBSan" OFF) + +add_library(torrentpeer SHARED + src/engine.c + src/loop.c + src/connection.c + src/transport.c + src/transport_mse.c + src/transport_utp.c + src/crypto.c + src/proto.c + src/scheduler.c + src/reqtab.c + src/ring.c + src/arena.c + src/peer_compat.c +) +target_include_directories(torrentpeer PUBLIC include) + +target_compile_options(torrentpeer PRIVATE + -O3 -Wall -Wextra -Wno-unused-parameter +) +if(PEER_NATIVE AND NOT PEER_ASAN) + target_compile_options(torrentpeer PRIVATE -march=native) +endif() +if(PEER_ASAN) + target_compile_options(torrentpeer PRIVATE -O1 -g -fsanitize=address,undefined -fno-omit-frame-pointer) + target_link_options(torrentpeer PRIVATE -fsanitize=address,undefined) +endif() + +find_package(Threads REQUIRED) +target_link_libraries(torrentpeer PRIVATE Threads::Threads) + +# P3a: link-time optimization for release builds (skipped under ASan). +option(PEER_LTO "Enable link-time optimization" ON) +if(PEER_LTO AND NOT PEER_ASAN) + include(CheckIPOSupported) + check_ipo_supported(RESULT _ipo_ok OUTPUT _ipo_msg) + if(_ipo_ok) + set_target_properties(torrentpeer PROPERTIES INTERPROCEDURAL_OPTIMIZATION ON) + endif() +endif() + +# Keep the .so name predictable for ctypes: libtorrentpeer.so +set_target_properties(torrentpeer PROPERTIES OUTPUT_NAME torrentpeer) diff --git a/ISSUES.md b/ISSUES.md new file mode 100644 index 0000000..5ff7d6e --- /dev/null +++ b/ISSUES.md @@ -0,0 +1,2 @@ +1) ~~Slow peers can hog pieces and can stall downloads towards the end. We need to a good dynamic end game.~~ +2) ~~The testing harness (swarm_download.py) needs to be able to resume a torrent download.~~ diff --git a/PLAN.md b/PLAN.md new file mode 100644 index 0000000..d956919 --- /dev/null +++ b/PLAN.md @@ -0,0 +1,366 @@ +# torrent-peer — Forward Work Plan + +Forward plan for evolving this from a single-peer, leech-only, TCP block fetcher +into a **unified multi-peer download engine**. Derived from REVIEW.md and the +architecture discussion that followed it. + +> **Pivot from the previous plan.** Earlier this library treated one +> `peer_handle` as one connection and pushed all concurrency to the parent +> ("more peers = more handles"). We are reversing that: the engine now owns +> *all* peer connections itself, via an event-loop pool. This brings cross-peer +> piece picking, endgame, and request-timeout handling **in scope**, and +> removes the thread-per-peer model. If the real target were only a few fast +> peers (seedbox-to-seedbox), the old single-peer model would have been fine; +> this plan assumes a general client over real swarms (many peers, many +> torrents, heavy churn) — the workload a reactor is built for. + +--- + +## Architectural decision: unified multi-peer engine + +**Why restructure.** BitTorrent's connection profile is "many peers, most slow +or idle, with churn" — the classic c10k reactor workload. Thread-per-peer (the +current `peer_start` → one net thread per fd, `peer.c:115`) costs MBs of stack +per thread × thousands, context-switch overhead, and the `RUNNING_TICK_MS` +2 ms-tick-×-N-threads waste. Every production client (libtorrent, rtorrent, +transmission) multiplexes many sockets over a few threads instead. + +**Shape: a fixed pool of event loops, with torrents pinned to a loop +(affinity).** Not one-loop-per-torrent (thread explosion + bad balance), and not +a fully global pool with a torrent's peers scattered across loops (which would +make the per-torrent piece picker contended). Instead: a small pool of loops +(≈ the cores spent on networking), each hosting *many* torrents, but **all of a +given torrent's peers live on its one assigned loop**. This yields a lock-free, +single-threaded piece picker per torrent, natural load balancing across loops, +and thread count decoupled from torrent count. + +**Escape hatch:** a single torrent hot enough to need more than one core +(saturating 10GbE from one swarm) is the only case wanting its peers sharded +across loops — and the only case that pays for a sharded/locked picker. Default +to affinity; treat cross-loop sharding as an opt-in for the rare mega-torrent. + +**What's reusable (most of the existing code).** The wire parser (`proto.c`), +the per-piece scheduling logic (`scheduler.c`), the arena handoff, and the SPSC +rings (`ring.h`) survive as a per-*connection* state machine. The restructure is +concentrated in `net.c` (the I/O driver and threading) and the public ABI. + +**What comes into scope because of the pivot:** +- Cross-peer / swarm-wide piece picking (rarest-first across all peers of a + torrent), replacing the single-peer priority scan. +- Endgame mode + `CANCEL` (was REVIEW P2d "parent's job" — now ours). +- Per-request in-flight tracking + timeout/re-request (REVIEW C1) and + unsolicited-block validation (REVIEW C2) — now first-class engine concerns. + +--- + +## Target object model + +``` +Engine process-global; owns the loop pool + worker pools + ├─ Loop[0..L) one OS thread each; owns an epoll/io_uring instance, + │ ├─ arena slab its set of fds, and ONE per-loop arena slab + │ ├─ command queue (eventfd) cross-thread control in (add torrent, set priority…) + │ └─ Connection[*] every conn on this loop is single-thread-owned + ├─ Torrent[*] pinned to one Loop; owns the piece picker, + │ ├─ piece picker availability counts, requested[], in-flight map — + │ ├─ priority vector touched ONLY by its loop ⇒ no lock + │ └─ peers ──────────────► (its Connections, all on the same loop) + └─ HashPool (optional) shared worker threads for in-library verification +``` + +- **Loop** is the sole owner of its connections, their `reader`/`outbuf`, and its + arena. Hot path (recv → parse → handoff → scheduler tick) is lock-free because + one thread touches all of it. +- **Torrent** state (picker, `have` counts, `requested`, in-flight) is touched + only by the torrent's loop ⇒ lock-free picking. A torrent is assigned to a + loop at add time (least-loaded loop), and all its peers are opened on that loop. +- **Control plane** (add/remove torrent, set priorities, add peer, stop) is + cross-thread: enqueue a command on the target loop's command queue and poke its + eventfd — the same pattern as today's `ctrl_efd` (`peer.c:124 poke`). +- **Data plane / consumer handoff:** two modes (see §7): + - *Harness/zero-copy mode (default, today's behavior):* per-connection SPSC + `ready_ring` + `free_ring` over the per-loop arena; the consumer drains many + rings. Slot lifetime spans the consumer. + - *In-library hashing mode:* the loop copies each block out of its slot into a + per-piece assembly buffer and **releases the slot immediately**, so slots + never leave the loop and no cross-thread free ring is needed. + +**ABI impact (decision needed).** The public surface moves from `peer_handle` +to an engine-centric API: `engine_create/destroy`, `engine_add_torrent`, +`torrent_set_priorities`, `engine_add_peer(torrent, ip, port)`, plus a +completion-draining call that spans connections. The Python ctypes layer +(`peer_ffi.py`) and harness change accordingly. Keep a thin single-peer +convenience wrapper so existing tests/usage migrate incrementally. + +--- + +## Guiding constraints (do not violate) + +- **One thread owns a connection end-to-end.** No connection state is shared + across loop threads; all cross-thread interaction is via command queues. This + is what keeps the hot path and the piece picker lock-free. +- **Keep the zero-copy harness path working.** In-library hashing, io_uring, + encryption, and µTP are additive/opt-in. The default (epoll + zero-copy arena + handoff) must not regress; the existing end-to-end test must keep passing. +- **Affinity by default.** A torrent's peers stay on its loop unless cross-loop + sharding is explicitly enabled for that torrent. + +--- + +## Scope table + +| Item | REVIEW id | Decision | +|---|---|---| +| Reactor / loop-pool restructure (de-thread-per-peer) | P2a (reframed) | **Keystone — implement** | +| Cross-peer rarest-first picker | P2a follow-on | Implement | +| Request timeout + in-flight table | C1 | Implement (engine core) | +| Unsolicited-block validation | C2 | Implement (engine core) | +| Endgame mode + CANCEL | P2d | **In scope now** (engine owns cross-peer policy) | +| Transport abstraction (vtable) | enabler | Implement (keystone-b) | +| IPv6 | P2e | Implement | +| Fast Extension (BEP-6) + Extension Protocol (BEP-10) | P2e | Implement | +| io_uring receive path | P2b | Implement (opt-in + epoll fallback) | +| In-library SHA hashing | P2c | Implement (opt-in) | +| MSE/PE encryption | P2e | Implement | +| µTP | P2e | Implement | + +P1 throughput items from REVIEW (event-driven slot release, don't pin +`SO_RCVBUF`, adaptive pipeline depth, true zero-copy recv, `outstanding` data +race) are folded into the relevant sections below rather than tracked +separately. + +--- + +## §0a. Keystone: reactor restructure (the big one) + +Convert from thread-per-peer to a loop pool. Do this first; it reshapes +everything else. + +**Today:** `net_thread_main` (`net.c:125`) is one thread per peer running its own +`epoll_wait` + `scheduler_tick` + `flush_out`, pulling bytes via `recv()` inside +`proto.c`. + +**Change:** +1. **Loop object:** one thread, one epoll (later io_uring) instance, a set of + connections, a per-loop arena, and a command-queue eventfd. The loop body is + today's `net_thread_main` body generalized to iterate over *all* ready + connections per wakeup instead of one fd. +2. **Push-model parser:** change `proto.c` from *pull* (`rd_refill`/`rd_take_body` + calling `recv`) to *push*: a `proto_feed(conn, buf, len)` entry point the loop + calls with received bytes. This is also a prerequisite for io_uring (§6) and + removes the per-body direct-`recv` syscall pattern. The parser state machine + (`RS_*`) is unchanged; only its byte source changes. +3. **Per-loop arena:** one slab shared by the loop's connections; slot allocation + is single-threaded ⇒ lock-free. Memory now scales with loops, not peers + (kills the ×N arena blowup of per-peer arenas). +4. **Event-driven slot release (REVIEW P1a):** returned credit wakes the owning + loop via its command queue (coalesced), removing the 2 ms-tick latency floor. +5. **Don't pin `SO_RCVBUF` (REVIEW P1c):** drop the fixed 8 MiB `SO_RCVBUF` + (`net.c:96`) so the kernel autotunes on high-BDP links; keep it an opt-in + tunable. +6. **Fix the `outstanding` data race (REVIEW P1f):** make per-connection + counters that status reads touch atomic. + +- **Effort:** ~1–1.5 weeks. **Risk:** medium-high (core rewrite of the I/O layer). +- **Tests:** existing end-to-end test passes with a 1-loop / 1-peer config; + add a test with multiple peers of one torrent on one loop, and multiple + torrents across loops. + +## §0b. Keystone: transport abstraction + +Decouple the parser from the raw socket so io_uring/MSE/µTP can slot in. Small; +do alongside §0a (the push-model parser is the shared piece). + +```c +typedef struct { + ssize_t (*recv)(void *ctx, void *buf, size_t n); /* -1/EAGAIN, 0 closed */ + ssize_t (*send)(void *ctx, const void *buf, size_t n); + int (*want_fds)(void *ctx, int *fds, int max); /* fds the loop polls */ + void (*close)(void *ctx); + void *ctx; +} transport; +``` + +Implementations: `transport_tcp` (current epoll path), later `transport_uring`, +`transport_mse` (wraps an inner transport), `transport_utp`. The loop drives +`t->recv`/`t->send`; encryption and µTP compose by wrapping an inner transport. + +- **Effort:** ~1 day on top of §0a. **Risk:** low. + +--- + +## §1. Cross-peer piece picker + in-flight tracking + timeout (correctness core) + +Replaces the single-peer `select_next_piece` scan (`scheduler.c:25`) with a +per-torrent picker shared across that torrent's peers (lock-free, since one loop +owns the torrent). + +**Design:** +- **Availability counts:** maintain per-piece rarity from every peer's + `bitfield`/`have` (and `HAVE_ALL`/`HAVE_NONE` from §5). Picker chooses + highest-priority, then rarest, then lowest index, among pieces some connected + peer has and that aren't fully in-flight. +- **In-flight map (REVIEW C1):** track outstanding `(piece, begin)` per request + with the issuing connection and a deadline. On timeout: drop the entry, + decrement that connection's `outstanding`, and re-arm the block for any peer + that has it. This closes the "silent drop → permanent stall" hole. +- **Unsolicited-block validation (REVIEW C2):** on an inbound piece message, look + it up in the in-flight map; if absent, drop the payload **without** consuming a + slot or decrementing `outstanding`. +- **Block-level requests:** the picker now hands out blocks (not whole pieces), + enabling the same block to be re-issued to another peer (timeout/endgame). +- **Adaptive pipeline depth (REVIEW P1d):** size per-connection in-flight to the + measured BDP (RTT from request→first-byte × rate from `update_rate`), + clamped by arena credit. + +- **Effort:** ~1 week. **Risk:** medium. **Tests:** multi-peer seed where one + peer stalls a block — assert it's re-requested from another and the download + completes; negative test for unsolicited blocks. + +## §2. Endgame mode + CANCEL + +Now in scope because the engine owns cross-peer policy. + +**Design:** when remaining blocks < threshold, allow a block to be requested from +multiple peers simultaneously; when one arrives, send `CANCEL` to the others and +drop their in-flight entries. Reuses the §1 in-flight map and block-level +requests. `CANCEL` send path lives in `proto.c` (currently CANCEL is only +recognized inbound). + +- **Effort:** ~2–3 days. **Risk:** low-medium (depends on §1). **Tests:** seed + with one deliberately slow peer near completion; assert endgame races finish + the tail and losers receive `CANCEL`. + +## §3. IPv6 (small; do early) + +`AF_INET`-only (`net.c:90`) roughly halves reachable peers. Replace the +hard-coded `sockaddr_in`/`inet_pton` with `sockaddr_storage` + `getaddrinfo`-style +handling; detect a literal `:` to choose `AF_INET6`. No conceptual ABI change +(peer addresses are already strings). + +- **Effort:** ~0.5 day. **Risk:** low. **Tests:** local seed bound to `[::1]`. + +## §4. Fast Extension (BEP-6) + Extension Protocol (BEP-10) + +**Why:** **Reject Request** turns §1's timeout-based recovery into *instant* +recovery — a peer that won't serve a block says so explicitly. `HAVE_ALL`/ +`HAVE_NONE` populate availability in one message. BEP-10 is the negotiation base +real peers expect (and the prerequisite for PEX/metadata later). + +**Design:** +- Handshake reserved bits: Fast-Ext (`reserved[7] |= 0x04`), Ext-Protocol + (`reserved[5] |= 0x10`); use a feature only if the peer also set its bit. +- Inbound in `handle_control`: `HAVE_ALL (0x0E)`/`HAVE_NONE (0x0F)` → + availability; `REJECT_REQUEST (0x10)` → look up in the §1 in-flight map, + decrement, re-arm immediately; `SUGGEST_PIECE`/`ALLOWED_FAST` → optional + selection bias (Allowed-Fast lets us request while choked). +- Outbound: send `HAVE_NONE` on connect when both support Fast; add a minimal + bencode reader for the BEP-10 extended handshake. + +- **Effort:** ~1.5 days (incl. minimal bencode). **Risk:** low-medium. **Tests:** + mock peer advertises Fast, sends `HAVE_ALL` then `REJECT_REQUEST`; assert + instant re-request (faster than the timeout) and correct completion. + +## §5. io_uring receive path (REVIEW P2b) + +**Why:** near-syscall-free, genuine zero-copy receive at multi-GB/s. **Reasons +it's not first:** kernel-version gating (multishot recv + provided buffers +≈ 5.19+), seccomp/sandbox policies that disable io_uring, and that it's a +throughput-only win whose payoff appears only after the other bottlenecks are +gone. So: opt-in, with the epoll path as permanent fallback. + +**Design:** a `transport_uring` selected per loop at runtime (probe at startup; +fall back to `transport_tcp`). Multishot `IORING_OP_RECV` with a provided-buffer +ring feeds the §0a push-model parser. Stretch: register the per-loop arena with +`io_uring_register_buffers` and use fixed-buffer reads into the destination slot, +removing the residual `memcpy` (REVIEW P1b). Batch request writes via +`IORING_OP_SEND`. + +- **Dependency:** `liburing`, optional + feature-gated. **Effort:** ~3–4 days. + **Risk:** medium (kernel variance). **Tests:** full suite against both backends + via a config knob; CI selects by kernel capability. + +## §6. In-library SHA hashing (REVIEW P2c) + +**Why:** with credit backpressure, **wire speed is hostage to hash speed**, and +hashing currently runs in Python. Make it an opt-in in-library mode (the harness +path stays default). + +**Design:** +- **Engine-level `HashPool`** of N worker threads (shared across loops). +- **Config/API:** `verify_in_library` flag, `worker_threads`, + `torrent_set_piece_hashes(...)`, and a results channel (`{piece, ok}` ring the + consumer drains; verified bytes via in-lib piece buffer / write-fd / callback). +- **Flow:** the loop copies each block from its slot into a pooled per-piece + assembly buffer and **releases the slot immediately** (wire stays full; slots + never leave the loop). When a piece is complete, enqueue a hash job; a worker + runs SHA-1 and reports the result. +- **SHA-1 with SHA-NI:** runtime-dispatched (`__builtin_cpu_supports("sha")`) + with a portable fallback. Reused by MSE key derivation (§7). + +- **Effort:** ~3 days. **Risk:** medium (new threading/memory subsystem). + **Tests:** enable in-lib verification against a local seed; assert all results + `ok` and bytes match; negative test with a wrong expected hash (retry path). + +## §7. MSE / PE encryption + +**Why:** many real-swarm peers refuse plaintext. A `transport_mse` wrapping an +inner transport (§0b) performs the MSE handshake as initiator before BT bytes +flow: DH key exchange (768-bit MSE prime) → obfuscated handshake +(`HASH('req1',S)`, `HASH('req2',SKEY) xor HASH('req3',S)`, `SKEY=info_hash`, +random pad) → cipher negotiation (`crypto_provide`/`select`, RC4 or plaintext, +keys from `HASH('keyA'/'keyB',S,SKEY)`, RC4-drop-1024) → transparent +encrypt/decrypt so `proto.c` is unchanged. Primitives: RC4 (~15 lines), SHA-1 +(reuse §6), bignum modexp (vendored small bigint or `libcrypto` BN — decision +below). Config: `enc_policy` (plaintext-only / prefer / require). + +- **Effort:** ~4–5 days. **Risk:** high (security-sensitive). **Tests:** + libtorrent seed with encryption forced; assert completion; cross-check RC4 and + plaintext selection; run under ASan. + +## §8. µTP (largest item) + +**Why:** a large fraction of real peers are µTP-only (BitTorrent over UDP, +LEDBAT). A `transport_utp` presents a reliable in-order byte stream to the parser +over a UDP socket: packet types `ST_SYN/DATA/STATE/FIN/RESET` with +`seq_nr`/`ack_nr`/timestamps + selective-ack; LEDBAT congestion control (one-way +delay target ~100 ms, AIMD-style window — the core complexity); retransmission +timers, send/recv windows, in-order reassembly; UDP fd registered via the +transport's `want_fds`, timers driven off the loop tick; connect as initiator. + +- **Effort:** ~1–2 weeks. **Risk:** high (timing/congestion correctness). + **Tests:** libtorrent seed forced to µTP-only over `127.0.0.1`; verify bytes + and LEDBAT back-off under induced delay. + +--- + +## Suggested ordering & milestones + +1. **§0a reactor restructure + §0b transport abstraction** (keystones; share the + push-model parser). Land with multi-peer/multi-torrent tests green. +2. **§1 picker + in-flight + timeout** and **§2 endgame/CANCEL** — the core that + makes multi-peer actually correct and fast. +3. **§3 IPv6** + **§4 Fast Extension/BEP-10** — quick wins; Fast-Ext upgrades §1 + recovery from timeout to instant. +4. **§6 in-library hashing** — removes the Python hash ceiling (independent). +5. **§5 io_uring** — throughput, on the transport abstraction. +6. **§7 MSE** — reach; reuses SHA-1 from §6. +7. **§8 µTP** — biggest; reach; last. + +§5/§6 are independent and can be parallelized; §7 and §8 both sit on §0b. + +## Decisions needed before starting + +- **Loop count / sizing:** fixed (e.g. `min(ncpu, N)`) vs. configurable; pinning + loop threads to cores? Default torrent→loop assignment policy (least-loaded)? +- **ABI shape:** engine-centric API + thin single-peer compat wrapper — confirm + the new surface and how the Python harness/consumer drains completions across + many connections. +- **Cross-loop sharding for mega-torrents:** ship affinity-only first and add + sharded picker later, or design the picker for optional sharding up front? +- **Crypto dependency (§7):** vendor minimal bigint + RC4 + SHA-1, or link + `libcrypto` (build dep + RC4 legacy-provider wrinkle)? +- **liburing (§5):** optional + runtime probe + epoll fallback (recommended). +- **In-library hashing output (§6):** results-ring + in-lib piece buffers only, + or also a write-to-fd path so the C side can persist without the harness? +- **Minimum kernel/libc baseline** for io_uring and µTP timers. diff --git a/README.md b/README.md new file mode 100644 index 0000000..9c72237 --- /dev/null +++ b/README.md @@ -0,0 +1,176 @@ +# torrent-peer + +A fast, BitTorrent-wire-compatible **leech peer** written in C, driven by a +Python **test harness**. The normal `.torrent` path parses metainfo locally and +uses the neighboring `torrent-tracker` library for DHT get_peers and HTTP/UDP +tracker announces; libtorrent remains in the tests for seed fixtures and as a +magnet-metadata fallback. + +Milestone 1 scope: download from connected peers as fast as possible. No +seeding, PEX, or native magnet metadata resolution yet. + +The peer **never hashes or persists data**. It only fetches the blocks the +harness asks for and hands the raw bytes back. The harness owns *what* to +download, reassembles pieces, and verifies SHA-1. + +## How data moves (and why it's fast) + +``` + enqueue_piece(i) request pipeline (BDP-deep) + harness ───────────────────▶ C peer ───────────────────────────▶ remote peer + ▲ ▲ │ net thread (epoll + TCP) + │ │ ready_ring (SPSC) │ recv() steered directly into arena slots + │ └──────────────────────────┘ + │ block_desc{piece,begin,len,slot} ← zero-copy: payload lives in arena + │ + └────── free_ring (SPSC) ── peer_release_slot(slot) ─────────────┐ + return spent slots = credit that lets new requests go out ┘ +``` + +- **SPSC lock-free rings (LMAX Disruptor pattern).** The net thread is the sole + producer of completed blocks (`ready_ring`) and sole consumer of returned + slots (`free_ring`); the harness thread is the mirror. Head/tail counters sit + on separate cache lines with acquire/release ordering — no locks on the hot + path. See `src/ring.h`. +- **Pre-allocated, page-faulted arena.** `num_slots × 16 KiB` of memory is + allocated and touched once up front (`src/arena.c`); there is **zero per-block + allocation** in steady state. The harness wraps it once as a `memoryview` and + reads each block zero-copy (`harness/peer_ffi.py`). +- **Payload steering.** The parser reads a piece message's 13-byte header from a + small staging buffer, then `recv()`s the bulk payload **straight into its + arena slot** — the only bytes copied are whatever were prefetched alongside + the header (`src/proto.c::rd_take_body`). +- **Credit-based flow control.** A block consumes one free slot on arrival; + requests are only issued while `outstanding < free_slots`, so a slow consumer + naturally throttles the wire without dropping data (`src/scheduler.c`). +- **Adaptive pipeline depth.** The in-flight target tracks the bandwidth-delay + product, `target ≈ 2 × rate × min_rtt / 16 KiB`, clamped to + `[32, max_pipeline]`. It uses the *minimum* observed request→block RTT (not an + average) so the estimate reflects the unloaded path instead of feeding back + the pipeline's own queueing delay (`src/scheduler.c`). +- **Request timeouts + re-queue.** Every issued block is tracked with its send + time; a block that goes unanswered past `request_timeout_ms` is re-queued and + re-issued ahead of fresh requests, so a silently dropped request can't stall + the download. Incoming blocks are matched against outstanding requests — an + unsolicited/duplicate block is drained and dropped without consuming a slot or + disturbing the credit accounting (`src/reqtab.c`, `src/proto.c`). + +## Piece selection + +The harness supplies a **priority vector — one `uint8_t` per piece** — via +`peer_set_priorities()` / `peer_set_priority()`. At each piece boundary the peer +picks the highest-priority piece that + +1. the **remote peer actually has** (tracked from its `bitfield`/`have` + messages — the peer never assumes availability), and +2. has **not already been fully requested**, + +breaking ties toward the lowest piece index. Priority `0` means "skip". Because +selection re-reads the vector live, the harness can implement any scheme by +rewriting priorities over time — sequential, rarest-first, or a deadline ramp +(e.g. fetch episode 1 first, then rarest-first, while slowly raising episode 2's +priorities so it lands inside 24 minutes) — without any peer-side changes. A +piece that fails verification is re-armed with `peer_request_piece()`. + +This replaces a fixed in-order schedule, which stalled whenever the connected +peer lacked the next piece. Selection is one O(num_pieces) scan per *piece* (not +per block); `src/scheduler.c`. + +io_uring zero-copy receive is the natural next upgrade and would drop into +`src/net.c` without touching the protocol parser or the arena handoff. + +## Layout + +| Path | Role | +|------|------| +| `include/peer.h` | public C ABI (the only surface ctypes binds to) | +| `src/peer.c` | lifecycle, helpers, control-plane piece queue | +| `src/net.c` | TCP connect, socket tuning, epoll loop, outgoing buffer | +| `src/proto.c` | handshake + wire framing + payload steering | +| `src/scheduler.c` | piece selection, adaptive depth, timeouts | +| `src/reqtab.c` | in-flight request table (match + timeout) | +| `src/ring.c/.h` | SPSC rings | +| `src/arena.c` | aligned, pre-faulted block arena | +| `harness/peer_ffi.py` | ctypes bindings + zero-copy arena view | +| `harness/torrent_meta.py` | local bencoded `.torrent` parser | +| `harness/tracker_ffi.py` | ctypes bindings for `../torrent-tracker` DHT/tracker helpers | +| `harness/harness.py` | tracker discovery + reassembly + SHA-1 verify | +| `tests/test_localseed.py` | end-to-end test against a local libtorrent seed | + +## Build + +```sh +cmake -S . -B build -DCMAKE_BUILD_TYPE=Release +cmake --build build # -> build/libtorrentpeer.so +``` + +Options: `-DPEER_NATIVE=OFF` (portable build, no `-march=native`), +`-DPEER_ASAN=ON` (AddressSanitizer/UBSan). + +## Test + +The end-to-end test creates a random file, seeds it with libtorrent on +localhost, downloads every piece through the C peer, and asserts the bytes match +(it builds the library automatically if needed): + +```sh +python tests/test_localseed.py # standalone +# or, if pytest is available: python -m pytest tests/ +``` + +Run the full path under ASan: + +```sh +cmake -S . -B build-asan -DPEER_ASAN=ON && cmake --build build-asan +ASAN_OPTIONS=detect_leaks=0 \ + LD_PRELOAD=$(gcc -print-file-name=libasan.so) \ + python tests/test_localseed.py +``` + +For offline interoperability against multiple seed clients, use the Docker +Compose harness in `interop/`. It creates a private trackerless fixture torrent, +starts libtorrent, Transmission, and aria2 seeders on an internal-only Docker +network, then verifies that this engine can download and hash-check the fixture +from each one: + +```sh +docker compose -f interop/docker-compose.yml up --build \ + --abort-on-container-exit --exit-code-from runner +``` + +## Use against a real torrent + +```sh +# explicit peer (skip tracker): +python harness/harness.py file.torrent --peer 1.2.3.4:51413 -o out.bin +# or discover peers via DHT and the torrent's HTTP/UDP trackers: +python harness/harness.py file.torrent -o out.bin +``` + +Tunables: `--slots N` (arena depth, 16 KiB each), `--pipeline N` (cap on +outstanding requests; the live depth adapts to the BDP under this cap), +`--timeout SECONDS`. Per-block request timeout and an optional `SO_RCVBUF` +override are `peer_config` fields (`request_timeout_ms`, `recv_buffer_bytes`). + +## Status + +Verified byte-for-byte on local-seed transfers up to 256 MiB; ~400+ MB/s over +loopback with the arena (16 MiB) far smaller than the file, i.e. with +credit-based backpressure fully engaged. Clean under ASan/UBSan. The test suite +covers full download + byte verification, a **partial seed** (peer downloads +only the pieces the remote has), **priority ordering**, and — via a raw-socket +mock peer — **request-timeout recovery** and **unsolicited-block rejection**. + +### Known gaps / next steps + +- Real-swarm auto-discovery is still hit-or-miss: many announced peers are + unreachable, behind NAT, or only usable with a different transport/encryption + mode. Use `harness/swarm_download.py --encryption` and `--utp` to choose the + dial mode. +- Single peer / single net thread. Aggregate throughput needs a multi-peer + connection manager with a shared piece picker (enables endgame mode + CANCEL). +- io_uring registered buffers + multishot recv would give a true zero-copy, + near-syscall-free receive path (drops into `src/net.c`). +- Optional in-library SHA-NI hashing on a worker pool, so wire speed isn't held + hostage to Python hash throughput at multi-GB/s. +- Wider peer reach: µTP, MSE/PE encryption, IPv6, Fast Extension (BEP-6). diff --git a/REVIEW.md b/REVIEW.md new file mode 100644 index 0000000..7d4f041 --- /dev/null +++ b/REVIEW.md @@ -0,0 +1,196 @@ +# torrent-peer — Performance & Correctness Review (work items) + +Audience: an AI/engineer who will implement fixes. Each item has a location, the +problem, why it matters for a high-performance BitTorrent client, and a concrete +fix direction. Items are grouped and roughly ordered by value. + +Context: this library is currently a **single-peer, leech-only, TCP-only** block +fetcher. The C peer connects to one peer, runs the request pipeline, and hands +raw blocks to a Python harness (which owns metadata/tracker/reassembly/SHA-1). +Several of the highest-value performance items live in scope the README has +explicitly deferred (multi-peer, µTP, encryption). They are still listed here +because the question is about hitting peak performance. + +--- + +## P0 — correctness / hard stalls (fix first) + +### C1. No per-request timeout → permanent stall on a dropped request +- Where: `src/scheduler.c` (request issue), `src/proto.c:247` (`outstanding--`), + `src/peer.c:148` (`peer_request_piece`). +- Problem: when a request is issued, `outstanding` is incremented and the piece + is marked `requested[i]=1` (`scheduler.c:38`). There is no timeout. If the + peer accepts the request but never sends the block (silent drop, or choked + without Fast-Extension Reject), `outstanding` stays high and `requested[i]` + stays 1 forever. The piece never completes and never recovers — the harness + only re-arms on hash *failure*, but it never receives bytes to hash. +- Why it matters: a single misbehaving/slow peer hangs the whole download. +- Fix: track per-block in-flight requests with a deadline (issue time + RTT-based + timeout). On expiry, decrement `outstanding`, clear the block's requested + state, and re-queue it. Add an internal timer tick (the net loop already wakes + every `RUNNING_TICK_MS`). Expose a configurable timeout in `peer_config`. + +### C2. Unsolicited / unmatched piece messages corrupt flow control +- Where: `src/proto.c:223-251` (`RS_PIECE_HDR` pops a free slot; `RS_PIECE_BODY` + pushes to harness and decrements `outstanding`). +- Problem: there is no check that the received `(piece, begin, len)` matches an + outstanding request. A buggy or malicious peer can push blocks you never + requested: this consumes arena slots, hands garbage to the harness, and + decrements `outstanding` incorrectly (`proto.c:247`), desyncing the + `outstanding <= free_slots` credit invariant. +- Why it matters: correctness + DoS resistance; also required before C1's + in-flight tracking can be trusted. +- Fix: maintain a set/multiset of outstanding `(piece, begin)` requests (a small + hash set keyed on piece*blocks+block index works). On a piece message, look it + up; if absent, drop the payload without consuming a slot and without touching + `outstanding`. Only decrement `outstanding` for a matched request. + +--- + +## P1 — performance ceilings in the current single-peer path + +### P1a. Slot release is not event-driven (2 ms polling latency) +- Where: `src/peer.c:198` (`peer_release_slot`), `src/net.c:30,154` + (`RUNNING_TICK_MS = 2`). +- Problem: `peer_release_slot` only pushes to `free_ring`; it does **not** poke + `ctrl_efd`. When the pipeline has drained (harness is the bottleneck), request + refill waits for the 2 ms epoll timeout instead of firing immediately on + returned credit. While blocks are actively arriving this is masked (EPOLLIN + drives `scheduler_tick`), but it is a real latency floor on the credit→request + loop whenever the wire idles on slots. +- Fix: signal `ctrl_efd` from `peer_release_slot`, coalesced so a burst of + releases costs at most one wakeup (e.g. only write the eventfd if a "needs + wake" flag transitions). Keep the periodic tick as a fallback. + +### P1b. "Zero-copy steering" mostly copies at high throughput +- Where: `src/net.c:29` (`RECV_STAGING_CAP = 256 KiB`), `src/proto.c:99-120` + (`rd_take_body`). +- Problem: `rd_refill` reads up to 256 KiB into staging; `rd_take_body` first + `memcpy`s the already-staged portion of the body and only `recv()`s the + remainder directly into the slot. When the socket is fast (kernel returns + 256 KiB at once — the regime that matters), most of each block is already in + staging and is copied, not steered. The README's "only bytes copied are those + prefetched alongside the header" is inverted under load. +- Why it matters: the advertised zero-copy property does not hold at speed; cost + grows toward multi-GB/s. +- Fix (small): keep it; the hot 16 KiB memcpy is cheap at a few hundred MB/s. +- Fix (real): use `readv`/`recvmmsg` scatter directly into slots, or io_uring + registered buffers (see P2b). That removes the copy *and* cuts syscalls. + +### P1c. `SO_RCVBUF` pinned to 8 MiB disables autotuning → caps high-BDP throughput +- Where: `src/net.c:96-97`, `SO_RCVBUF_BYTES = 8<<20`. +- Problem: manually setting `SO_RCVBUF` disables kernel receive-window + autotuning and clamps the window. On a long-fat path (e.g. 1 Gbps × 100 ms + ≈ 12.5 MB BDP) this throttles below line rate. +- Fix: by default do **not** set `SO_RCVBUF` (let autotuning scale). Make it an + opt-in tunable for low-latency LAN cases. Optionally set `TCP_CONGESTION=bbr` + and consider `TCP_QUICKACK`. + +### P1d. Adaptive pipeline depth +- Where: `src/peer.c:17` (`DEFAULT_PIPELINE = 2048`), `src/scheduler.c:46`. +- Problem: pipeline depth is a fixed cap, not tracking the measured + bandwidth-delay product. Too small under-fills on high RTT; too large wastes + arena. README acknowledges this. +- Fix: measure RTT (request→first-byte) and achieved rate (already sampled in + `update_rate`, `net.c:78`) and size in-flight bytes to ~BDP, clamped by arena. + +### P1e. One piece in flight at a time starves small-piece torrents +- Where: `src/scheduler.c:53-70` (`have_cur_piece` gate). +- Problem: the scheduler fully requests `cur_piece` before selecting the next. + For small piece sizes the in-flight window collapses toward one piece's worth + of blocks, starving the pipeline. +- Fix: allow outstanding requests to span multiple pieces; select a new piece + whenever the in-flight window has room, not only at piece completion. + +### P1f. Data race on `outstanding` (and `rate_bps`) +- Where: `src/peer.c:221` reads `hh->outstanding`; net thread writes it at + `src/proto.c:247` / `src/scheduler.c:68`. +- Problem: non-atomic cross-thread read = C11 data race (UB). Benign in + practice; `rate_bps` is similarly racy and acknowledged. +- Fix: make `outstanding` an `atomic_uint` with relaxed ordering, or snapshot it + into an atomic for status reads. + +--- + +## P2 — features required to actually hit peak performance + +### P2a. Multi-peer / multi-connection (biggest lever) +- Problem: a single TCP peer rarely saturates a fast link; BitTorrent throughput + is aggregate across many peers. The current thread-per-peer model + (`peer_start` spawns one net thread, `peer.c:115`) does not scale to + hundreds/thousands of sockets. +- Fix: introduce an event-loop pool that shards many sockets across a small set + of epoll/io_uring loops. The per-handle arena+rings already isolate state + cleanly; the work is a connection manager + a shared piece-picker across peers. +- Depends on / enables: endgame mode (P2d), rarest-first across the swarm. + +### P2b. io_uring (registered buffers + multishot recv) +- Where: replaces the recv path in `src/net.c`; README flags this. +- Problem/fix: register the arena once, use multishot recv to deliver blocks + with a near-zero syscall hot path and genuine zero-copy (fixes P1b). The + protocol parser and arena handoff above it stay unchanged. + +### P2c. In-library, multi-threaded, SHA-NI hashing +- Problem: because of the (good) credit backpressure, **wire speed is a hostage + to hash speed**, and hashing currently runs in Python in the harness. At + multi-GB/s, SHA-1 verification in Python becomes the real throttle and slots + return slowly. +- Fix: offer optional in-library piece verification using hardware SHA (SHA-NI) + on a small worker-thread pool, releasing slots as soon as a block is hashed + into its piece buffer. Keep the harness path for v2/custom schemes. + +### P2d. Request timeouts → endgame mode + CANCEL +- Where: builds on C1; `CANCEL` is parsed but never sent (`proto.c` only handles + inbound control in `handle_control`). +- Fix: near completion, request the last outstanding blocks from multiple peers + and send `CANCEL` to the losers. Requires multi-peer (P2a) to be fully useful, + but the CANCEL send path and "duplicate request allowed near end" logic belong + here. + +### P2e. Wider peer pool: µTP, MSE/PE encryption, IPv6, Fast Extension (BEP-6) +- Where: IPv4-only today (`AF_INET`, `net.c:90`); no extension handshake. +- Problem: µTP and MSE encryption are how you reach many real-swarm peers at + all (→ aggregate throughput). Fast Extension's **Reject Request** is directly + relevant: without it a peer that won't serve a request silently drops it — + exactly the case C1 must otherwise recover from via timeout. IPv4-only halves + reachability. +- Fix: implement BEP-6 (at least Reject Request, Have All/None, Allowed Fast), + BEP-10 extension protocol, then µTP and MSE as larger efforts. + +--- + +## P3 — smaller / build + +### P3a. Build flags +- Where: `CMakeLists.txt`. +- Fix: enable LTO (`-flto` / `INTERPROCEDURAL_OPTIMIZATION`). Consider per-call + `-mtune`. Keep `-march=native` opt-in as it already is. + +### P3b. Huge pages + NUMA-local arena +- Where: `src/arena.c`. +- Fix: for the multi-GB/s regime, back the arena with huge pages + (`MAP_HUGETLB`/THP via `madvise(MADV_HUGEPAGE)`) to cut TLB pressure, and + allocate it NUMA-local to the net thread (or interleaved) on multi-socket + hosts. + +### P3c. `pri_lock` held across the full O(num_pieces) scan +- Where: `src/scheduler.c:28-39` (`select_next_piece`). +- Problem: the lock is held for the whole scan, contending with harness priority + updates. Low impact (updates are rare). +- Fix: snapshot or use a finer structure (bucketed/heap) only if profiling shows + it matters; continuously changing priorities limits the payoff. + +--- + +## Things that are already good (do not regress) + +- SPSC rings (`src/ring.h`): power-of-two mask, free-running counters, + cache-line-isolated head/tail, correct acquire/release pairing. +- Pre-faulted, page-aligned arena with no hot-path allocation; correct + over-alignment handling in `peer_create` (`src/peer.c:47-54`). +- Credit-based backpressure invariant `outstanding <= free_slots` + (`src/scheduler.c:50`) — every arriving block is guaranteed a slot. +- Live priority-vector piece selection (`src/scheduler.c:25`) decouples policy + from mechanism. +- Resumable incremental parser, batched request writes, `TCP_NODELAY`, + ASan/UBSan build, end-to-end test. diff --git a/harness/engine_ffi.py b/harness/engine_ffi.py new file mode 100644 index 0000000..ac3c99e --- /dev/null +++ b/harness/engine_ffi.py @@ -0,0 +1,202 @@ +""" +ctypes bindings for the multi-peer engine ABI (include/engine.h). + +The engine owns a pool of event-loop threads; torrents are pinned to a loop and +every connection of a torrent lives there. Each loop has its own arena, so a +delivered block names both the loop and the slot. We wrap each loop's arena once +as a zero-copy ``memoryview`` and slice per block; returning the slot via +``release()`` is what lets the engine issue new requests (credit-based flow +control). +""" +from __future__ import annotations + +import ctypes as C +import os + +# peer_state / peer_error mirrors of include/engine.h +STATE_IDLE, STATE_CONNECTING, STATE_HANDSHAKE, STATE_CHOKED, \ + STATE_RUNNING, STATE_STOPPED, STATE_ERROR = range(7) +STATE_NAMES = ["IDLE", "CONNECTING", "HANDSHAKE", "CHOKED", + "RUNNING", "STOPPED", "ERROR"] +ERROR_NAMES = ["OK", "CONNECT", "HANDSHAKE", "CLOSED", "PROTOCOL", "IO", "NOMEM"] + +BLOCK_SIZE = 16384 + + +class EngineConfig(C.Structure): + _fields_ = [ + ("loop_count", C.c_uint32), + ("slots_per_loop", C.c_uint32), + ("max_pipeline", C.c_uint32), + ("request_timeout_ms", C.c_uint32), + ("recv_buffer_bytes", C.c_uint32), + ("encryption", C.c_uint32), + ("utp", C.c_uint32), + ("connect_timeout_ms", C.c_uint32), + ("fallback", C.c_uint32), + ] + + +class EngineBlock(C.Structure): + _fields_ = [ + ("torrent", C.c_uint32), + ("piece", C.c_uint32), + ("begin", C.c_uint32), + ("len", C.c_uint32), + ("loop", C.c_uint32), + ("slot", C.c_uint32), + ] + + +class TorrentStatus(C.Structure): + _fields_ = [ + ("state", C.c_int32), + ("error", C.c_int32), + ("bytes_received", C.c_uint64), + ("blocks_received", C.c_uint64), + ("peers", C.c_uint32), + ("peers_connected", C.c_uint32), + ("peers_failed", C.c_uint32), + ("outstanding", C.c_uint32), + ("free_slots", C.c_uint32), + ("pipeline_target", C.c_uint32), + ("rate_bps", C.c_double), + ("rtt_min_ms", C.c_double), + ] + + +def _default_lib_path() -> str: + here = os.path.dirname(os.path.abspath(__file__)) + cand = [ + os.path.join(here, "..", "build", "libtorrentpeer.so"), + os.path.join(here, "..", "build", "lib", "libtorrentpeer.so"), + ] + for p in cand: + if os.path.exists(p): + return os.path.abspath(p) + return os.path.abspath(cand[0]) + + +def _load(lib_path: str | None) -> C.CDLL: + lib = C.CDLL(lib_path or _default_lib_path()) + lib.engine_create.restype = C.c_void_p + lib.engine_create.argtypes = [C.POINTER(EngineConfig)] + lib.engine_destroy.restype = None + lib.engine_destroy.argtypes = [C.c_void_p] + lib.engine_add_torrent.restype = C.c_int32 + lib.engine_add_torrent.argtypes = [ + C.c_void_p, C.POINTER(C.c_uint8), C.POINTER(C.c_uint8), + C.c_uint64, C.c_uint64, C.c_uint32, + ] + lib.engine_add_peer.restype = C.c_int + lib.engine_add_peer.argtypes = [C.c_void_p, C.c_uint32, C.c_char_p, C.c_uint16] + lib.engine_set_priorities.restype = C.c_int + lib.engine_set_priorities.argtypes = [ + C.c_void_p, C.c_uint32, C.POINTER(C.c_uint8), C.c_uint32] + lib.engine_set_priority.restype = C.c_int + lib.engine_set_priority.argtypes = [C.c_void_p, C.c_uint32, C.c_uint32, C.c_uint8] + lib.engine_request_piece.restype = C.c_int + lib.engine_request_piece.argtypes = [C.c_void_p, C.c_uint32, C.c_uint32] + lib.engine_poll_ready.restype = C.c_uint32 + lib.engine_poll_ready.argtypes = [C.c_void_p, C.POINTER(EngineBlock), C.c_uint32] + lib.engine_release_slot.restype = None + lib.engine_release_slot.argtypes = [C.c_void_p, C.c_uint32, C.c_uint32] + lib.engine_wait.restype = C.c_int + lib.engine_wait.argtypes = [C.c_void_p, C.c_int] + lib.engine_arena_base.restype = C.c_void_p + lib.engine_arena_base.argtypes = [C.c_void_p, C.c_uint32] + lib.engine_arena_bytes.restype = C.c_uint64 + lib.engine_arena_bytes.argtypes = [C.c_void_p, C.c_uint32] + lib.engine_loop_count.restype = C.c_uint32 + lib.engine_loop_count.argtypes = [C.c_void_p] + lib.engine_torrent_status.restype = None + lib.engine_torrent_status.argtypes = [C.c_void_p, C.c_uint32, C.POINTER(TorrentStatus)] + return lib + + +class Engine: + """Pythonic wrapper around one engine instance (a pool of loops).""" + + def __init__(self, cfg: EngineConfig | None = None, + lib_path: str | None = None, poll_batch: int = 1024): + self._lib = _load(lib_path) + self._e = self._lib.engine_create(C.byref(cfg) if cfg else None) + if not self._e: + raise RuntimeError("engine_create failed") + + # One zero-copy memoryview per loop arena. + self.nloops = self._lib.engine_loop_count(self._e) + self._arenas = [] + self.arenas = [] + for i in range(self.nloops): + base = self._lib.engine_arena_base(self._e, i) + nbytes = self._lib.engine_arena_bytes(self._e, i) + buf = (C.c_char * nbytes).from_address(base) + self._arenas.append(buf) + self.arenas.append(memoryview(buf).cast("B")) + + self._batch = poll_batch + self._blocks = (EngineBlock * poll_batch)() + + def add_torrent(self, info_hash: bytes, peer_id: bytes, piece_length: int, + total_size: int, num_pieces: int) -> int: + ih = (C.c_uint8 * 20).from_buffer_copy(info_hash) + pid = (C.c_uint8 * 20).from_buffer_copy(peer_id) + tid = self._lib.engine_add_torrent(self._e, ih, pid, piece_length, + total_size, num_pieces) + if tid < 0: + raise RuntimeError("engine_add_torrent failed") + return tid + + def add_peer(self, torrent_id: int, ip: str, port: int) -> None: + if self._lib.engine_add_peer(self._e, torrent_id, ip.encode(), port) != 0: + raise RuntimeError("engine_add_peer failed") + + def set_priorities(self, torrent_id: int, priorities) -> None: + buf = bytes(priorities) + arr = (C.c_uint8 * len(buf)).from_buffer_copy(buf) + if self._lib.engine_set_priorities(self._e, torrent_id, arr, len(buf)) != 0: + raise ValueError("set_priorities: length must equal num_pieces") + + def set_priority(self, torrent_id: int, piece_index: int, priority: int) -> None: + if self._lib.engine_set_priority(self._e, torrent_id, piece_index, priority) != 0: + raise ValueError(f"set_priority({piece_index}) out of range") + + def request_piece(self, torrent_id: int, piece_index: int) -> None: + if self._lib.engine_request_piece(self._e, torrent_id, piece_index) != 0: + raise ValueError(f"request_piece({piece_index}) out of range") + + def poll_ready(self): + """Return a list of EngineBlock for completed blocks (may be empty).""" + n = self._lib.engine_poll_ready(self._e, self._blocks, self._batch) + return [self._blocks[i] for i in range(n)] + + def block_data(self, loop: int, slot: int, length: int) -> memoryview: + off = slot * BLOCK_SIZE + return self.arenas[loop][off:off + length] + + def release(self, loop: int, slot: int) -> None: + self._lib.engine_release_slot(self._e, loop, slot) + + def wait(self, timeout_ms: int) -> int: + return self._lib.engine_wait(self._e, timeout_ms) + + def status(self, torrent_id: int) -> TorrentStatus: + st = TorrentStatus() + self._lib.engine_torrent_status(self._e, torrent_id, C.byref(st)) + return st + + def close(self) -> None: + if self._e: + for mv in self.arenas: + mv.release() + self.arenas = [] + self._arenas = [] + self._lib.engine_destroy(self._e) + self._e = None + + def __enter__(self): + return self + + def __exit__(self, *exc): + self.close() diff --git a/harness/harness.py b/harness/harness.py new file mode 100644 index 0000000..b7029be --- /dev/null +++ b/harness/harness.py @@ -0,0 +1,202 @@ +""" +Test/driver harness for the C peer. + +The Python harness parses .torrent metadata and can discover peers through the +sibling torrent-tracker C library's DHT and HTTP/UDP tracker helpers. The C peer +does the fast part: pull the requested blocks from one peer. This harness owns +what to download, reassembles pieces, and verifies SHA-1 hashes -- the peer +never hashes or persists anything. +""" +from __future__ import annotations + +import argparse +import hashlib +import os +import secrets +import time + +from peer_ffi import Peer, PeerConfig, STATE_ERROR, STATE_NAMES, ERROR_NAMES +from torrent_meta import Metadata, load_metadata, load_torrent +from tracker_ffi import DHTClient, TrackerClient + + +def make_peer_id() -> bytes: + return b"-PC0001-" + secrets.token_bytes(12) + + +def discover_peers(torrent_path: str, max_wait: float = 20.0) -> list[tuple[str, int]]: + """Discover peer endpoints through DHT first, then torrent trackers.""" + tf = load_torrent(torrent_path) + peer_id = make_peer_id() + key = secrets.randbits(32) + deadline = time.time() + max_wait + seen: set[tuple[str, int]] = set() + + dht_budget = max(0.5, min(6.0, max_wait / 2.0)) + dht_result = DHTClient().lookup(tf.metadata.info_hash, timeout=dht_budget) + seen.update(dht_result.peers) + if seen: + return sorted(seen) + + client = TrackerClient() + for url in tf.trackers: + if time.time() >= deadline: + break + result = client.announce( + url, tf.metadata, peer_id, port=6881, key=key, numwant=50, + event="started", timeout=max(0.5, deadline - time.time())) + if result.ok: + seen.update(result.peers) + if seen: + break + return sorted(seen) + + +class Downloader: + def __init__(self, meta: Metadata, *, peer_id: bytes | None = None, + num_slots: int = 0, max_pipeline: int = 0, + request_timeout_ms: int = 0, recv_buffer_bytes: int = 0, + lib_path: str | None = None): + self.meta = meta + cfg = PeerConfig() + cfg.info_hash[:] = meta.info_hash + cfg.peer_id[:] = peer_id or make_peer_id() + cfg.piece_length = meta.piece_length + cfg.total_size = meta.total_size + cfg.num_pieces = meta.num_pieces + cfg.num_slots = num_slots + cfg.max_pipeline = max_pipeline + cfg.request_timeout_ms = request_timeout_ms + cfg.recv_buffer_bytes = recv_buffer_bytes + self.peer = Peer(cfg, lib_path) + + self.buffers: list[bytearray | None] = [None] * meta.num_pieces + self.received = [0] * meta.num_pieces + self.done = bytearray(meta.num_pieces) + self.done_count = 0 + + def download(self, ip: str, port: int, pieces=None, priorities=None, + timeout: float = 60.0, progress_every: float = 1.0, + output: str | None = None) -> bytes | None: + meta = self.meta + pieces = list(pieces) if pieces is not None else list(range(meta.num_pieces)) + + out_fh = open(output, "wb") if output else None + if out_fh: + out_fh.truncate(meta.total_size) + + # Build the priority vector: caller-supplied scheme, or uniform "1" over + # the wanted pieces (0 = not wanted). The peer masks this with what the + # remote actually has, so a peer missing a piece is simply skipped. + prio = bytearray(priorities) if priorities is not None \ + else bytearray(meta.num_pieces) + for i in pieces: + self.buffers[i] = bytearray(meta.piece_len(i)) + if priorities is None: + prio[i] = 1 + + self.peer.start(ip, port) + self.peer.set_priorities(prio) + + want = len(pieces) + deadline = time.time() + timeout + last_print = 0.0 + last_progress_bytes = 0 + last_progress_time = time.time() + + while self.done_count < want: + st = self.peer.status() + if st.state == STATE_ERROR: + raise RuntimeError(f"peer error: {ERROR_NAMES[st.error]}") + + descs = self.peer.poll_ready() + if not descs: + self.peer.wait(100) + now = time.time() + if st.bytes_received != last_progress_bytes: + last_progress_bytes = st.bytes_received + last_progress_time = now + if now > deadline and now - last_progress_time > timeout: + raise TimeoutError( + f"stalled: {self.done_count}/{want} pieces, " + f"state={STATE_NAMES[st.state]}") + if now - last_print >= progress_every: + last_print = now + print(f" {self.done_count}/{want} pieces " + f"{st.rate_bps/1e6:.1f} MB/s " + f"outstanding={st.outstanding} free={st.free_slots}") + continue + + for d in descs: + buf = self.buffers[d.piece] + buf[d.begin:d.begin + d.len] = self.peer.block_data(d.slot, d.len) + self.peer.release(d.slot) + self.received[d.piece] += d.len + if (not self.done[d.piece] + and self.received[d.piece] >= meta.piece_len(d.piece)): + digest = hashlib.sha1(bytes(buf)).digest() + if digest != meta.piece_hashes[d.piece]: + raise ValueError(f"piece {d.piece} hash mismatch") + self.done[d.piece] = 1 + self.done_count += 1 + # Drop priority so a verified piece is no longer a + # selection candidate (the peer also won't re-request it). + self.peer.set_priority(d.piece, 0) + if out_fh: + out_fh.seek(d.piece * meta.piece_length) + out_fh.write(buf) + if not output: + pass # keep in memory for return + else: + self.buffers[d.piece] = None # free once flushed + + if out_fh: + out_fh.close() + return None + return b"".join(bytes(self.buffers[i]) for i in pieces) + + def close(self): + self.peer.stop() + self.peer.close() + + +def main() -> int: + ap = argparse.ArgumentParser(description="Drive the C peer to download a torrent.") + ap.add_argument("torrent", help="path to .torrent file") + ap.add_argument("--peer", help="explicit peer ip:port (skip tracker)") + ap.add_argument("--output", "-o", help="write downloaded data here") + ap.add_argument("--slots", type=int, default=0, help="arena slots (16 KiB each)") + ap.add_argument("--pipeline", type=int, default=0, help="max outstanding requests") + ap.add_argument("--timeout", type=float, default=120.0) + args = ap.parse_args() + + meta = load_metadata(args.torrent) + print(f"torrent: {meta.name} {meta.total_size} bytes " + f"{meta.num_pieces} pieces x {meta.piece_length}") + + if args.peer: + host, port = args.peer.rsplit(":", 1) + endpoints = [(host, int(port))] + else: + print("discovering peers via tracker/DHT...") + endpoints = discover_peers(args.torrent) + if not endpoints: + print("no peers found") + return 1 + print(f"found {len(endpoints)} peer(s); using {endpoints[0]}") + + dl = Downloader(meta, num_slots=args.slots, max_pipeline=args.pipeline) + try: + ip, port = endpoints[0] + t0 = time.time() + dl.download(ip, port, timeout=args.timeout, output=args.output) + dt = time.time() - t0 + mb = meta.total_size / 1e6 + print(f"done: {mb:.1f} MB in {dt:.2f}s = {mb/dt:.1f} MB/s") + finally: + dl.close() + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/harness/peer_ffi.py b/harness/peer_ffi.py new file mode 100644 index 0000000..dbff8b8 --- /dev/null +++ b/harness/peer_ffi.py @@ -0,0 +1,181 @@ +""" +ctypes bindings for libtorrentpeer.so. + +The arena is wrapped once as a zero-copy ``memoryview``; ``block_data()`` returns +a slice into it, so the harness never copies a block until it chooses to (e.g. +into a per-piece buffer for hashing). Returning the slot via ``release()`` is what +lets the peer issue new requests (credit-based flow control). +""" +from __future__ import annotations + +import ctypes as C +import os + +# peer_state / peer_error mirrors of include/peer.h +STATE_IDLE, STATE_CONNECTING, STATE_HANDSHAKE, STATE_CHOKED, \ + STATE_RUNNING, STATE_STOPPED, STATE_ERROR = range(7) +STATE_NAMES = ["IDLE", "CONNECTING", "HANDSHAKE", "CHOKED", + "RUNNING", "STOPPED", "ERROR"] +ERROR_NAMES = ["OK", "CONNECT", "HANDSHAKE", "CLOSED", "PROTOCOL", "IO", "NOMEM"] + +BLOCK_SIZE = 16384 + + +class PeerConfig(C.Structure): + _fields_ = [ + ("info_hash", C.c_uint8 * 20), + ("peer_id", C.c_uint8 * 20), + ("piece_length", C.c_uint64), + ("total_size", C.c_uint64), + ("num_pieces", C.c_uint32), + ("num_slots", C.c_uint32), + ("max_pipeline", C.c_uint32), + ("request_timeout_ms", C.c_uint32), + ("recv_buffer_bytes", C.c_uint32), + ] + + +class BlockDesc(C.Structure): + _fields_ = [ + ("piece", C.c_uint32), + ("begin", C.c_uint32), + ("len", C.c_uint32), + ("slot", C.c_uint32), + ] + + +class PeerStatus(C.Structure): + _fields_ = [ + ("state", C.c_int32), + ("error", C.c_int32), + ("bytes_received", C.c_uint64), + ("blocks_received", C.c_uint64), + ("outstanding", C.c_uint32), + ("free_slots", C.c_uint32), + ("pipeline_target", C.c_uint32), + ("rate_bps", C.c_double), + ("rtt_min_ms", C.c_double), + ] + + +def _default_lib_path() -> str: + here = os.path.dirname(os.path.abspath(__file__)) + cand = [ + os.path.join(here, "..", "build", "libtorrentpeer.so"), + os.path.join(here, "..", "build", "lib", "libtorrentpeer.so"), + ] + for p in cand: + if os.path.exists(p): + return os.path.abspath(p) + return os.path.abspath(cand[0]) + + +def _load(lib_path: str | None) -> C.CDLL: + lib = C.CDLL(lib_path or _default_lib_path()) + lib.peer_create.restype = C.c_void_p + lib.peer_create.argtypes = [C.POINTER(PeerConfig)] + lib.peer_start.restype = C.c_int + lib.peer_start.argtypes = [C.c_void_p, C.c_char_p, C.c_uint16] + lib.peer_set_priorities.restype = C.c_int + lib.peer_set_priorities.argtypes = [C.c_void_p, C.POINTER(C.c_uint8), C.c_uint32] + lib.peer_set_priority.restype = C.c_int + lib.peer_set_priority.argtypes = [C.c_void_p, C.c_uint32, C.c_uint8] + lib.peer_request_piece.restype = C.c_int + lib.peer_request_piece.argtypes = [C.c_void_p, C.c_uint32] + lib.peer_stop.restype = None + lib.peer_stop.argtypes = [C.c_void_p] + lib.peer_destroy.restype = None + lib.peer_destroy.argtypes = [C.c_void_p] + lib.peer_arena_base.restype = C.c_void_p + lib.peer_arena_base.argtypes = [C.c_void_p] + lib.peer_arena_bytes.restype = C.c_uint64 + lib.peer_arena_bytes.argtypes = [C.c_void_p] + lib.peer_poll_ready.restype = C.c_uint32 + lib.peer_poll_ready.argtypes = [C.c_void_p, C.POINTER(BlockDesc), C.c_uint32] + lib.peer_release_slot.restype = None + lib.peer_release_slot.argtypes = [C.c_void_p, C.c_uint32] + lib.peer_wait.restype = C.c_int + lib.peer_wait.argtypes = [C.c_void_p, C.c_int] + lib.peer_get_status.restype = None + lib.peer_get_status.argtypes = [C.c_void_p, C.POINTER(PeerStatus)] + return lib + + +class Peer: + """Pythonic wrapper around one peer_handle.""" + + def __init__(self, cfg: PeerConfig, lib_path: str | None = None, + poll_batch: int = 1024): + self._lib = _load(lib_path) + self._h = self._lib.peer_create(C.byref(cfg)) + if not self._h: + raise RuntimeError("peer_create failed (bad config or OOM)") + + base = self._lib.peer_arena_base(self._h) + nbytes = self._lib.peer_arena_bytes(self._h) + arena_t = (C.c_char * nbytes) + self._arena = arena_t.from_address(base) + # zero-copy view over the slab, as unsigned bytes for clean slicing + self.arena = memoryview(self._arena).cast("B") + + self._batch = poll_batch + self._descs = (BlockDesc * poll_batch)() + + def start(self, ip: str, port: int) -> None: + rc = self._lib.peer_start(self._h, ip.encode(), port) + if rc != 0: + raise RuntimeError("peer_start failed") + + def set_priorities(self, priorities) -> None: + """Set the whole per-piece priority vector (len must == num_pieces).""" + buf = bytes(priorities) + arr = (C.c_uint8 * len(buf)).from_buffer_copy(buf) + if self._lib.peer_set_priorities(self._h, arr, len(buf)) != 0: + raise ValueError("set_priorities: length must equal num_pieces") + + def set_priority(self, piece_index: int, priority: int) -> None: + if self._lib.peer_set_priority(self._h, piece_index, priority) != 0: + raise ValueError(f"set_priority({piece_index}) out of range") + + def request_piece(self, piece_index: int) -> None: + """Re-arm a piece for (re-)download (e.g. after a hash failure).""" + if self._lib.peer_request_piece(self._h, piece_index) != 0: + raise ValueError(f"request_piece({piece_index}) out of range") + + def poll_ready(self): + """Return a list of BlockDesc for completed blocks (may be empty).""" + n = self._lib.peer_poll_ready(self._h, self._descs, self._batch) + return [self._descs[i] for i in range(n)] + + def block_data(self, slot: int, length: int) -> memoryview: + off = slot * BLOCK_SIZE + return self.arena[off:off + length] + + def release(self, slot: int) -> None: + self._lib.peer_release_slot(self._h, slot) + + def wait(self, timeout_ms: int) -> int: + return self._lib.peer_wait(self._h, timeout_ms) + + def status(self) -> PeerStatus: + st = PeerStatus() + self._lib.peer_get_status(self._h, C.byref(st)) + return st + + def stop(self) -> None: + if self._h: + self._lib.peer_stop(self._h) + + def close(self) -> None: + if self._h: + # Drop the memoryview before freeing the arena it points into. + self.arena.release() + del self._arena + self._lib.peer_destroy(self._h) + self._h = None + + def __enter__(self): + return self + + def __exit__(self, *exc): + self.close() diff --git a/harness/seed_server.py b/harness/seed_server.py new file mode 100644 index 0000000..272a463 --- /dev/null +++ b/harness/seed_server.py @@ -0,0 +1,364 @@ +""" +seed_server.py - A lightweight emulated BitTorrent client (seed side) for +load-testing the peer/engine against "as many peers as possible". + +Rather than spinning up N heavyweight libtorrent sessions, this serves one +torrent's data from a single asyncio event loop across many listener sockets. +Each listener is a distinct endpoint, so from the engine's point of view each is +a separate peer: one `add_peer(tid, ip, port)` per listener. + +It speaks the plaintext BitTorrent v1 wire protocol a real client would on the +seed side: validates the handshake + info-hash, advertises a full bitfield, +unchokes, and answers `request` messages with `piece` data read (mmap'd) from +disk. That is exactly the subset the engine drives, and it is plaintext because +the engine does not negotiate MSE encryption yet. + +Usage +----- +Serve an existing torrent (data already on disk under --data): + python harness/seed_server.py some.torrent --data /path/to/datadir --peers 64 + +Generate a random test torrent, serve it, and print the .torrent path: + python harness/seed_server.py --generate 256M --peers 64 --out /tmp/seedtest + +Generate + serve + drive the engine against every peer and report throughput: + python harness/seed_server.py --generate 256M --peers 64 --out /tmp/seedtest --self-test + +In a test, use the SeedSwarm class directly to start listeners in-process and +read `swarm.endpoints`. +""" +from __future__ import annotations + +import argparse +import asyncio +import mmap +import os +import secrets +import struct +import threading +import time + +PSTR = b"BitTorrent protocol" +HANDSHAKE_LEN = 68 + +# Wire message ids. +MSG_CHOKE, MSG_UNCHOKE, MSG_INTERESTED, MSG_NOT_INTERESTED = 0, 1, 2, 3 +MSG_HAVE, MSG_BITFIELD, MSG_REQUEST, MSG_PIECE, MSG_CANCEL = 4, 5, 6, 7, 8 + +DRAIN_HIGH_WATER = 1 << 20 # let blocks queue up, apply backpressure past 1 MiB +MAX_BLOCK = 1 << 17 # reject absurd request lengths (128 KiB) + + +# --------------------------------------------------------------------------- # +# Data source: read (offset, length) from the torrent's concatenated files. +# --------------------------------------------------------------------------- # +class DataSource: + """Maps the linear piece space onto one or more on-disk files (BT v1 lays + files out back-to-back). Files are mmap'd for cheap repeated reads.""" + + def __init__(self, files: list[tuple[str, int]]): + self._maps = [] # (global_offset, size, mmap_or_none) + self._handles = [] + off = 0 + for path, size in files: + mm = None + if size > 0: + fh = open(path, "rb") + self._handles.append(fh) + mm = mmap.mmap(fh.fileno(), size, prot=mmap.PROT_READ) + self._maps.append((off, size, mm)) + off += size + self.total = off + self._single = self._maps[0][2] if len(self._maps) == 1 else None + + def read(self, offset: int, length: int) -> bytes: + if self._single is not None: # fast path: one file + return self._single[offset:offset + length] + out = bytearray() + remaining = length + for foff, size, mm in self._maps: + if remaining <= 0: + break + if offset >= foff + size or offset < foff: + continue + local = offset - foff + take = min(size - local, remaining) + out += mm[local:local + take] + offset += take + remaining -= take + return bytes(out) + + def close(self): + for _, _, mm in self._maps: + if mm is not None: + mm.close() + for fh in self._handles: + fh.close() + + +def _full_bitfield(num_pieces: int) -> bytes: + nbytes = (num_pieces + 7) // 8 + bf = bytearray(b"\xff" * nbytes) + rem = num_pieces & 7 + if rem: # clear pad bits past the end + bf[-1] = (0xFF << (8 - rem)) & 0xFF + return bytes(bf) + + +def _msg(mid: int, payload: bytes = b"") -> bytes: + return struct.pack(">IB", 1 + len(payload), mid) + payload + + +# --------------------------------------------------------------------------- # +# The swarm: many listeners sharing one event loop, on a background thread. +# --------------------------------------------------------------------------- # +class SeedSwarm: + def __init__(self, info_hash: bytes, piece_length: int, num_pieces: int, + source: DataSource, host: str = "127.0.0.1"): + self.info_hash = info_hash + self.piece_length = piece_length + self.num_pieces = num_pieces + self.source = source + self.host = host + self.bitfield = _full_bitfield(num_pieces) + self.endpoints: list[tuple[str, int]] = [] + self.served_bytes = 0 + self._loop = asyncio.new_event_loop() + self._servers: list[asyncio.AbstractServer] = [] + self._thread = threading.Thread(target=self._run, daemon=True) + + # -- connection handler (one per accepted peer) ------------------------- # + async def _handle(self, reader: asyncio.StreamReader, + writer: asyncio.StreamWriter): + try: + hs = await reader.readexactly(HANDSHAKE_LEN) + if hs[1:20] != PSTR or hs[28:48] != self.info_hash: + writer.close() + return + peer_id = b"-SD0001-" + secrets.token_bytes(12) + writer.write(bytes([len(PSTR)]) + PSTR + b"\x00" * 8 + + self.info_hash + peer_id) + writer.write(_msg(MSG_BITFIELD, self.bitfield)) + writer.write(_msg(MSG_UNCHOKE)) + await writer.drain() + + while True: + (length,) = struct.unpack(">I", await reader.readexactly(4)) + if length == 0: + continue # keep-alive + body = await reader.readexactly(length) + mid = body[0] + if mid == MSG_REQUEST and length >= 13: + index, begin, blen = struct.unpack(">III", body[1:13]) + if blen > MAX_BLOCK: + continue + data = self.source.read(index * self.piece_length + begin, blen) + writer.write(struct.pack(">IB", 9 + len(data), MSG_PIECE) + + struct.pack(">II", index, begin) + data) + self.served_bytes += len(data) + if writer.transport.get_write_buffer_size() > DRAIN_HIGH_WATER: + await writer.drain() + # interested / not-interested / cancel / choke: nothing to do — + # we are already unchoked and answer requests as they arrive. + except (asyncio.IncompleteReadError, ConnectionResetError, + BrokenPipeError, ConnectionError): + pass + finally: + try: + writer.close() + except Exception: + pass + + async def _make_servers(self, count: int): + servers, endpoints = [], [] + for _ in range(count): + srv = await asyncio.start_server(self._handle, self.host, 0) + servers.append(srv) + endpoints.append((self.host, srv.sockets[0].getsockname()[1])) + return servers, endpoints + + def _run(self): + asyncio.set_event_loop(self._loop) + self._loop.run_forever() + + def start(self, count: int): + self._thread.start() + fut = asyncio.run_coroutine_threadsafe(self._make_servers(count), self._loop) + self._servers, self.endpoints = fut.result(timeout=15) + return self.endpoints + + def stop(self): + def _shutdown(): + for srv in self._servers: + srv.close() + self._loop.stop() + if self._loop.is_running(): + self._loop.call_soon_threadsafe(_shutdown) + self._thread.join(timeout=5) + try: + self._loop.close() + except Exception: + pass + self.source.close() + + +# --------------------------------------------------------------------------- # +# Torrent loading / generation. +# --------------------------------------------------------------------------- # +def load_torrent(torrent_path: str, data_dir: str): + """Return (info_hash, piece_length, num_pieces, total_size, DataSource).""" + from torrent_meta import load_torrent as parse_torrent + tf = parse_torrent(torrent_path) + files = [] + for path, size in tf.files: + files.append((os.path.join(data_dir, path), size)) + missing = [p for p, _ in files if not os.path.exists(p)] + if missing: + raise FileNotFoundError(f"data files not found under {data_dir}: {missing}") + meta = tf.metadata + return (meta.info_hash, meta.piece_length, meta.num_pieces, + meta.total_size, DataSource(files)) + + +def generate_torrent(out_dir: str, size: int, piece_size: int = 256 * 1024): + """Create a random data file + .torrent under out_dir. Returns + (torrent_path, info_hash, piece_length, num_pieces, total_size, DataSource).""" + import libtorrent as lt + os.makedirs(out_dir, exist_ok=True) + data_path = os.path.join(out_dir, "data.bin") + with open(data_path, "wb") as f: + rem = size + while rem > 0: + n = min(rem, 8 << 20) + f.write(os.urandom(n)) + rem -= n + fs = lt.file_storage() + lt.add_files(fs, data_path) + t = lt.create_torrent(fs, piece_size=piece_size) + t.set_priv(False) + lt.set_piece_hashes(t, out_dir) + torrent_path = os.path.join(out_dir, "test.torrent") + with open(torrent_path, "wb") as f: + f.write(lt.bencode(t.generate())) + ih, pl, npc, total, src = load_torrent(torrent_path, out_dir) + return torrent_path, ih, pl, npc, total, src + + +def parse_size(s: str) -> int: + s = s.strip().upper() + mult = 1 + if s and s[-1] in "KMG": + mult = {"K": 1024, "M": 1024**2, "G": 1024**3}[s[-1]] + s = s[:-1] + return int(float(s) * mult) + + +# --------------------------------------------------------------------------- # +# Optional: drive the engine against every peer and verify + report rate. +# --------------------------------------------------------------------------- # +def _self_test(torrent_path: str, data_dir: str, endpoints, timeout: float): + import hashlib + from engine_ffi import Engine, EngineConfig, STATE_ERROR, ERROR_NAMES + from harness import load_metadata + + meta = load_metadata(torrent_path) + lib = os.path.join(os.path.dirname(os.path.dirname(os.path.abspath(__file__))), + "build", "libtorrentpeer.so") + loops = min(8, max(1, (os.cpu_count() or 1))) + with Engine(EngineConfig(loop_count=loops, slots_per_loop=4096, + max_pipeline=1024), lib_path=lib) as eng: + tid = eng.add_torrent(meta.info_hash, b"-PC0001-" + secrets.token_bytes(12), + meta.piece_length, meta.total_size, meta.num_pieces) + eng.set_priorities(tid, [1] * meta.num_pieces) + for ip, port in endpoints: + eng.add_peer(tid, ip, port) + + buffers = [bytearray(meta.piece_len(i)) for i in range(meta.num_pieces)] + received = [0] * meta.num_pieces + done = bytearray(meta.num_pieces) + done_count = 0 + t0 = time.time() + deadline = t0 + timeout + while done_count < meta.num_pieces: + st = eng.status(tid) + if st.state == STATE_ERROR: + raise RuntimeError(f"engine error: {ERROR_NAMES[st.error]}") + descs = eng.poll_ready() + if not descs: + eng.wait(200) + if time.time() > deadline: + raise TimeoutError(f"stalled at {done_count}/{meta.num_pieces}") + continue + for x in descs: + buf = buffers[x.piece] + buf[x.begin:x.begin + x.len] = eng.block_data(x.loop, x.slot, x.len) + eng.release(x.loop, x.slot) + received[x.piece] += x.len + if not done[x.piece] and received[x.piece] >= meta.piece_len(x.piece): + if hashlib.sha1(bytes(buf)).digest() != meta.piece_hashes[x.piece]: + raise ValueError(f"piece {x.piece} hash mismatch") + done[x.piece] = 1 + done_count += 1 + eng.set_priority(tid, x.piece, 0) + dt = time.time() - t0 + mb = meta.total_size / 1e6 + print(f"self-test OK: {mb:.1f} MB from {len(endpoints)} peers " + f"in {dt:.2f}s = {mb/dt:.1f} MB/s (peers={eng.status(tid).peers})") + + +def main() -> int: + ap = argparse.ArgumentParser(description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("torrent", nargs="?", help="path to an existing .torrent") + ap.add_argument("--data", help="directory (or file) holding the torrent's data") + ap.add_argument("--generate", metavar="SIZE", + help="generate a random torrent of this size (e.g. 256M, 1G)") + ap.add_argument("--out", default="/tmp/seedtest", + help="output dir for --generate (default: /tmp/seedtest)") + ap.add_argument("--peers", type=int, default=32, help="number of listeners") + ap.add_argument("--host", default="127.0.0.1") + ap.add_argument("--piece-size", type=int, default=256 * 1024) + ap.add_argument("--ports-file", help="write the endpoints (one ip:port/line) here") + ap.add_argument("--self-test", action="store_true", + help="drive the engine against all peers, verify, report rate") + ap.add_argument("--timeout", type=float, default=300.0) + args = ap.parse_args() + + torrent_path = args.torrent + if args.generate: + size = parse_size(args.generate) + torrent_path, ih, pl, npc, total, src = generate_torrent( + args.out, size, args.piece_size) + data_dir = args.out + print(f"generated {torrent_path} ({total} bytes, {npc} pieces x {pl})") + else: + if not torrent_path or not args.data: + ap.error("provide a .torrent and --data, or use --generate SIZE") + data_dir = args.data if os.path.isdir(args.data) else os.path.dirname(args.data) + ih, pl, npc, total, src = load_torrent(torrent_path, data_dir) + + swarm = SeedSwarm(ih, pl, npc, src, host=args.host) + swarm.start(args.peers) + print(f"seeding {total} bytes on {len(swarm.endpoints)} peers " + f"({args.host}:{swarm.endpoints[0][1]}..{swarm.endpoints[-1][1]})") + if args.ports_file: + with open(args.ports_file, "w") as f: + f.writelines(f"{ip}:{port}\n" for ip, port in swarm.endpoints) + print(f"endpoints written to {args.ports_file}") + + try: + if args.self_test: + _self_test(torrent_path, data_dir, swarm.endpoints, args.timeout) + else: + print("serving; press Ctrl-C to stop") + while True: + time.sleep(1.0) + except KeyboardInterrupt: + pass + finally: + swarm.stop() + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/harness/swarm_download.py b/harness/swarm_download.py new file mode 100644 index 0000000..bb91155 --- /dev/null +++ b/harness/swarm_download.py @@ -0,0 +1,543 @@ +""" +swarm_download.py - Download a real torrent from a real swarm using the engine, +to measure how it performs against actual peers. + +Division of labour: for .torrent files, this harness parses metainfo locally and +uses the sibling torrent-tracker library for DHT get_peers and HTTP/UDP tracker +announces. Magnet metadata resolution still falls back to libtorrent. Every +discovered peer endpoint is fed to the engine, which does all the data transfer. +Pieces are reassembled and SHA-1-verified here; the engine never hashes or +persists anything. + +Reality check (important for interpreting the numbers): a real public swarm +contains unreachable peers, peers behind NAT, peers that only accept a different +transport/encryption combination, and peers that do not actually have useful +pieces. Those show up as "failed". The headline metric this prints is therefore +how many discovered peers were actually usable, and the sustained rate across +them. That is the honest "how well does it work today" answer. + +Usage +----- + python harness/swarm_download.py path/to/file.torrent + python harness/swarm_download.py 'magnet:?xt=urn:btih:...' + python harness/swarm_download.py file.torrent --max-peers 200 --output /tmp/out --timeout 600 + python harness/swarm_download.py file.torrent --output /tmp/out --resume + +Pick a well-seeded torrent (e.g. a current Linux distro ISO) for a meaningful +test; obscure or dead torrents will show few usable peers regardless. +""" +from __future__ import annotations + +import argparse +import hashlib +import os +import secrets +import sys +import tempfile +import time + +ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) +sys.path.insert(0, os.path.join(ROOT, "harness")) + +import libtorrent as lt # noqa: E402 # only used for magnet metadata fallback + +from harness import load_metadata # noqa: E402 +from engine_ffi import (BLOCK_SIZE, Engine, EngineConfig, STATE_NAMES, # noqa: E402 + ERROR_NAMES) +from torrent_meta import load_torrent # noqa: E402 +from tracker_ffi import DHTClient, TrackerClient # noqa: E402 + +LIB = os.path.join(ROOT, "build", "libtorrentpeer.so") + + +def _discovery_session() -> lt.session: + s = lt.session({ + "listen_interfaces": "0.0.0.0:0,[::]:0", + "enable_dht": True, "enable_lsd": True, + "enable_upnp": True, "enable_natpmp": True, + "alert_mask": 0, + }) + for host, port in (("router.bittorrent.com", 6881), + ("dht.transmissionbt.com", 6881), + ("router.utorrent.com", 6881)): + try: + s.add_dht_node((host, port)) + except Exception: + pass + return s + + +class LibtorrentDiscovery: + def __init__(self, handle, session): + self.handle = handle + self.session = session + + def collect(self, _max_peers: int) -> set[tuple[str, int]]: + eps = set() + try: + for pi in self.handle.get_peer_info(): + ip = pi.ip + if isinstance(ip, tuple) and len(ip) == 2 and ip[1]: + eps.add((ip[0], int(ip[1]))) + except Exception: + pass + return eps + + def close(self): + try: + self.session.remove_torrent(self.handle) + except Exception: + pass + + +class TrackerDiscovery: + def __init__(self, trackers: list[str], meta, *, timeout: float, + max_trackers: int, use_dht: bool, dht_timeout: float, + dht_queries: int): + self.trackers = trackers[:max_trackers] if max_trackers > 0 else trackers + self.meta = meta + self.timeout = timeout + self.client = TrackerClient() + self.dht = DHTClient() if use_dht else None + self.dht_timeout = dht_timeout + self.dht_queries = dht_queries + self.dht_done = False + self.peer_id = b"-PC0001-" + secrets.token_bytes(12) + self.key = secrets.randbits(32) + self.endpoints: set[tuple[str, int]] = set() + self.index = 0 + self.next_cycle_at = 0.0 + + def collect(self, max_peers: int) -> set[tuple[str, int]]: + now = time.time() + if len(self.endpoints) >= max_peers: + return set(self.endpoints) + + if self.dht and not self.dht_done: + self.dht_done = True + result = self.dht.lookup( + self.meta.info_hash, + timeout=self.dht_timeout, + max_queries=self.dht_queries, + max_peers=max_peers - len(self.endpoints), + ) + self.endpoints.update(result.peers) + msg = (f"dht: {len(result.peers)} peers, " + f"{result.nodes_queried} queried/{result.nodes_discovered} learned " + f"({result.elapsed_ms:.0f} ms)") + if result.error: + msg += f": {result.error}" + print(msg, flush=True) + if len(self.endpoints) >= max_peers: + return set(self.endpoints) + + if self.index >= len(self.trackers): + if now < self.next_cycle_at: + return set(self.endpoints) + self.index = 0 + + if not self.trackers: + return set() + + url = self.trackers[self.index] + self.index += 1 + if self.index >= len(self.trackers): + self.next_cycle_at = now + 60.0 + + result = self.client.announce( + url, self.meta, self.peer_id, port=6881, key=self.key, + numwant=max(1, min(200, max_peers - len(self.endpoints))), + event="started", timeout=self.timeout) + if result.ok: + self.endpoints.update(result.peers) + print(f"tracker: {url} returned {len(result.peers)} peers " + f"({result.elapsed_ms:.0f} ms)", flush=True) + else: + print(f"tracker: {url} failed: {result.error}", flush=True) + return set(self.endpoints) + + def close(self): + pass + + +def _resolve_magnet(arg: str, scratch: str): + ses = _discovery_session() + params = lt.parse_magnet_uri(arg) + params.save_path = scratch + params.flags |= lt.torrent_flags.upload_mode + h = ses.add_torrent(params) + print("resolving magnet metadata from the swarm using libtorrent...", flush=True) + deadline = time.time() + 120 + while time.time() < deadline and not h.status().has_metadata: + time.sleep(0.5) + if not h.status().has_metadata: + raise TimeoutError("could not fetch metadata for magnet within 120s") + tpath = os.path.join(scratch, "resolved.torrent") + with open(tpath, "wb") as f: + f.write(lt.bencode(lt.create_torrent(h.torrent_file()).generate())) + return tpath, LibtorrentDiscovery(h, ses) + + +def _completed_bytes(meta, done: bytearray) -> int: + return sum(meta.piece_len(i) for i, is_done in enumerate(done) if is_done) + + +def verify_existing_output(out_fh, meta) -> tuple[bytearray, int, int]: + """Hash pieces already present in the output file. + + Returns (done_bitfield, done_count, verified_bytes). Only pieces whose full + range exists and whose SHA-1 matches the torrent metadata are marked done. + This intentionally ignores partial pieces because the engine currently only + persists data after a whole piece has passed verification. + """ + done = bytearray(meta.num_pieces) + done_count = 0 + file_size = os.fstat(out_fh.fileno()).st_size + + for piece in range(meta.num_pieces): + piece_len = meta.piece_len(piece) + offset = piece * meta.piece_length + if offset + piece_len > file_size: + continue + + out_fh.seek(offset) + data = out_fh.read(piece_len) + if len(data) != piece_len: + continue + if hashlib.sha1(data).digest() != meta.piece_hashes[piece]: + continue + + done[piece] = 1 + done_count += 1 + + return done, done_count, _completed_bytes(meta, done) + + +class PieceAssembler: + """Reassemble pieces while ignoring duplicate blocks. + + Endgame mode deliberately re-requests unfinished pieces from multiple peers. + That means the same block may arrive more than once. Counting raw bytes would + mark a piece complete too early, so completion is based on unique block + offsets within each piece. + """ + + def __init__(self, meta, done: bytearray): + self.meta = meta + self.done = done + self.buffers: dict[int, bytearray] = {} + self.received = [0] * meta.num_pieces + self.seen: dict[int, bytearray] = {} + + def _block_count(self, piece: int) -> int: + piece_len = self.meta.piece_len(piece) + return (piece_len + BLOCK_SIZE - 1) // BLOCK_SIZE + + def reset_piece(self, piece: int) -> int: + previous = self.received[piece] + self.received[piece] = 0 + self.buffers.pop(piece, None) + self.seen.pop(piece, None) + return previous + + def add_block(self, piece: int, begin: int, data) -> tuple[bool, bool]: + if self.done[piece]: + return False, False + if begin % BLOCK_SIZE != 0: + raise ValueError(f"unaligned block for piece {piece}: begin={begin}") + piece_len = self.meta.piece_len(piece) + length = len(data) + if begin + length > piece_len: + raise ValueError( + f"block overruns piece {piece}: begin={begin} len={length} " + f"piece_len={piece_len}") + + buf = self.buffers.get(piece) + if buf is None: + buf = bytearray(piece_len) + self.buffers[piece] = buf + seen = self.seen.get(piece) + if seen is None: + seen = bytearray(self._block_count(piece)) + self.seen[piece] = seen + + block = begin // BLOCK_SIZE + buf[begin:begin + length] = data + if seen[block]: + return False, self.received[piece] >= piece_len + + seen[block] = 1 + self.received[piece] += length + return True, self.received[piece] >= piece_len + + def piece_bytes(self, piece: int) -> bytes: + return bytes(self.buffers[piece]) + + def finish_piece(self, piece: int) -> None: + self.done[piece] = 1 + self.buffers.pop(piece, None) + self.seen.pop(piece, None) + + +class EndgameController: + """Conservative piece-level endgame for slow tail pieces.""" + + def __init__(self, meta, *, min_pieces: int, peer_factor: float, + interval: float): + self.meta = meta + self.min_pieces = min_pieces + self.peer_factor = peer_factor + self.interval = interval + self.active = False + self.last_rearm = 0.0 + + def maybe_rearm(self, eng: Engine, tid: int, done: bytearray, + done_count: int, connected: int, now: float) -> None: + remaining = self.meta.num_pieces - done_count + if remaining <= 0: + return + + threshold = max(self.min_pieces, + int(max(1, connected) * self.peer_factor)) + if remaining > threshold: + return + + if not self.active: + self.active = True + print(f"endgame: {remaining} pieces left; duplicating tail requests", + flush=True) + for piece, is_done in enumerate(done): + if not is_done: + eng.set_priority(tid, piece, 255) + self.last_rearm = 0.0 + + if now - self.last_rearm < self.interval: + return + + for piece, is_done in enumerate(done): + if not is_done: + eng.request_piece(tid, piece) + self.last_rearm = now + + +def open_output(path: str, meta, resume: bool): + if resume: + existed = os.path.exists(path) + out_fh = open(path, "r+b" if existed else "w+b") + if existed: + print("resume: verifying existing output pieces...", flush=True) + done, done_count, verified_bytes = verify_existing_output(out_fh, meta) + pct = 100.0 * done_count / meta.num_pieces if meta.num_pieces else 100.0 + print(f"resume: found {done_count}/{meta.num_pieces} verified pieces " + f"({pct:4.1f}%, {verified_bytes/1e6:.1f} MB)", + flush=True) + else: + done = bytearray(meta.num_pieces) + done_count = 0 + print("resume: output file does not exist yet; starting fresh", + flush=True) + else: + out_fh = open(path, "w+b") + done = bytearray(meta.num_pieces) + done_count = 0 + + out_fh.truncate(meta.total_size) + return out_fh, done, done_count + + +def main() -> int: + ap = argparse.ArgumentParser( + description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("torrent", help="path to a .torrent file or a magnet: URI") + ap.add_argument("--max-peers", type=int, default=100, + help="cap on peers fed to the engine (default 100)") + ap.add_argument("--output", "-o", help="write verified data here (else discard)") + ap.add_argument("--resume", action="store_true", + help="resume from an existing --output file by hashing " + "verified pieces and skipping them") + ap.add_argument("--timeout", type=float, default=600.0, + help="overall stall timeout in seconds") + ap.add_argument("--tracker-timeout", type=float, default=4.0, + help="per-tracker announce timeout for direct tracker " + "discovery (default 4)") + ap.add_argument("--max-trackers", type=int, default=16, + help="maximum trackers to announce to from a .torrent " + "(0 = all, default 16)") + ap.add_argument("--no-dht", action="store_true", + help="disable DHT get_peers discovery for .torrent files") + ap.add_argument("--dht-timeout", type=float, default=6.0, + help="total DHT lookup budget in seconds (default 6)") + ap.add_argument("--dht-queries", type=int, default=32, + help="maximum DHT nodes to query per torrent (default 32)") + ap.add_argument("--lib", default=LIB) + ap.add_argument("--loops", type=int, default=0, help="engine loops (0=auto)") + ap.add_argument("--encryption", type=int, default=1, choices=[0, 1, 2], + help="0=plaintext, 1=MSE offer RC4+plaintext (default), " + "2=MSE require RC4. Encryption reaches far more of a " + "real swarm.") + ap.add_argument("--utp", type=int, default=0, choices=[0, 1], + help="0=TCP (default), 1=µTP/UDP. A swarm has a mix; this " + "selects which transport the engine dials with.") + ap.add_argument("--no-endgame", action="store_true", + help="disable duplicate tail-piece requests") + ap.add_argument("--endgame-min-pieces", type=int, default=16, + help="enter endgame when remaining pieces are at or below " + "this count, also scaled by connected peers " + "(default 16)") + ap.add_argument("--endgame-peer-factor", type=float, default=2.0, + help="also enter endgame below peers*factor remaining " + "pieces (default 2.0)") + ap.add_argument("--endgame-interval", type=float, default=3.0, + help="seconds between tail-piece re-arms in endgame " + "(default 3.0)") + args = ap.parse_args() + if args.resume and not args.output: + ap.error("--resume requires --output") + + scratch = tempfile.mkdtemp(prefix="swarm_dl_") + if args.torrent.startswith("magnet:"): + tpath, discovery = _resolve_magnet(args.torrent, scratch) + else: + torrent_file = load_torrent(args.torrent) + tpath = args.torrent + discovery = TrackerDiscovery( + torrent_file.trackers, + torrent_file.metadata, + timeout=args.tracker_timeout, + max_trackers=args.max_trackers, + use_dht=not args.no_dht, + dht_timeout=args.dht_timeout, + dht_queries=args.dht_queries, + ) + meta = load_metadata(tpath) + print(f"torrent: {meta.name} {meta.total_size/1e6:.1f} MB " + f"{meta.num_pieces} pieces x {meta.piece_length}", flush=True) + + out_fh = None + done = bytearray(meta.num_pieces) + done_count = 0 + if args.output: + out_fh, done, done_count = open_output(args.output, meta, args.resume) + + cfg = EngineConfig(loop_count=args.loops, slots_per_loop=4096, + max_pipeline=1024, encryption=args.encryption, + utp=args.utp) + print(f"transport: {'µTP' if args.utp else 'TCP'}, " + f"encryption={['off','offer','require'][args.encryption]}", flush=True) + eng = Engine(cfg, lib_path=args.lib) + tid = eng.add_torrent(meta.info_hash, b"-PC0001-" + secrets.token_bytes(12), + meta.piece_length, meta.total_size, meta.num_pieces) + eng.set_priorities(tid, [0 if done[i] else 1 + for i in range(meta.num_pieces)]) + + assembler = PieceAssembler(meta, done) + endgame = None if args.no_endgame else EndgameController( + meta, + min_pieces=max(1, args.endgame_min_pieces), + peer_factor=max(1.0, args.endgame_peer_factor), + interval=max(0.5, args.endgame_interval), + ) + added: set = set() + peak_connected = 0 + useful_bytes = _completed_bytes(meta, done) + + t0 = time.time() + deadline = t0 + args.timeout + last_print = 0.0 + last_useful_bytes = useful_bytes + last_progress_t = t0 + + try: + while done_count < meta.num_pieces: + # Feed any newly-discovered peers to the engine, up to the cap. + if len(added) < args.max_peers: + for ip, port in discovery.collect(args.max_peers): + if (ip, port) in added: + continue + added.add((ip, port)) + try: + eng.add_peer(tid, ip, port) + except Exception: + pass + if len(added) >= args.max_peers: + break + + st = eng.status(tid) + peak_connected = max(peak_connected, st.peers_connected) + now = time.time() + if endgame: + endgame.maybe_rearm( + eng, tid, done, done_count, st.peers_connected, now) + + descs = eng.poll_ready() + if not descs: + eng.wait(200) + now = time.time() + if useful_bytes != last_useful_bytes: + last_useful_bytes = useful_bytes + last_progress_t = now + if now - last_print >= 1.0: + last_print = now + pct = 100.0 * done_count / meta.num_pieces + print(f" {done_count}/{meta.num_pieces} pieces ({pct:4.1f}%) " + f"{st.rate_bps/1e6:6.1f} MB/s " + f"peers {st.peers_connected} up / {st.peers_failed} failed " + f"/ {len(added)} tried outstanding={st.outstanding}", + flush=True) + if now > deadline or (now - last_progress_t) > args.timeout: + print("stalled; giving up", flush=True) + break + continue + + for x in descs: + if done[x.piece]: + # A block can arrive after the piece was completed and + # de-prioritized because it was already in flight. Drop it. + eng.release(x.loop, x.slot) + continue + + block = bytes(eng.block_data(x.loop, x.slot, x.len)) + eng.release(x.loop, x.slot) + added_unique, complete = assembler.add_block( + x.piece, x.begin, block) + if added_unique: + useful_bytes += x.len + if complete: + piece_data = assembler.piece_bytes(x.piece) + if hashlib.sha1(piece_data).digest() != meta.piece_hashes[x.piece]: + # Corrupt/garbage block from a misbehaving peer: re-arm. + useful_bytes -= assembler.reset_piece(x.piece) + eng.request_piece(tid, x.piece) + continue + done[x.piece] = 1 + done_count += 1 + eng.set_priority(tid, x.piece, 0) + if out_fh: + out_fh.seek(x.piece * meta.piece_length) + out_fh.write(piece_data) + out_fh.flush() + assembler.finish_piece(x.piece) # free as we go + + dt = time.time() - t0 + st = eng.status(tid) + mb = _completed_bytes(meta, done) / 1e6 + print("-" * 70) + print(f"downloaded {done_count}/{meta.num_pieces} pieces " + f"({mb:.1f} MB) in {dt:.1f}s = {mb/dt:.1f} MB/s" if dt > 0 else "") + print(f"peers: {len(added)} discovered+tried, " + f"{peak_connected} usable at peak, " + f"{st.peers_failed} failed (likely incompatible transport/" + f"encryption or unreachable)") + if done_count < meta.num_pieces: + print(f"engine state: {STATE_NAMES[st.state]} " + f"err={ERROR_NAMES[st.error]}") + return 0 if done_count == meta.num_pieces else 2 + finally: + if out_fh: + out_fh.close() + eng.close() + discovery.close() + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/harness/torrent_meta.py b/harness/torrent_meta.py new file mode 100644 index 0000000..a1e2d2f --- /dev/null +++ b/harness/torrent_meta.py @@ -0,0 +1,223 @@ +from __future__ import annotations + +import hashlib +import os +from dataclasses import dataclass +from typing import Any + + +class BencodeError(ValueError): + pass + + +class BDecoder: + def __init__(self, data: bytes): + self.data = data + self.info_span: tuple[int, int] | None = None + + def parse(self) -> Any: + value, pos = self._value(0, top=True) + if pos != len(self.data): + raise BencodeError(f"trailing data at byte {pos}") + return value + + def _value(self, pos: int, *, top: bool = False) -> tuple[Any, int]: + if pos >= len(self.data): + raise BencodeError("unexpected end of bencode") + c = self.data[pos] + if c == ord("i"): + return self._int(pos) + if c == ord("l"): + return self._list(pos) + if c == ord("d"): + return self._dict(pos, top=top) + if ord("0") <= c <= ord("9"): + return self._bytes(pos) + raise BencodeError(f"invalid bencode byte {c!r} at {pos}") + + def _int(self, pos: int) -> tuple[int, int]: + end = self.data.find(b"e", pos) + if end < 0: + raise BencodeError("unterminated integer") + raw = self.data[pos + 1:end] + if not raw: + raise BencodeError("empty integer") + return int(raw), end + 1 + + def _bytes(self, pos: int) -> tuple[bytes, int]: + colon = self.data.find(b":", pos) + if colon < 0: + raise BencodeError("unterminated byte string length") + n = int(self.data[pos:colon]) + start = colon + 1 + end = start + n + if end > len(self.data): + raise BencodeError("byte string exceeds input") + return self.data[start:end], end + + def _list(self, pos: int) -> tuple[list[Any], int]: + out: list[Any] = [] + pos += 1 + while pos < len(self.data) and self.data[pos] != ord("e"): + value, pos = self._value(pos) + out.append(value) + if pos >= len(self.data): + raise BencodeError("unterminated list") + return out, pos + 1 + + def _dict(self, pos: int, *, top: bool = False) -> tuple[dict[bytes, Any], int]: + out: dict[bytes, Any] = {} + pos += 1 + while pos < len(self.data) and self.data[pos] != ord("e"): + key, pos = self._bytes(pos) + value_start = pos + value, pos = self._value(pos) + if top and key == b"info": + self.info_span = (value_start, pos) + out[key] = value + if pos >= len(self.data): + raise BencodeError("unterminated dict") + return out, pos + 1 + + +@dataclass +class Metadata: + info_hash: bytes + piece_length: int + total_size: int + num_pieces: int + piece_hashes: list[bytes] + name: str + + def piece_len(self, index: int) -> int: + if index + 1 == self.num_pieces: + return self.total_size - index * self.piece_length + return self.piece_length + + +@dataclass +class TorrentFile: + path: str + raw: bytes + metainfo: dict[bytes, Any] + info: dict[bytes, Any] + info_raw: bytes + metadata: Metadata + trackers: list[str] + files: list[tuple[str, int]] + + +def _text(value: Any, default: str = "") -> str: + if isinstance(value, bytes): + return value.decode("utf-8", "replace") + return default + + +def _file_tree_size(node: Any) -> int: + if not isinstance(node, dict): + return 0 + total = 0 + file_marker = node.get(b"") + if isinstance(file_marker, dict): + total += int(file_marker.get(b"length", 0)) + for key, child in node.items(): + if key != b"": + total += _file_tree_size(child) + return total + + +def _total_size(info: dict[bytes, Any]) -> int: + if b"length" in info: + return int(info[b"length"]) + if b"files" in info: + return sum(int(f.get(b"length", 0)) for f in info[b"files"]) + if b"file tree" in info: + return _file_tree_size(info[b"file tree"]) + return 0 + + +def _trackers(meta: dict[bytes, Any]) -> list[str]: + urls: list[str] = [] + announce = meta.get(b"announce") + if isinstance(announce, bytes): + urls.append(_text(announce)) + tiers = meta.get(b"announce-list") + if isinstance(tiers, list): + for tier in tiers: + if not isinstance(tier, list): + continue + for item in tier: + if isinstance(item, bytes): + urls.append(_text(item)) + seen: set[str] = set() + out: list[str] = [] + for url in urls: + if url and url not in seen: + seen.add(url) + out.append(url) + return out + + +def _path_text(parts: list[Any]) -> str: + decoded = [] + for part in parts: + if not isinstance(part, bytes): + raise BencodeError("file path component is not bytes") + decoded.append(part.decode("utf-8", "replace")) + return os.path.join(*decoded) if decoded else "" + + +def _files(info: dict[bytes, Any]) -> list[tuple[str, int]]: + name = _text(info.get(b"name"), "") + if b"length" in info: + return [(name, int(info[b"length"]))] + files = info.get(b"files") + if isinstance(files, list): + out = [] + for entry in files: + if not isinstance(entry, dict): + continue + path = entry.get(b"path") + if not isinstance(path, list): + continue + out.append((os.path.join(name, _path_text(path)), + int(entry.get(b"length", 0)))) + return out + return [] + + +def load_torrent(path: str) -> TorrentFile: + raw = open(path, "rb").read() + dec = BDecoder(raw) + meta = dec.parse() + if not isinstance(meta, dict) or dec.info_span is None: + raise BencodeError("metainfo does not contain a top-level info dict") + info = meta.get(b"info") + if not isinstance(info, dict): + raise BencodeError("metainfo info value is not a dict") + info_raw = raw[dec.info_span[0]:dec.info_span[1]] + + pieces = info.get(b"pieces") + if not isinstance(pieces, bytes) or len(pieces) % 20 != 0: + raise BencodeError("only v1/hybrid torrents with a valid pieces string are supported") + piece_length = int(info.get(b"piece length", 0)) + if piece_length <= 0: + raise BencodeError("missing or invalid piece length") + piece_hashes = [pieces[i:i + 20] for i in range(0, len(pieces), 20)] + total_size = _total_size(info) + if total_size <= 0: + raise BencodeError("missing torrent payload size") + metadata = Metadata( + info_hash=hashlib.sha1(info_raw).digest(), + piece_length=piece_length, + total_size=total_size, + num_pieces=len(piece_hashes), + piece_hashes=piece_hashes, + name=_text(info.get(b"name"), os.path.basename(path)), + ) + return TorrentFile(path, raw, meta, info, info_raw, metadata, _trackers(meta), + _files(info)) + + +def load_metadata(path: str) -> Metadata: + return load_torrent(path).metadata diff --git a/harness/tracker_ffi.py b/harness/tracker_ffi.py new file mode 100644 index 0000000..5cae7fe --- /dev/null +++ b/harness/tracker_ffi.py @@ -0,0 +1,498 @@ +from __future__ import annotations + +import ctypes as C +import os +import random +import socket +import ssl +import struct +import time +from dataclasses import dataclass +from urllib.parse import urlsplit, urlunsplit +from urllib.request import Request, urlopen + +TRACKER_OK = 0 +TRACKER_EVENT_NONE = 0 +TRACKER_EVENT_COMPLETED = 1 +TRACKER_EVENT_STARTED = 2 +TRACKER_EVENT_STOPPED = 3 +TRACKER_ADDR_IPV4 = 4 +TRACKER_ADDR_IPV6 = 6 +TRACKER_MAX_PEERS = 256 +TRACKER_MAX_URL_DATA = 512 +DHT_MAX_TRANSACTION = 16 +DHT_MAX_TOKEN = 64 +DHT_MAX_NODES = 256 +DHT_MAX_ERROR = 128 +DHT_MSG_QUERY = 1 +DHT_MSG_RESPONSE = 2 +DHT_QUERY_GET_PEERS = 3 +DEFAULT_DHT_BOOTSTRAP = ( + ("router.bittorrent.com", 6881), + ("dht.transmissionbt.com", 6881), + ("router.utorrent.com", 6881), +) + + +class TrackerPeer(C.Structure): + _fields_ = [ + ("family", C.c_uint8), + ("addr", C.c_uint8 * 16), + ("port", C.c_uint16), + ("peer_id", C.c_uint8 * 20), + ("has_peer_id", C.c_uint8), + ] + + +class TrackerAnnounceRequest(C.Structure): + _fields_ = [ + ("info_hash", C.c_uint8 * 20), + ("peer_id", C.c_uint8 * 20), + ("port", C.c_uint16), + ("uploaded", C.c_uint64), + ("downloaded", C.c_uint64), + ("left", C.c_uint64), + ("numwant", C.c_int32), + ("key", C.c_uint32), + ("ip4", C.c_uint32), + ("event", C.c_int), + ("compact", C.c_uint8), + ("no_peer_id", C.c_uint8), + ("has_key", C.c_uint8), + ("has_ip4", C.c_uint8), + ("ip", C.c_char * 64), + ("tracker_id", C.c_char * 128), + ("url_data", C.c_char * TRACKER_MAX_URL_DATA), + ] + + +class TrackerAnnounceResponse(C.Structure): + _fields_ = [ + ("interval", C.c_uint32), + ("min_interval", C.c_uint32), + ("complete", C.c_uint32), + ("incomplete", C.c_uint32), + ("tracker_id", C.c_char_p), + ("peers", C.POINTER(TrackerPeer)), + ("peer_count", C.c_size_t), + ("compact", C.c_uint8), + ] + + +class DHTNode(C.Structure): + _fields_ = [ + ("id", C.c_uint8 * 20), + ("family", C.c_uint8), + ("addr", C.c_uint8 * 16), + ("port", C.c_uint16), + ] + + +class DHTMessage(C.Structure): + _fields_ = [ + ("type", C.c_int), + ("query", C.c_int), + ("transaction", C.c_uint8 * DHT_MAX_TRANSACTION), + ("transaction_len", C.c_size_t), + ("id", C.c_uint8 * 20), + ("target", C.c_uint8 * 20), + ("info_hash", C.c_uint8 * 20), + ("port", C.c_uint16), + ("implied_port", C.c_uint8), + ("want_ipv4", C.c_uint8), + ("want_ipv6", C.c_uint8), + ("token", C.c_uint8 * DHT_MAX_TOKEN), + ("token_len", C.c_size_t), + ("nodes", DHTNode * DHT_MAX_NODES), + ("node_count", C.c_size_t), + ("peers", TrackerPeer * TRACKER_MAX_PEERS), + ("peer_count", C.c_size_t), + ("error_code", C.c_int), + ("error_message", C.c_char * DHT_MAX_ERROR), + ] + + +@dataclass +class TrackerResult: + tracker: str + ok: bool + protocol: str + peers: list[tuple[str, int]] + interval: int = 0 + complete: int = 0 + incomplete: int = 0 + error: str = "" + elapsed_ms: float = 0.0 + + +@dataclass +class DHTResult: + peers: list[tuple[str, int]] + nodes_queried: int + nodes_discovered: int + elapsed_ms: float + error: str = "" + + +def _default_lib_path() -> str: + env = os.environ.get("TORRENT_TRACKER_LIB") + if env: + return env + root = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) + candidates = [ + os.path.join(root, "..", "torrent-tracker", "build", "libtorrenttracker.so"), + os.path.join(root, "build", "libtorrenttracker.so"), + ] + for path in candidates: + if os.path.exists(path): + return os.path.abspath(path) + return os.path.abspath(candidates[0]) + + +def _load(lib_path: str | None = None) -> C.CDLL: + lib = C.CDLL(lib_path or _default_lib_path()) + lib.tracker_http_write_announce_query.restype = C.c_int + lib.tracker_http_write_announce_query.argtypes = [ + C.POINTER(TrackerAnnounceRequest), C.c_char_p, C.c_size_t, C.POINTER(C.c_size_t)] + lib.tracker_http_parse_announce_response.restype = C.c_int + lib.tracker_http_parse_announce_response.argtypes = [ + C.c_void_p, C.c_size_t, C.POINTER(TrackerPeer), C.c_size_t, + C.POINTER(TrackerAnnounceResponse)] + lib.tracker_udp_write_connect_request.restype = C.c_int + lib.tracker_udp_write_connect_request.argtypes = [ + C.c_uint32, C.c_void_p, C.c_size_t, C.POINTER(C.c_size_t)] + lib.tracker_udp_parse_connect_response.restype = C.c_int + lib.tracker_udp_parse_connect_response.argtypes = [ + C.c_void_p, C.c_size_t, C.c_uint32, C.POINTER(C.c_uint64)] + lib.tracker_udp_write_announce_request.restype = C.c_int + lib.tracker_udp_write_announce_request.argtypes = [ + C.c_uint64, C.c_uint32, C.POINTER(TrackerAnnounceRequest), + C.c_void_p, C.c_size_t, C.POINTER(C.c_size_t)] + lib.tracker_udp_parse_announce_response.restype = C.c_int + lib.tracker_udp_parse_announce_response.argtypes = [ + C.c_void_p, C.c_size_t, C.c_uint32, C.c_int, C.POINTER(TrackerPeer), + C.c_size_t, C.POINTER(TrackerAnnounceResponse)] + lib.dht_write_get_peers_query.restype = C.c_int + lib.dht_write_get_peers_query.argtypes = [ + C.c_void_p, C.c_size_t, C.c_void_p, C.c_void_p, C.c_uint8, C.c_uint8, + C.c_void_p, C.c_size_t, C.POINTER(C.c_size_t)] + lib.dht_parse_message.restype = C.c_int + lib.dht_parse_message.argtypes = [C.c_void_p, C.c_size_t, C.POINTER(DHTMessage)] + lib.dht_write_peers_response.restype = C.c_int + lib.dht_write_peers_response.argtypes = [ + C.c_void_p, C.c_size_t, C.c_void_p, C.c_void_p, C.c_size_t, + C.POINTER(TrackerPeer), C.c_size_t, C.c_void_p, C.c_size_t, + C.POINTER(C.c_size_t)] + return lib + + +class TrackerClient: + def __init__(self, lib_path: str | None = None): + self.lib = _load(lib_path) + + def _request(self, meta, peer_id: bytes, port: int, key: int, numwant: int, + event: str) -> TrackerAnnounceRequest: + req = TrackerAnnounceRequest() + C.memset(C.byref(req), 0, C.sizeof(req)) + req.info_hash[:] = meta.info_hash + req.peer_id[:] = peer_id + req.port = port + req.left = meta.total_size + req.numwant = numwant + req.key = key + req.has_key = 1 + req.compact = 1 + req.no_peer_id = 1 + req.event = { + "completed": TRACKER_EVENT_COMPLETED, + "started": TRACKER_EVENT_STARTED, + "stopped": TRACKER_EVENT_STOPPED, + }.get(event, TRACKER_EVENT_NONE) + return req + + @staticmethod + def _peers(peers, count: int) -> list[tuple[str, int]]: + out: list[tuple[str, int]] = [] + for i in range(count): + p = peers[i] + if p.family == TRACKER_ADDR_IPV4: + host = socket.inet_ntop(socket.AF_INET, bytes(p.addr[:4])) + elif p.family == TRACKER_ADDR_IPV6: + host = socket.inet_ntop(socket.AF_INET6, bytes(p.addr[:16])) + else: + continue + out.append((host, int(p.port))) + return out + + def announce_http(self, url: str, meta, peer_id: bytes, port: int, key: int, + numwant: int, event: str, timeout: float) -> TrackerResult: + start = time.monotonic() + try: + req = self._request(meta, peer_id, port, key, numwant, event) + query = C.create_string_buffer(2048) + written = C.c_size_t() + rc = self.lib.tracker_http_write_announce_query( + C.byref(req), query, C.sizeof(query), C.byref(written)) + if rc != TRACKER_OK: + raise RuntimeError(f"tracker_http_write_announce_query failed: {rc}") + + parts = urlsplit(url) + q = parts.query + suffix = query.value.decode("ascii") + q = f"{q}&{suffix}" if q else suffix + announce_url = urlunsplit((parts.scheme, parts.netloc, parts.path, q, + parts.fragment)) + request = Request(announce_url, + headers={"User-Agent": "torrent-peer/0.1"}) + ctx = ssl.create_default_context() + with urlopen(request, timeout=timeout, context=ctx) as resp: + raw = resp.read(2 * 1024 * 1024) + + raw_buf = C.create_string_buffer(raw, len(raw)) + out_peers = (TrackerPeer * TRACKER_MAX_PEERS)() + parsed = TrackerAnnounceResponse() + rc = self.lib.tracker_http_parse_announce_response( + raw_buf, len(raw), out_peers, TRACKER_MAX_PEERS, C.byref(parsed)) + if rc != TRACKER_OK: + raise RuntimeError(f"tracker_http_parse_announce_response failed: {rc}") + return TrackerResult( + tracker=url, + ok=True, + protocol=parts.scheme, + peers=self._peers(out_peers, parsed.peer_count), + interval=int(parsed.interval), + complete=int(parsed.complete), + incomplete=int(parsed.incomplete), + elapsed_ms=(time.monotonic() - start) * 1000.0, + ) + except Exception as exc: + return TrackerResult(url, False, "http", [], error=str(exc), + elapsed_ms=(time.monotonic() - start) * 1000.0) + + @staticmethod + def _url_data(url: str) -> bytes: + parts = urlsplit(url) + data = (parts.path or "").encode("utf-8") + if parts.query: + data += b"?" + parts.query.encode("utf-8") + return data[:TRACKER_MAX_URL_DATA - 1] + + @staticmethod + def _roundtrip(sock: socket.socket, packet: bytes, txid: int, + timeout: float) -> bytes: + deadline = time.monotonic() + timeout + delay = min(timeout, 1.0) + while True: + sock.send(packet) + remaining = deadline - time.monotonic() + if remaining <= 0: + raise TimeoutError("UDP tracker timed out") + sock.settimeout(min(delay, remaining)) + try: + raw = sock.recv(65535) + except socket.timeout: + delay = min(delay * 2.0, 8.0) + continue + if len(raw) >= 8 and struct.unpack_from("!I", raw, 4)[0] == txid: + return raw + + def announce_udp(self, url: str, meta, peer_id: bytes, port: int, key: int, + numwant: int, event: str, timeout: float) -> TrackerResult: + start = time.monotonic() + parts = urlsplit(url) + if not parts.hostname: + return TrackerResult(url, False, "udp", [], error="missing UDP tracker host") + tracker_port = parts.port or 80 + try: + infos = socket.getaddrinfo(parts.hostname, tracker_port, 0, + socket.SOCK_DGRAM) + last_error: Exception | None = None + for family, socktype, proto, _canon, sockaddr in infos: + if family not in (socket.AF_INET, socket.AF_INET6): + continue + try: + with socket.socket(family, socktype, proto) as sock: + sock.connect(sockaddr) + + txid = random.getrandbits(32) + buf = C.create_string_buffer(2048) + written = C.c_size_t() + rc = self.lib.tracker_udp_write_connect_request( + txid, buf, C.sizeof(buf), C.byref(written)) + if rc != TRACKER_OK: + raise RuntimeError(f"connect request failed: {rc}") + raw = self._roundtrip(sock, buf.raw[:written.value], txid, timeout) + conn_id = C.c_uint64() + raw_buf = C.create_string_buffer(raw, len(raw)) + rc = self.lib.tracker_udp_parse_connect_response( + raw_buf, len(raw), txid, C.byref(conn_id)) + if rc != TRACKER_OK: + raise RuntimeError(f"connect response failed: {rc}") + + req = self._request(meta, peer_id, port, key, numwant, event) + url_data = self._url_data(url) + if url_data: + req.url_data = url_data + txid = random.getrandbits(32) + rc = self.lib.tracker_udp_write_announce_request( + conn_id.value, txid, C.byref(req), buf, C.sizeof(buf), + C.byref(written)) + if rc != TRACKER_OK: + raise RuntimeError(f"announce request failed: {rc}") + raw = self._roundtrip(sock, buf.raw[:written.value], txid, timeout) + raw_buf = C.create_string_buffer(raw, len(raw)) + out_peers = (TrackerPeer * TRACKER_MAX_PEERS)() + parsed = TrackerAnnounceResponse() + tracker_family = (TRACKER_ADDR_IPV6 if family == socket.AF_INET6 + else TRACKER_ADDR_IPV4) + rc = self.lib.tracker_udp_parse_announce_response( + raw_buf, len(raw), txid, tracker_family, out_peers, + TRACKER_MAX_PEERS, C.byref(parsed)) + if rc != TRACKER_OK: + raise RuntimeError(f"announce response failed: {rc}") + return TrackerResult( + tracker=url, + ok=True, + protocol="udp", + peers=self._peers(out_peers, parsed.peer_count), + interval=int(parsed.interval), + complete=int(parsed.complete), + incomplete=int(parsed.incomplete), + elapsed_ms=(time.monotonic() - start) * 1000.0, + ) + except Exception as exc: + last_error = exc + continue + raise RuntimeError(str(last_error or "no usable tracker address")) + except Exception as exc: + return TrackerResult(url, False, "udp", [], error=str(exc), + elapsed_ms=(time.monotonic() - start) * 1000.0) + + def announce(self, url: str, meta, peer_id: bytes, port: int, key: int, + numwant: int = 50, event: str = "started", + timeout: float = 8.0) -> TrackerResult: + scheme = urlsplit(url).scheme.lower() + if scheme in ("http", "https"): + return self.announce_http(url, meta, peer_id, port, key, numwant, + event, timeout) + if scheme == "udp": + return self.announce_udp(url, meta, peer_id, port, key, numwant, + event, timeout) + return TrackerResult(url, False, scheme or "unknown", [], + error=f"unsupported tracker scheme {scheme!r}") + + +class DHTClient: + def __init__(self, lib_path: str | None = None, + bootstrap: tuple[tuple[str, int], ...] = DEFAULT_DHT_BOOTSTRAP): + self.lib = _load(lib_path) + self.bootstrap = bootstrap + self.node_id = os.urandom(20) + self._tx = random.randrange(1, 0xffff) + + @staticmethod + def _peer_endpoint(peer: TrackerPeer) -> tuple[str, int] | None: + if peer.family == TRACKER_ADDR_IPV4: + host = socket.inet_ntop(socket.AF_INET, bytes(peer.addr[:4])) + elif peer.family == TRACKER_ADDR_IPV6: + host = socket.inet_ntop(socket.AF_INET6, bytes(peer.addr[:16])) + else: + return None + return host, int(peer.port) + + @staticmethod + def _node_endpoint(node: DHTNode) -> tuple[str, int] | None: + if node.family == TRACKER_ADDR_IPV4: + host = socket.inet_ntop(socket.AF_INET, bytes(node.addr[:4])) + elif node.family == TRACKER_ADDR_IPV6: + host = socket.inet_ntop(socket.AF_INET6, bytes(node.addr[:16])) + else: + return None + return host, int(node.port) + + def _next_tx(self) -> bytes: + self._tx = (self._tx + 1) & 0xffff + return self._tx.to_bytes(2, "big") + + def _get_peers_packet(self, info_hash: bytes, tx: bytes) -> bytes: + buf = C.create_string_buffer(2048) + written = C.c_size_t() + tx_buf = C.create_string_buffer(tx, len(tx)) + id_buf = C.create_string_buffer(self.node_id, len(self.node_id)) + hash_buf = C.create_string_buffer(info_hash, len(info_hash)) + rc = self.lib.dht_write_get_peers_query( + tx_buf, len(tx), id_buf, hash_buf, 1, 1, buf, C.sizeof(buf), + C.byref(written)) + if rc != TRACKER_OK: + raise RuntimeError(f"dht_write_get_peers_query failed: {rc}") + return buf.raw[:written.value] + + def lookup(self, info_hash: bytes, *, timeout: float = 6.0, + max_queries: int = 32, max_peers: int = 100) -> DHTResult: + start = time.monotonic() + deadline = start + timeout + peers: set[tuple[str, int]] = set() + queue: list[tuple[str, int]] = list(self.bootstrap) + seen_nodes: set[tuple[str, int]] = set() + queried = 0 + discovered = 0 + last_error = "" + + while queue and queried < max_queries and len(peers) < max_peers: + if time.monotonic() >= deadline: + break + host, port = queue.pop(0) + if (host, port) in seen_nodes: + continue + seen_nodes.add((host, port)) + queried += 1 + + remaining = max(0.05, deadline - time.monotonic()) + try: + infos = socket.getaddrinfo(host, port, 0, socket.SOCK_DGRAM) + except OSError as exc: + last_error = str(exc) + continue + + for family, socktype, proto, _canon, sockaddr in infos: + if family not in (socket.AF_INET, socket.AF_INET6): + continue + tx = self._next_tx() + packet = self._get_peers_packet(info_hash, tx) + try: + with socket.socket(family, socktype, proto) as sock: + sock.settimeout(min(1.0, remaining)) + sock.sendto(packet, sockaddr) + raw, _addr = sock.recvfrom(4096) + except OSError as exc: + last_error = str(exc) + continue + + msg = DHTMessage() + raw_buf = C.create_string_buffer(raw, len(raw)) + rc = self.lib.dht_parse_message(raw_buf, len(raw), C.byref(msg)) + if rc != TRACKER_OK: + last_error = f"dht_parse_message failed: {rc}" + continue + got_tx = bytes(msg.transaction[:msg.transaction_len]) + if got_tx != tx or msg.type != DHT_MSG_RESPONSE: + continue + + for i in range(int(msg.peer_count)): + ep = self._peer_endpoint(msg.peers[i]) + if ep: + peers.add(ep) + for i in range(int(msg.node_count)): + ep = self._node_endpoint(msg.nodes[i]) + if ep and ep not in seen_nodes and ep not in queue: + queue.append(ep) + discovered += 1 + break + + return DHTResult( + peers=sorted(peers), + nodes_queried=queried, + nodes_discovered=discovered, + elapsed_ms=(time.monotonic() - start) * 1000.0, + error="" if peers else last_error, + ) diff --git a/include/engine.h b/include/engine.h new file mode 100644 index 0000000..29baa76 --- /dev/null +++ b/include/engine.h @@ -0,0 +1,157 @@ +/* + * engine.h - Public ABI for the multi-peer download engine. + * + * The engine owns all peer connections via a fixed pool of event loops (one OS + * thread each). Torrents are pinned to a loop ("affinity"); every connection of + * a torrent lives on that loop, so each loop thread is the sole owner of its + * connections, its arena, and its torrents' piece state. The hot path + * (recv -> parse -> handoff -> schedule) is therefore lock-free. + * + * Data plane: each loop owns one arena slab and an SPSC ready-ring; the consumer + * thread drains completed blocks across loops with engine_poll_ready() and + * returns spent slots with engine_release_slot(). Control plane (add torrent, + * add peer, set priorities) is delivered to the owning loop via a command queue. + * + * The legacy single-peer peer_* API (peer.h) is a thin wrapper over a 1-loop / + * 1-torrent / 1-peer engine. + */ +#ifndef TORRENT_ENGINE_H +#define TORRENT_ENGINE_H + +#include +#include + +#ifdef __cplusplus +extern "C" { +#endif + +#define PEER_BLOCK_SIZE 16384u /* BitTorrent block size */ + +/* Connection/torrent state, mirrored to status. */ +typedef enum { + PEER_STATE_IDLE = 0, + PEER_STATE_CONNECTING = 1, + PEER_STATE_HANDSHAKE = 2, + PEER_STATE_CHOKED = 3, + PEER_STATE_RUNNING = 4, + PEER_STATE_STOPPED = 5, + PEER_STATE_ERROR = 6 +} peer_state; + +typedef enum { + PEER_OK = 0, + PEER_ERR_CONNECT = 1, + PEER_ERR_HANDSHAKE = 2, + PEER_ERR_CLOSED = 3, + PEER_ERR_PROTOCOL = 4, + PEER_ERR_IO = 5, + PEER_ERR_NOMEM = 6 +} peer_error; + +typedef struct { + uint32_t loop_count; /* event-loop threads (0 => min(ncpu, 8)) */ + uint32_t slots_per_loop; /* arena depth per loop in 16 KiB slots (0=>def)*/ + uint32_t max_pipeline; /* per-connection outstanding-request cap (0=>def)*/ + uint32_t request_timeout_ms;/* re-request a block after this long (0=>def) */ + uint32_t recv_buffer_bytes; /* SO_RCVBUF override; 0 => kernel autotuning */ + uint32_t encryption; /* 0 = plaintext only; 1 = MSE, offer RC4 + + * plaintext (most compatible); 2 = MSE, require + * RC4 (refuse plaintext) */ + uint32_t utp; /* 0 = TCP; 1 = µTP (UDP). May combine with + * encryption to run MSE over µTP. */ + uint32_t connect_timeout_ms;/* drop a peer that hasn't finished connecting + + * handshaking within this long (0 => 10000). + * Reclaims unreachable/silent peers instead of + * leaving them stuck. */ + uint32_t fallback; /* 1 => if a peer fails before the BitTorrent + * handshake, retry the same endpoint over the + * next transport/encryption combo (TCP+MSE -> + * TCP+plain -> µTP+MSE -> µTP+plain, ordered by + * the utp/encryption prefs above). Reaches far + * more of a real swarm. 0 => single attempt. */ +} engine_config; + +/* + * One delivered block. Payload lives at: + * (uint8_t*)engine_arena_base(e, loop) + (uint64_t)slot * PEER_BLOCK_SIZE + * valid until returned via engine_release_slot(e, loop, slot). + */ +typedef struct { + uint32_t torrent; /* torrent id this block belongs to */ + uint32_t piece; + uint32_t begin; + uint32_t len; + uint32_t loop; /* arena that holds the slot */ + uint32_t slot; /* slot index within that loop's arena */ +} engine_block; + +/* Aggregated status for one torrent. */ +typedef struct { + int32_t state; /* peer_state (best connection's state) */ + int32_t error; /* peer_error of a failed connection, if any */ + uint64_t bytes_received; + uint64_t blocks_received; + uint32_t peers; /* connections attached (incl. failed) */ + uint32_t peers_connected; /* handshake completed (choked or running) */ + uint32_t peers_failed; /* connections that errored out */ + uint32_t outstanding; /* in-flight requests summed across peers */ + uint32_t free_slots; /* free arena slots on the torrent's loop */ + uint32_t pipeline_target; /* summed adaptive target across peers */ + double rate_bps; /* summed download rate */ + double rtt_min_ms; /* smallest observed request->block RTT */ +} torrent_status; + +typedef struct engine engine; + +/* Lifecycle. */ +engine *engine_create(const engine_config *cfg); +void engine_destroy(engine *e); + +/* Register a torrent. Returns its id (>= 0) or -1 on error. The arrays are + * copied. piece_length/total_size/num_pieces describe the torrent geometry. */ +int32_t engine_add_torrent(engine *e, const uint8_t info_hash[20], + const uint8_t peer_id[20], uint64_t piece_length, + uint64_t total_size, uint32_t num_pieces); + +/* Open a connection to a peer for a torrent (ip = dotted-quad or IPv6 literal). */ +int engine_add_peer(engine *e, uint32_t torrent_id, const char *ip, uint16_t port); + +/* Priority vector / single priority / re-arm — see peer.h docs for semantics. */ +int engine_set_priorities(engine *e, uint32_t torrent_id, + const uint8_t *priorities, uint32_t count); +int engine_set_priority(engine *e, uint32_t torrent_id, uint32_t piece, + uint8_t priority); +int engine_request_piece(engine *e, uint32_t torrent_id, uint32_t piece); + +/* Engine-wide download throttle in bytes/sec; 0 = unlimited (default). Bounds + * the aggregate receive rate by gating outgoing block requests. Safe to call at + * any time from any thread. */ +void engine_set_download_rate(engine *e, uint64_t bytes_per_sec); + +/* Data plane (single consumer thread). */ +uint32_t engine_poll_ready(engine *e, engine_block *out, uint32_t max); +void engine_release_slot(engine *e, uint32_t loop, uint32_t slot); +int engine_wait(engine *e, int timeout_ms); + +void *engine_arena_base(engine *e, uint32_t loop); +uint64_t engine_arena_bytes(engine *e, uint32_t loop); +uint32_t engine_loop_count(engine *e); + +void engine_torrent_status(engine *e, uint32_t torrent_id, torrent_status *out); + +/* Diagnostic: write a human-readable dump of one torrent's piece-selection and + * per-connection state to `out`. Reports, for every still-wanted piece + * (priority > 0), whether a peer has claimed it (requested), how many connected + * peers advertise it (availability), and how many in-flight block requests it + * has across all peers — the data needed to tell apart a stuck piece no peer + * has, one a dead/idle peer claimed but never delivered, and one the scheduler + * is simply not picking. Reads loop-owned state from the caller's thread without + * locking (like engine_torrent_status), so it is a best-effort snapshot meant + * for debugging, not control. */ +void engine_dump_torrent(engine *e, uint32_t torrent_id, FILE *out); + +#ifdef __cplusplus +} +#endif + +#endif /* TORRENT_ENGINE_H */ diff --git a/include/peer.h b/include/peer.h new file mode 100644 index 0000000..03e16e6 --- /dev/null +++ b/include/peer.h @@ -0,0 +1,134 @@ +/* + * peer.h - Legacy single-peer ABI, now a thin compatibility shim over the + * multi-peer engine (engine.h). A peer_handle is a private 1-loop / 1-torrent / + * 1-peer engine; the semantics below are unchanged so existing callers and + * tests keep working. New code should use engine.h directly. + */ +#ifndef TORRENT_PEER_H +#define TORRENT_PEER_H + +#include + +#include "engine.h" /* PEER_BLOCK_SIZE, peer_state, peer_error */ + +#ifdef __cplusplus +extern "C" { +#endif + +/* + * Immutable configuration passed to peer_create(). The arrays are copied, so + * the caller need not keep them alive afterwards. + */ +typedef struct { + uint8_t info_hash[20]; /* torrent info-hash (BitTorrent v1, SHA-1) */ + uint8_t peer_id[20]; /* our 20-byte peer id */ + uint64_t piece_length; /* bytes per piece (last piece may be shorter) */ + uint64_t total_size; /* total torrent payload size */ + uint32_t num_pieces; /* number of pieces */ + uint32_t num_slots; /* arena depth in 16 KiB slots (0 => default) */ + uint32_t max_pipeline; /* cap on outstanding block requests (0 => def) */ + uint32_t request_timeout_ms; /* re-request a block after this long with no + * reply (0 => default). Guards against silent + * request drops. */ + uint32_t recv_buffer_bytes; /* SO_RCVBUF override; 0 => leave kernel + * autotuning alone (recommended). */ +} peer_config; + +/* + * One received block. The payload lives at: + * (uint8_t*)peer_arena_base(h) + (uint64_t)slot * PEER_BLOCK_SIZE + * and stays valid until the slot is returned with peer_release_slot(). + * 16 bytes, trivially copyable, no pointers (ABI/relocation friendly). + */ +typedef struct { + uint32_t piece; /* piece index */ + uint32_t begin; /* byte offset of this block within the piece */ + uint32_t len; /* block length in bytes (<= PEER_BLOCK_SIZE) */ + uint32_t slot; /* arena slot holding the payload */ +} block_desc; + +/* Snapshot of peer progress; filled by peer_get_status(). */ +typedef struct { + int32_t state; /* peer_state */ + int32_t error; /* peer_error */ + uint64_t bytes_received; /* total payload bytes delivered to ready ring*/ + uint64_t blocks_received; /* total blocks delivered */ + uint32_t outstanding; /* requests sent but not yet received */ + uint32_t free_slots; /* arena slots currently available */ + uint32_t pipeline_target; /* current adaptive in-flight target (blocks) */ + double rate_bps; /* EWMA download rate, bytes/sec */ + double rtt_min_ms; /* smallest observed request->block RTT */ +} peer_status; + +typedef struct peer_handle peer_handle; + +/* Lifecycle ------------------------------------------------------------- */ + +/* Allocate a peer and its arena/rings. Returns NULL on bad config or OOM. */ +peer_handle *peer_create(const peer_config *cfg); + +/* Connect + handshake on a new network thread. ip is dotted-quad IPv4. + * Returns 0 if the thread launched, negative on immediate failure. The actual + * connection result surfaces asynchronously via peer_get_status(). */ +int peer_start(peer_handle *h, const char *ip, uint16_t port); + +/* + * Piece selection is priority-driven. The harness supplies one priority byte + * per piece; the peer always works on the highest-priority piece that + * (a) the remote peer actually HAS (per its bitfield/have messages), and + * (b) has not already been fully requested, + * breaking ties toward the lowest piece index. Priority 0 means "do not + * request". Re-evaluation happens at every piece boundary, so updating + * priorities while running steers the download live (sequential, rarest-first, + * deadline ramps, or any mix). The peer never assumes the remote has a piece it + * has not advertised, which is the key fix over a fixed in-order schedule. + */ + +/* Replace the whole priority vector. `count` must equal num_pieces. Does not + * touch the "already requested" state. Thread-safe. Returns 0, or negative if + * count != num_pieces. */ +int peer_set_priorities(peer_handle *h, const uint8_t *priorities, uint32_t count); + +/* Update one piece's priority. Thread-safe. Negative if out of range. */ +int peer_set_priority(peer_handle *h, uint32_t piece_index, uint8_t priority); + +/* Re-arm a piece for (re-)download, clearing its internal "already requested" + * mark so it becomes selectable again. Newly created peers start fully armed, + * so this is only needed to retry a piece that failed hash verification. + * Thread-safe. Negative if out of range. */ +int peer_request_piece(peer_handle *h, uint32_t piece_index); + +/* Stop the network thread and close the socket. Idempotent. */ +void peer_stop(peer_handle *h); + +/* Free the peer and its arena. The arena pointer is invalid afterwards. */ +void peer_destroy(peer_handle *h); + +/* Data plane ------------------------------------------------------------ */ + +/* Base of the block arena. Wrap [base, base+peer_arena_bytes()) in a Python + * memoryview once and slice it per block for zero-copy reads. */ +void *peer_arena_base(const peer_handle *h); +uint64_t peer_arena_bytes(const peer_handle *h); + +/* Drain up to max completed blocks into out[]. Returns the count (0..max), + * never blocks. Call from the single harness consumer thread. */ +uint32_t peer_poll_ready(peer_handle *h, block_desc *out, uint32_t max); + +/* Return a consumed slot so the peer can reuse it. This is what unblocks new + * requests (credit-based flow control). */ +void peer_release_slot(peer_handle *h, uint32_t slot); + +/* Block until ready blocks are available or timeout_ms elapses (negative = + * wait forever). Returns 1 if readable, 0 on timeout, negative on error. + * Optional: callers may instead poll peer_poll_ready() directly. */ +int peer_wait(peer_handle *h, int timeout_ms); + +/* Fill *out with a status snapshot. */ +void peer_get_status(const peer_handle *h, peer_status *out); + +#ifdef __cplusplus +} +#endif + +#endif /* TORRENT_PEER_H */ diff --git a/interop/Dockerfile b/interop/Dockerfile new file mode 100644 index 0000000..b85979f --- /dev/null +++ b/interop/Dockerfile @@ -0,0 +1,32 @@ +FROM ubuntu:24.04 + +ENV DEBIAN_FRONTEND=noninteractive + +RUN apt-get update && apt-get install -y --no-install-recommends \ + aria2 \ + build-essential \ + ca-certificates \ + cmake \ + deluge-console \ + deluged \ + procps \ + python3 \ + python3-libtorrent \ + qbittorrent-nox \ + rtorrent \ + transmission-cli \ + transmission-daemon \ + && rm -rf /var/lib/apt/lists/* + +WORKDIR /work + +COPY CMakeLists.txt /work/CMakeLists.txt +COPY include /work/include +COPY src /work/src +COPY harness /work/harness +COPY interop /work/interop + +RUN cmake -S /work -B /work/build -DCMAKE_BUILD_TYPE=Release -DPEER_NATIVE=OFF \ + && cmake --build /work/build + +ENV PYTHONPATH=/work/harness:/work/interop diff --git a/interop/README.md b/interop/README.md new file mode 100644 index 0000000..1cac41e --- /dev/null +++ b/interop/README.md @@ -0,0 +1,91 @@ +# Offline Client Interop + +This harness starts several seed-only BitTorrent clients on a Docker Compose +network with `internal: true`, generates one private test torrent, +mounts the same data into every seeder, and drives this engine against each +client independently. + +It is meant to catch interoperability failures: bad handshakes, encryption/uTP +negotiation mistakes, malformed requests, or behavior that makes common clients +reject us. It is not a public-swarm or tracker test. + +## Clients + +Default matrix: + +- `libtorrent-plain-tcp` +- `libtorrent-mse-tcp` with RC4 required +- `libtorrent-utp` +- `libtorrent-utp-mse` +- `transmission-tcp` +- `transmission-utp` +- `transmission-mse-tcp` with encryption required +- `aria2-tcp` +- `qbittorrent-tcp` +- `deluge-tcp` +- `rtorrent-tcp` + +All clients run with DHT, PEX, local peer discovery, UPnP, and NAT-PMP disabled. +The fixture torrent includes the deterministic dummy announce URL +`http://fixture:9/announce` because rTorrent rejects trackerless torrents, but +the runner still injects peers directly. The Compose network is internal-only, +so containers cannot route to the internet during the test run. + +## Run + +From the repository root: + +```sh +docker compose -f interop/docker-compose.yml up --build \ + --abort-on-container-exit --exit-code-from runner +``` + +Useful overrides: + +```sh +FIXTURE_SIZE=128M TEST_TIMEOUT=180 \ +docker compose -f interop/docker-compose.yml up --build \ + --abort-on-container-exit --exit-code-from runner +``` + +Results are written to `interop/results/results.json`. + +Clean generated containers, networks, and the fixture volume: + +```sh +docker compose -f interop/docker-compose.yml down -v +``` + +## Run One Client + +The runner supports `--only`, but Compose still starts all default dependencies. +For focused debugging, run a shell after the stack is up: + +```sh +docker compose -f interop/docker-compose.yml run --rm runner \ + python3 /work/interop/run_matrix.py \ + --fixture /fixture \ + --clients /work/interop/clients.json \ + --results /results/one.json \ + --only transmission-tcp +``` + +## Adding Clients + +Add a seeder service to `docker-compose.yml`, disable all discovery/tracker/NAT +features for that client, expose it only on `torrent_lab`, then add an entry to +`clients.json`: + +```json +{ + "name": "new-client-tcp", + "host": "seed-new-client", + "port": 6900, + "utp": 0, + "encryption": 0, + "fallback": 0 +} +``` + +The engine currently needs a numeric IP, so the runner resolves the Compose DNS +name to IPv4 before calling `engine_add_peer`. diff --git a/interop/clients.json b/interop/clients.json new file mode 100644 index 0000000..e90acd0 --- /dev/null +++ b/interop/clients.json @@ -0,0 +1,90 @@ +[ + { + "name": "libtorrent-plain-tcp", + "host": "seed-libtorrent-plain", + "port": 6881, + "utp": 0, + "encryption": 0, + "fallback": 0 + }, + { + "name": "libtorrent-mse-tcp", + "host": "seed-libtorrent-mse", + "port": 6882, + "utp": 0, + "encryption": 2, + "fallback": 0 + }, + { + "name": "libtorrent-utp", + "host": "seed-libtorrent-utp", + "port": 6883, + "utp": 1, + "encryption": 0, + "fallback": 0 + }, + { + "name": "libtorrent-utp-mse", + "host": "seed-libtorrent-utp-mse", + "port": 6884, + "utp": 1, + "encryption": 1, + "fallback": 0 + }, + { + "name": "transmission-tcp", + "host": "seed-transmission", + "port": 6891, + "utp": 0, + "encryption": 0, + "fallback": 0 + }, + { + "name": "transmission-utp", + "host": "seed-transmission-utp", + "port": 6896, + "utp": 1, + "encryption": 0, + "fallback": 0 + }, + { + "name": "transmission-mse-tcp", + "host": "seed-transmission-mse", + "port": 6897, + "utp": 0, + "encryption": 2, + "fallback": 0 + }, + { + "name": "aria2-tcp", + "host": "seed-aria2", + "port": 6892, + "utp": 0, + "encryption": 0, + "fallback": 0 + }, + { + "name": "qbittorrent-tcp", + "host": "seed-qbittorrent", + "port": 6893, + "utp": 0, + "encryption": 0, + "fallback": 0 + }, + { + "name": "deluge-tcp", + "host": "seed-deluge", + "port": 6894, + "utp": 0, + "encryption": 0, + "fallback": 0 + }, + { + "name": "rtorrent-tcp", + "host": "seed-rtorrent", + "port": 6895, + "utp": 0, + "encryption": 0, + "fallback": 0 + } +] diff --git a/interop/common.py b/interop/common.py new file mode 100644 index 0000000..40524eb --- /dev/null +++ b/interop/common.py @@ -0,0 +1,63 @@ +from __future__ import annotations + +import hashlib +import json +import os +import socket +import time +import sys + +ROOT = os.environ.get("TORRENT_PEER_ROOT", "/work") +sys.path.insert(0, os.path.join(ROOT, "harness")) + +from torrent_meta import Metadata, load_metadata # noqa: E402 + + +def parse_size(text: str) -> int: + s = text.strip().upper() + mult = 1 + if s[-1:] in ("K", "M", "G"): + mult = {"K": 1024, "M": 1024**2, "G": 1024**3}[s[-1]] + s = s[:-1] + return int(s) * mult + + +def file_sha1(path: str) -> str: + h = hashlib.sha1() + with open(path, "rb") as f: + for chunk in iter(lambda: f.read(1024 * 1024), b""): + h.update(chunk) + return h.hexdigest() + + +def write_manifest(path: str, **items) -> None: + tmp = path + ".tmp" + with open(tmp, "w", encoding="utf-8") as f: + json.dump(items, f, indent=2, sort_keys=True) + f.write("\n") + os.replace(tmp, path) + + +def wait_for_tcp(host: str, port: int, timeout: float = 30.0) -> None: + deadline = time.time() + timeout + last_error = None + while time.time() < deadline: + try: + with socket.create_connection((host, port), timeout=1.0): + return + except OSError as exc: + last_error = exc + time.sleep(0.25) + raise TimeoutError(f"timed out waiting for {host}:{port}: {last_error}") + + +def resolve_ipv4(host: str) -> str: + infos = socket.getaddrinfo(host, None, socket.AF_INET, socket.SOCK_STREAM) + if not infos: + raise OSError(f"no IPv4 address for {host}") + return infos[0][4][0] + + +def touch_ready(path: str = "/tmp/seed-ready") -> None: + with open(path, "w", encoding="utf-8") as f: + f.write("ready\n") diff --git a/interop/docker-compose.yml b/interop/docker-compose.yml new file mode 100644 index 0000000..fc0c75b --- /dev/null +++ b/interop/docker-compose.yml @@ -0,0 +1,379 @@ +name: torrent-peer-interop + +services: + fixture: + build: + context: .. + dockerfile: interop/Dockerfile + command: + - /bin/sh + - -c + - > + python3 /work/interop/make_fixture.py --out /fixture --size ${FIXTURE_SIZE:-32M} + && touch /tmp/fixture-ready + && tail -f /dev/null + volumes: + - fixture:/fixture + networks: + - torrent_lab + healthcheck: + test: ["CMD", "test", "-f", "/tmp/fixture-ready"] + interval: 2s + timeout: 1s + retries: 30 + + seed-libtorrent-plain: + build: + context: .. + dockerfile: interop/Dockerfile + depends_on: + fixture: + condition: service_healthy + command: + - python3 + - /work/interop/seed_libtorrent.py + - --torrent + - /fixture/test.torrent + - --data + - /fixture + - --port + - "6881" + - --mode + - plain + volumes: + - fixture:/fixture:ro + networks: + - torrent_lab + healthcheck: + test: ["CMD", "test", "-f", "/tmp/seed-ready"] + interval: 2s + timeout: 1s + retries: 30 + + seed-libtorrent-mse: + build: + context: .. + dockerfile: interop/Dockerfile + depends_on: + fixture: + condition: service_healthy + command: + - python3 + - /work/interop/seed_libtorrent.py + - --torrent + - /fixture/test.torrent + - --data + - /fixture + - --port + - "6882" + - --mode + - mse + volumes: + - fixture:/fixture:ro + networks: + - torrent_lab + healthcheck: + test: ["CMD", "test", "-f", "/tmp/seed-ready"] + interval: 2s + timeout: 1s + retries: 30 + + seed-libtorrent-utp: + build: + context: .. + dockerfile: interop/Dockerfile + depends_on: + fixture: + condition: service_healthy + command: + - python3 + - /work/interop/seed_libtorrent.py + - --torrent + - /fixture/test.torrent + - --data + - /fixture + - --port + - "6883" + - --mode + - utp + volumes: + - fixture:/fixture:ro + networks: + - torrent_lab + healthcheck: + test: ["CMD", "test", "-f", "/tmp/seed-ready"] + interval: 2s + timeout: 1s + retries: 30 + + seed-libtorrent-utp-mse: + build: + context: .. + dockerfile: interop/Dockerfile + depends_on: + fixture: + condition: service_healthy + command: + - python3 + - /work/interop/seed_libtorrent.py + - --torrent + - /fixture/test.torrent + - --data + - /fixture + - --port + - "6884" + - --mode + - utp-mse + volumes: + - fixture:/fixture:ro + networks: + - torrent_lab + healthcheck: + test: ["CMD", "test", "-f", "/tmp/seed-ready"] + interval: 2s + timeout: 1s + retries: 30 + + seed-transmission: + build: + context: .. + dockerfile: interop/Dockerfile + depends_on: + fixture: + condition: service_healthy + command: + - python3 + - /work/interop/seed_transmission.py + - --torrent + - /fixture/test.torrent + - --data + - /fixture + - --peer-port + - "6891" + - --rpc-port + - "9091" + volumes: + - fixture:/fixture:ro + networks: + - torrent_lab + healthcheck: + test: ["CMD", "test", "-f", "/tmp/seed-ready"] + interval: 2s + timeout: 1s + retries: 45 + + seed-transmission-utp: + build: + context: .. + dockerfile: interop/Dockerfile + depends_on: + fixture: + condition: service_healthy + command: + - python3 + - /work/interop/seed_transmission.py + - --torrent + - /fixture/test.torrent + - --data + - /fixture + - --peer-port + - "6896" + - --rpc-port + - "9092" + - --utp + volumes: + - fixture:/fixture:ro + networks: + - torrent_lab + healthcheck: + test: ["CMD", "test", "-f", "/tmp/seed-ready"] + interval: 2s + timeout: 1s + retries: 45 + + seed-transmission-mse: + build: + context: .. + dockerfile: interop/Dockerfile + depends_on: + fixture: + condition: service_healthy + command: + - python3 + - /work/interop/seed_transmission.py + - --torrent + - /fixture/test.torrent + - --data + - /fixture + - --peer-port + - "6897" + - --rpc-port + - "9093" + - --encryption + - required + volumes: + - fixture:/fixture:ro + networks: + - torrent_lab + healthcheck: + test: ["CMD", "test", "-f", "/tmp/seed-ready"] + interval: 2s + timeout: 1s + retries: 45 + + seed-aria2: + build: + context: .. + dockerfile: interop/Dockerfile + depends_on: + fixture: + condition: service_healthy + command: + - python3 + - /work/interop/seed_aria2.py + - --torrent + - /fixture/test.torrent + - --data + - /fixture + - --port + - "6892" + volumes: + - fixture:/fixture:ro + networks: + - torrent_lab + healthcheck: + test: ["CMD", "test", "-f", "/tmp/seed-ready"] + interval: 2s + timeout: 1s + retries: 45 + + seed-qbittorrent: + build: + context: .. + dockerfile: interop/Dockerfile + depends_on: + fixture: + condition: service_healthy + command: + - python3 + - /work/interop/seed_qbittorrent.py + - --torrent + - /fixture/test.torrent + - --data + - /fixture + - --port + - "6893" + volumes: + - fixture:/fixture:ro + networks: + - torrent_lab + healthcheck: + test: ["CMD", "test", "-f", "/tmp/seed-ready"] + interval: 2s + timeout: 1s + retries: 45 + + seed-deluge: + build: + context: .. + dockerfile: interop/Dockerfile + depends_on: + fixture: + condition: service_healthy + command: + - python3 + - /work/interop/seed_deluge.py + - --torrent + - /fixture/test.torrent + - --data + - /fixture + - --port + - "6894" + - --daemon-port + - "58846" + volumes: + - fixture:/fixture:ro + networks: + - torrent_lab + healthcheck: + test: ["CMD", "test", "-f", "/tmp/seed-ready"] + interval: 2s + timeout: 1s + retries: 45 + + seed-rtorrent: + build: + context: .. + dockerfile: interop/Dockerfile + depends_on: + fixture: + condition: service_healthy + command: + - python3 + - /work/interop/seed_rtorrent.py + - --torrent + - /fixture/test.torrent + - --data + - /fixture + - --port + - "6895" + volumes: + - fixture:/fixture:ro + networks: + - torrent_lab + healthcheck: + test: ["CMD", "test", "-f", "/tmp/seed-ready"] + interval: 2s + timeout: 1s + retries: 45 + + runner: + build: + context: .. + dockerfile: interop/Dockerfile + depends_on: + seed-libtorrent-plain: + condition: service_healthy + seed-libtorrent-mse: + condition: service_healthy + seed-libtorrent-utp: + condition: service_healthy + seed-libtorrent-utp-mse: + condition: service_healthy + seed-transmission: + condition: service_healthy + seed-transmission-utp: + condition: service_healthy + seed-transmission-mse: + condition: service_healthy + seed-aria2: + condition: service_healthy + seed-qbittorrent: + condition: service_healthy + seed-deluge: + condition: service_healthy + seed-rtorrent: + condition: service_healthy + command: + - python3 + - /work/interop/run_matrix.py + - --fixture + - /fixture + - --clients + - /work/interop/clients.json + - --results + - /results/results.json + - --timeout + - ${TEST_TIMEOUT:-90} + volumes: + - fixture:/fixture:ro + - ./results:/results + networks: + - torrent_lab + +volumes: + fixture: + +networks: + torrent_lab: + internal: true diff --git a/interop/make_fixture.py b/interop/make_fixture.py new file mode 100644 index 0000000..8bf0b92 --- /dev/null +++ b/interop/make_fixture.py @@ -0,0 +1,76 @@ +from __future__ import annotations + +import argparse +import hashlib +import os + +import libtorrent as lt + +from common import file_sha1, parse_size, write_manifest + + +def deterministic_bytes(offset: int, size: int) -> bytes: + out = bytearray() + counter = offset // 32 + while len(out) < size: + out.extend(hashlib.sha256(f"torrent-peer-interop:{counter}".encode()).digest()) + counter += 1 + return bytes(out[:size]) + + +def write_data(path: str, size: int) -> None: + with open(path, "wb") as f: + off = 0 + while off < size: + n = min(1024 * 1024, size - off) + f.write(deterministic_bytes(off, n)) + off += n + + +def make_torrent(root: str, size: int, piece_size: int) -> str: + os.makedirs(root, exist_ok=True) + data_path = os.path.join(root, "data.bin") + torrent_path = os.path.join(root, "test.torrent") + write_data(data_path, size) + + fs = lt.file_storage() + lt.add_files(fs, data_path) + t = lt.create_torrent(fs, piece_size=piece_size) + t.set_priv(True) + t.add_tracker("http://fixture:9/announce") + lt.set_piece_hashes(t, root) + with open(torrent_path, "wb") as f: + f.write(lt.bencode(t.generate())) + return torrent_path + + +def main() -> int: + ap = argparse.ArgumentParser(description="Create the offline interop fixture.") + ap.add_argument("--out", required=True) + ap.add_argument("--size", default="32M") + ap.add_argument("--piece-size", default="256K") + args = ap.parse_args() + + size = parse_size(args.size) + piece_size = parse_size(args.piece_size) + torrent_path = make_torrent(args.out, size, piece_size) + data_path = os.path.join(args.out, "data.bin") + + ti = lt.torrent_info(torrent_path) + write_manifest( + os.path.join(args.out, "manifest.json"), + name=ti.name(), + size=size, + piece_size=piece_size, + num_pieces=ti.num_pieces(), + sha1=file_sha1(data_path), + announce="http://fixture:9/announce", + torrent=os.path.basename(torrent_path), + data=os.path.basename(data_path), + ) + print(f"fixture ready: {size} bytes, {ti.num_pieces()} pieces", flush=True) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/interop/run_matrix.py b/interop/run_matrix.py new file mode 100644 index 0000000..a745e42 --- /dev/null +++ b/interop/run_matrix.py @@ -0,0 +1,206 @@ +from __future__ import annotations + +import argparse +import hashlib +import json +import os +import secrets +import sys +import time +import traceback + +ROOT = "/work" +sys.path.insert(0, os.path.join(ROOT, "harness")) + +from common import load_metadata, resolve_ipv4, write_manifest # noqa: E402 +from engine_ffi import Engine, EngineConfig, ERROR_NAMES, STATE_ERROR, STATE_NAMES # noqa: E402 + + +LIB = "/work/build/libtorrentpeer.so" + + +def make_peer_id() -> bytes: + return b"-PC0001-" + secrets.token_bytes(12) + + +def load_clients(path: str, only: set[str] | None) -> list[dict]: + with open(path, "r", encoding="utf-8") as f: + clients = json.load(f) + if only: + clients = [c for c in clients if c["name"] in only] + if not clients: + raise ValueError("no clients selected") + return clients + + +def download_from_client(meta, client: dict, timeout: float, lib_path: str) -> dict: + ip = resolve_ipv4(client["host"]) + port = int(client["port"]) + cfg = EngineConfig( + loop_count=1, + slots_per_loop=int(client.get("slots_per_loop", 1024)), + max_pipeline=int(client.get("max_pipeline", 256)), + request_timeout_ms=int(client.get("request_timeout_ms", 10000)), + encryption=int(client.get("encryption", 0)), + utp=int(client.get("utp", 0)), + connect_timeout_ms=int(client.get("connect_timeout_ms", 5000)), + fallback=int(client.get("fallback", 0)), + ) + + started = time.time() + buffers = [bytearray(meta.piece_len(i)) for i in range(meta.num_pieces)] + received = [0] * meta.num_pieces + done = bytearray(meta.num_pieces) + done_count = 0 + last_progress = started + last_bytes = 0 + status_snap = None + + with Engine(cfg, lib_path=lib_path, poll_batch=2048) as eng: + tid = eng.add_torrent( + meta.info_hash, + make_peer_id(), + meta.piece_length, + meta.total_size, + meta.num_pieces, + ) + eng.set_priorities(tid, [1] * meta.num_pieces) + eng.add_peer(tid, ip, port) + + while done_count < meta.num_pieces: + status_snap = eng.status(tid) + if status_snap.state == STATE_ERROR: + raise RuntimeError(f"engine error: {ERROR_NAMES[status_snap.error]}") + + descs = eng.poll_ready() + if not descs: + eng.wait(200) + now = time.time() + if status_snap.bytes_received != last_bytes: + last_bytes = status_snap.bytes_received + last_progress = now + if now - started > timeout or now - last_progress > timeout: + raise TimeoutError( + f"stalled after {now - started:.1f}s: " + f"{done_count}/{meta.num_pieces} pieces, " + f"state={STATE_NAMES[status_snap.state]}, " + f"connected={status_snap.peers_connected}, " + f"failed={status_snap.peers_failed}, " + f"outstanding={status_snap.outstanding}" + ) + continue + + for block in descs: + buf = buffers[block.piece] + buf[block.begin:block.begin + block.len] = eng.block_data( + block.loop, block.slot, block.len + ) + eng.release(block.loop, block.slot) + received[block.piece] += block.len + if not done[block.piece] and received[block.piece] >= meta.piece_len(block.piece): + digest = hashlib.sha1(bytes(buf)).digest() + if digest != meta.piece_hashes[block.piece]: + raise ValueError(f"piece {block.piece} hash mismatch") + done[block.piece] = 1 + done_count += 1 + eng.set_priority(tid, block.piece, 0) + + status_snap = eng.status(tid) + + full = hashlib.sha1() + for buf in buffers: + full.update(buf) + elapsed = time.time() - started + return { + "name": client["name"], + "ok": True, + "host": client["host"], + "ip": ip, + "port": port, + "utp": cfg.utp, + "encryption": cfg.encryption, + "fallback": cfg.fallback, + "pieces": done_count, + "bytes": meta.total_size, + "sha1": full.hexdigest(), + "elapsed_s": elapsed, + "mbps": (meta.total_size / 1e6 / elapsed) if elapsed > 0 else 0.0, + "peers_connected": int(status_snap.peers_connected if status_snap else 0), + "peers_failed": int(status_snap.peers_failed if status_snap else 0), + "state": STATE_NAMES[status_snap.state] if status_snap else "UNKNOWN", + "error": ERROR_NAMES[status_snap.error] if status_snap else "OK", + } + + +def main() -> int: + ap = argparse.ArgumentParser(description="Run offline client interop matrix.") + ap.add_argument("--fixture", required=True) + ap.add_argument("--clients", required=True) + ap.add_argument("--results", required=True) + ap.add_argument("--timeout", type=float, default=90.0) + ap.add_argument("--lib", default=LIB) + ap.add_argument("--only", action="append", + help="client name to run; may be repeated") + args = ap.parse_args() + + torrent_path = os.path.join(args.fixture, "test.torrent") + manifest_path = os.path.join(args.fixture, "manifest.json") + meta = load_metadata(torrent_path) + with open(manifest_path, "r", encoding="utf-8") as f: + manifest = json.load(f) + + clients = load_clients(args.clients, set(args.only or []) or None) + print(f"fixture: {meta.name}, {meta.total_size} bytes, " + f"{meta.num_pieces} pieces", flush=True) + print(f"clients: {', '.join(c['name'] for c in clients)}", flush=True) + + results = [] + for client in clients: + print(f"==> {client['name']} ({client['host']}:{client['port']})", flush=True) + try: + result = download_from_client(meta, client, args.timeout, args.lib) + if result["sha1"] != manifest["sha1"]: + raise ValueError( + f"full-file sha1 mismatch: got {result['sha1']} expected {manifest['sha1']}" + ) + print(f" ok: {result['bytes']/1e6:.1f} MB in " + f"{result['elapsed_s']:.2f}s ({result['mbps']:.1f} MB/s)", + flush=True) + except Exception as exc: + result = { + "name": client["name"], + "ok": False, + "host": client.get("host"), + "port": client.get("port"), + "utp": client.get("utp", 0), + "encryption": client.get("encryption", 0), + "fallback": client.get("fallback", 0), + "error": str(exc), + "traceback": traceback.format_exc(), + } + print(f" FAIL: {exc}", flush=True) + results.append(result) + + os.makedirs(os.path.dirname(args.results), exist_ok=True) + write_manifest( + args.results, + fixture=manifest, + timeout_s=args.timeout, + results=results, + passed=sum(1 for r in results if r["ok"]), + failed=sum(1 for r in results if not r["ok"]), + ) + + print("-" * 72) + for r in results: + if r["ok"]: + print(f"PASS {r['name']:<24} {r['mbps']:8.1f} MB/s " + f"{r['state']} enc={r['encryption']} utp={r['utp']}") + else: + print(f"FAIL {r['name']:<24} {r['error']}") + return 0 if all(r["ok"] for r in results) else 1 + + +if __name__ == "__main__": + raise SystemExit(main()) + diff --git a/interop/seed_aria2.py b/interop/seed_aria2.py new file mode 100644 index 0000000..e1f2dae --- /dev/null +++ b/interop/seed_aria2.py @@ -0,0 +1,66 @@ +from __future__ import annotations + +import argparse +import signal +import subprocess +import time + +from common import touch_ready, wait_for_tcp + + +def main() -> int: + ap = argparse.ArgumentParser(description="Seed a fixture with aria2.") + ap.add_argument("--torrent", required=True) + ap.add_argument("--data", required=True) + ap.add_argument("--port", type=int, required=True) + args = ap.parse_args() + + cmd = [ + "aria2c", + "--dir", args.data, + "--seed-time=1000000", + "--check-integrity=true", + "--allow-overwrite=false", + "--auto-file-renaming=false", + "--enable-dht=false", + "--enable-dht6=false", + "--enable-peer-exchange=false", + "--bt-enable-lpd=false", + "--listen-port", str(args.port), + "--dht-listen-port", str(args.port), + "--summary-interval=0", + args.torrent, + ] + proc = subprocess.Popen(cmd) + stop = False + + def _stop(signum, frame): + nonlocal stop + stop = True + proc.terminate() + + signal.signal(signal.SIGTERM, _stop) + signal.signal(signal.SIGINT, _stop) + + try: + wait_for_tcp("127.0.0.1", args.port, timeout=45) + if proc.poll() is not None: + return proc.returncode or 1 + # aria2 may still be checking files; the runner also has retry/timeout, + # so readiness here means the peer port is accepting connections. + touch_ready() + print(f"aria2 seeding on {args.port}", flush=True) + while not stop and proc.poll() is None: + time.sleep(1) + return proc.returncode or 0 + finally: + if proc.poll() is None: + proc.terminate() + try: + proc.wait(timeout=10) + except subprocess.TimeoutExpired: + proc.kill() + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/interop/seed_deluge.py b/interop/seed_deluge.py new file mode 100644 index 0000000..8e82b81 --- /dev/null +++ b/interop/seed_deluge.py @@ -0,0 +1,82 @@ +from __future__ import annotations + +import argparse +import os +import signal +import subprocess +import time + +from common import touch_ready, wait_for_tcp + + +def console(config: str, command: str, check: bool = True) -> subprocess.CompletedProcess: + return subprocess.run( + ["deluge-console", "-c", config, command], + text=True, + stdout=subprocess.PIPE, + stderr=subprocess.STDOUT, + check=check, + timeout=15, + ) + + +def main() -> int: + ap = argparse.ArgumentParser(description="Seed a fixture with Deluge.") + ap.add_argument("--torrent", required=True) + ap.add_argument("--data", required=True) + ap.add_argument("--port", type=int, required=True) + ap.add_argument("--daemon-port", type=int, default=58846) + args = ap.parse_args() + + config = "/tmp/deluge-config" + os.makedirs(config, exist_ok=True) + proc = subprocess.Popen([ + "deluged", + "-d", + "-c", config, + "-i", "0.0.0.0", + "-p", str(args.daemon_port), + "-L", "warning", + ]) + stop = False + + def _stop(signum, frame): + nonlocal stop + stop = True + proc.terminate() + + signal.signal(signal.SIGTERM, _stop) + signal.signal(signal.SIGINT, _stop) + + try: + wait_for_tcp("127.0.0.1", args.daemon_port, timeout=45) + settings = [ + ("listen_ports", f"({args.port}, {args.port})"), + ("random_port", "False"), + ("dht", "False"), + ("lsd", "False"), + ("upnp", "False"), + ("natpmp", "False"), + ("add_paused", "False"), + ("download_location", args.data), + ] + for key, value in settings: + console(config, f"config -s {key} {value}") + console(config, f"add -p {args.data} {args.torrent}") + wait_for_tcp("127.0.0.1", args.port, timeout=45) + touch_ready() + print(f"Deluge seeding on {args.port}", flush=True) + while not stop and proc.poll() is None: + time.sleep(1) + return proc.returncode or 0 + finally: + if proc.poll() is None: + proc.terminate() + try: + proc.wait(timeout=10) + except subprocess.TimeoutExpired: + proc.kill() + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/interop/seed_libtorrent.py b/interop/seed_libtorrent.py new file mode 100644 index 0000000..5d89a2d --- /dev/null +++ b/interop/seed_libtorrent.py @@ -0,0 +1,84 @@ +from __future__ import annotations + +import argparse +import os +import signal +import time + +import libtorrent as lt + +from common import touch_ready + + +def settings_for(mode: str, port: int) -> dict: + tcp = mode in ("plain", "mse") + utp = mode in ("utp", "utp-mse") + encrypted = mode in ("mse", "utp-mse") + settings = { + "listen_interfaces": f"0.0.0.0:{port}", + "enable_dht": False, + "enable_lsd": False, + "enable_upnp": False, + "enable_natpmp": False, + "enable_outgoing_tcp": tcp, + "enable_incoming_tcp": tcp, + "enable_outgoing_utp": utp, + "enable_incoming_utp": utp, + "announce_to_all_trackers": False, + "announce_to_all_tiers": False, + "alert_mask": 0, + } + if encrypted: + settings.update({ + "in_enc_policy": int(lt.enc_policy.forced), + "out_enc_policy": int(lt.enc_policy.forced), + "allowed_enc_level": int(lt.enc_level.rc4), + "prefer_rc4": True, + }) + else: + settings.update({ + "in_enc_policy": int(lt.enc_policy.disabled), + "out_enc_policy": int(lt.enc_policy.disabled), + }) + return settings + + +def main() -> int: + ap = argparse.ArgumentParser(description="Seed a fixture with libtorrent.") + ap.add_argument("--torrent", required=True) + ap.add_argument("--data", required=True) + ap.add_argument("--port", required=True, type=int) + ap.add_argument("--mode", required=True, choices=["plain", "mse", "utp", "utp-mse"]) + args = ap.parse_args() + + stop = False + + def _stop(signum, frame): + nonlocal stop + stop = True + + signal.signal(signal.SIGTERM, _stop) + signal.signal(signal.SIGINT, _stop) + + ses = lt.session(settings_for(args.mode, args.port)) + h = ses.add_torrent({ + "ti": lt.torrent_info(args.torrent), + "save_path": args.data, + "flags": lt.torrent_flags.seed_mode, + }) + deadline = time.time() + 60 + while time.time() < deadline and not h.status().is_seeding: + time.sleep(0.25) + if not h.status().is_seeding: + raise TimeoutError(f"libtorrent {args.mode} did not enter seed mode") + touch_ready() + print(f"libtorrent {args.mode} seeding on {args.port}", flush=True) + while not stop: + time.sleep(1) + ses.remove_torrent(h) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) + diff --git a/interop/seed_qbittorrent.py b/interop/seed_qbittorrent.py new file mode 100644 index 0000000..844edf2 --- /dev/null +++ b/interop/seed_qbittorrent.py @@ -0,0 +1,88 @@ +from __future__ import annotations + +import argparse +import os +import signal +import subprocess +import time + +from common import touch_ready, wait_for_tcp + + +def write_config(profile: str, data_dir: str, port: int) -> None: + cfg_dir = os.path.join(profile, "qBittorrent", "config") + os.makedirs(cfg_dir, exist_ok=True) + # qBittorrent stores settings in an INI-like file with escaped keys. + # These disable internet/discovery paths and fix the peer port. + with open(os.path.join(cfg_dir, "qBittorrent.conf"), "w", encoding="utf-8") as f: + f.write(f"""[BitTorrent] +Session\\AddTorrentPaused=false +Session\\BTProtocol=TCP +Session\\DHTEnabled=false +Session\\DefaultSavePath={data_dir} +Session\\DisableAutoTMMByDefault=true +Session\\LSDEnabled=false +Session\\PeXEnabled=false +Session\\Port={port} +Session\\QueueingSystemEnabled=false +Session\\UPnP=false + +[LegalNotice] +Accepted=true + +[Preferences] +WebUI\\Enabled=false +""") + + +def main() -> int: + ap = argparse.ArgumentParser(description="Seed a fixture with qBittorrent-nox.") + ap.add_argument("--torrent", required=True) + ap.add_argument("--data", required=True) + ap.add_argument("--port", type=int, required=True) + args = ap.parse_args() + + profile = "/tmp/qbt-profile" + write_config(profile, args.data, args.port) + cmd = [ + "qbittorrent-nox", + f"--profile={profile}", + "--configuration=interop", + f"--torrenting-port={args.port}", + f"--save-path={args.data}", + "--add-paused=false", + "--skip-dialog=true", + "--skip-hash-check", + args.torrent, + ] + proc = subprocess.Popen(cmd) + stop = False + + def _stop(signum, frame): + nonlocal stop + stop = True + proc.terminate() + + signal.signal(signal.SIGTERM, _stop) + signal.signal(signal.SIGINT, _stop) + + try: + wait_for_tcp("127.0.0.1", args.port, timeout=45) + if proc.poll() is not None: + return proc.returncode or 1 + touch_ready() + print(f"qBittorrent seeding on {args.port}", flush=True) + while not stop and proc.poll() is None: + time.sleep(1) + return proc.returncode or 0 + finally: + if proc.poll() is None: + proc.terminate() + try: + proc.wait(timeout=10) + except subprocess.TimeoutExpired: + proc.kill() + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/interop/seed_rtorrent.py b/interop/seed_rtorrent.py new file mode 100644 index 0000000..140250f --- /dev/null +++ b/interop/seed_rtorrent.py @@ -0,0 +1,80 @@ +from __future__ import annotations + +import argparse +import os +import signal +import subprocess +import time + +from common import touch_ready, wait_for_tcp + + +def write_rc(path: str, torrent: str, data: str, port: int, session: str) -> None: + os.makedirs(session, exist_ok=True) + with open(path, "w", encoding="utf-8") as f: + f.write(f""" +directory.default.set = {data} +session.path.set = {session} +network.port_range.set = {port}-{port} +network.port_random.set = no +dht.mode.set = disable +protocol.pex.set = no +trackers.use_udp.set = no +network.http.max_open.set = 0 +pieces.hash.on_completion.set = no +""") + + +def main() -> int: + ap = argparse.ArgumentParser(description="Seed a fixture with rTorrent.") + ap.add_argument("--torrent", required=True) + ap.add_argument("--data", required=True) + ap.add_argument("--port", type=int, required=True) + args = ap.parse_args() + + rc = "/tmp/rtorrent.rc" + session = "/tmp/rtorrent-session" + write_rc(rc, args.torrent, args.data, args.port, session) + env = os.environ.copy() + env.setdefault("TERM", "xterm") + command = f"rtorrent -n -o import={rc}" + proc = subprocess.Popen( + ["script", "-q", "-e", "-c", command, "/dev/null"], + env=env, + stdin=subprocess.PIPE, + ) + stop = False + + def _stop(signum, frame): + nonlocal stop + stop = True + proc.terminate() + subprocess.run(["pkill", "-TERM", "rtorrent"], check=False) + + signal.signal(signal.SIGTERM, _stop) + signal.signal(signal.SIGINT, _stop) + + try: + wait_for_tcp("127.0.0.1", args.port, timeout=45) + if proc.stdin: + proc.stdin.write(b"\x7f" + args.torrent.encode("utf-8") + b"\n") + proc.stdin.flush() + time.sleep(2) + if proc.poll() is not None: + return proc.returncode or 1 + touch_ready() + print(f"rTorrent seeding on {args.port}", flush=True) + while not stop: + if subprocess.run(["pgrep", "rtorrent"], stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL).returncode != 0: + return 1 + time.sleep(1) + return 0 + finally: + subprocess.run(["pkill", "-TERM", "rtorrent"], check=False) + if proc.poll() is None: + proc.terminate() + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/interop/seed_transmission.py b/interop/seed_transmission.py new file mode 100644 index 0000000..3128289 --- /dev/null +++ b/interop/seed_transmission.py @@ -0,0 +1,109 @@ +from __future__ import annotations + +import argparse +import json +import os +import shutil +import signal +import subprocess +import time + +from common import touch_ready, wait_for_tcp + + +def run_remote(rpc_port: int, *args: str, check: bool = True) -> subprocess.CompletedProcess: + return subprocess.run( + ["transmission-remote", f"127.0.0.1:{rpc_port}", *args], + text=True, + stdout=subprocess.PIPE, + stderr=subprocess.STDOUT, + check=check, + ) + + +def write_settings(config_dir: str, peer_port: int, rpc_port: int, data_dir: str) -> None: + os.makedirs(config_dir, exist_ok=True) + settings = { + "download-dir": data_dir, + "incomplete-dir-enabled": False, + "dht-enabled": False, + "pex-enabled": False, + "lpd-enabled": False, + "utp-enabled": False, + "port-forwarding-enabled": False, + "peer-port": peer_port, + "peer-port-random-on-start": False, + "rpc-enabled": True, + "rpc-bind-address": "127.0.0.1", + "rpc-port": rpc_port, + "rpc-whitelist-enabled": False, + "start-added-torrents": True, + "trash-original-torrent-files": False, + } + with open(os.path.join(config_dir, "settings.json"), "w", encoding="utf-8") as f: + json.dump(settings, f, indent=2, sort_keys=True) + + +def main() -> int: + ap = argparse.ArgumentParser(description="Seed a fixture with Transmission.") + ap.add_argument("--torrent", required=True) + ap.add_argument("--data", required=True) + ap.add_argument("--peer-port", type=int, required=True) + ap.add_argument("--rpc-port", type=int, required=True) + ap.add_argument("--utp", action="store_true") + ap.add_argument("--encryption", choices=["tolerated", "preferred", "required"], + default="tolerated") + args = ap.parse_args() + + config_dir = "/tmp/transmission-config" + shutil.rmtree(config_dir, ignore_errors=True) + write_settings(config_dir, args.peer_port, args.rpc_port, args.data) + + proc = subprocess.Popen(["transmission-daemon", "-f", "-g", config_dir]) + stop = False + + def _stop(signum, frame): + nonlocal stop + stop = True + proc.terminate() + + signal.signal(signal.SIGTERM, _stop) + signal.signal(signal.SIGINT, _stop) + + try: + wait_for_tcp("127.0.0.1", args.rpc_port, timeout=30) + run_remote(args.rpc_port, "--no-dht", "--no-pex", "--no-lpd", "--no-portmap") + run_remote(args.rpc_port, "--utp" if args.utp else "--no-utp") + run_remote(args.rpc_port, { + "tolerated": "--encryption-tolerated", + "preferred": "--encryption-preferred", + "required": "--encryption-required", + }[args.encryption]) + run_remote(args.rpc_port, "-a", args.torrent, "-w", args.data) + run_remote(args.rpc_port, "-t", "all", "--start") + wait_for_tcp("127.0.0.1", args.peer_port, timeout=30) + deadline = time.time() + 60 + while time.time() < deadline: + info = run_remote(args.rpc_port, "-t", "all", "-i", check=False).stdout + if "Percent Done: 100%" in info or "Seeding" in info: + touch_ready() + print(f"transmission seeding on {args.peer_port}", flush=True) + break + time.sleep(1) + else: + raise TimeoutError("Transmission did not report a complete seed") + + while not stop and proc.poll() is None: + time.sleep(1) + return proc.returncode or 0 + finally: + if proc.poll() is None: + proc.terminate() + try: + proc.wait(timeout=10) + except subprocess.TimeoutExpired: + proc.kill() + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/src/arena.c b/src/arena.c new file mode 100644 index 0000000..c078080 --- /dev/null +++ b/src/arena.c @@ -0,0 +1,41 @@ +/* + * arena.c - The block arena: a single page-aligned slab of num_slots * 16 KiB. + * + * Block payloads are received directly into slabs and exposed to the harness + * zero-copy. We touch every page once so the pages are faulted in up front and + * never allocate per block on the hot path. + */ +#include "arena.h" + +#include +#include +#include +#include + +uint8_t *arena_alloc(uint32_t num_slots, uint64_t *out_bytes) { + uint64_t bytes = (uint64_t)num_slots * PEER_BLOCK_SIZE; + long pagesz = sysconf(_SC_PAGESIZE); + if (pagesz <= 0) pagesz = 4096; + + void *p = NULL; + if (posix_memalign(&p, (size_t)pagesz, (size_t)bytes) != 0 || !p) + return NULL; + + /* P3b: hint transparent huge pages to cut TLB pressure on the multi-GB/s + * receive path. Best-effort; ignored where THP is unavailable. */ +#ifdef MADV_HUGEPAGE + madvise(p, (size_t)bytes, MADV_HUGEPAGE); +#endif + + /* Pre-fault: write one byte per page so the hot path never page-faults. */ + uint8_t *base = (uint8_t *)p; + for (uint64_t off = 0; off < bytes; off += (uint64_t)pagesz) + base[off] = 0; + + *out_bytes = bytes; + return base; +} + +void arena_free(uint8_t *base) { + free(base); +} diff --git a/src/arena.h b/src/arena.h new file mode 100644 index 0000000..ab390f7 --- /dev/null +++ b/src/arena.h @@ -0,0 +1,14 @@ +/* arena.h - aligned, pre-faulted block arena. */ +#ifndef TORRENT_PEER_ARENA_H +#define TORRENT_PEER_ARENA_H + +#include + +#include "../include/engine.h" + +/* Allocate num_slots * PEER_BLOCK_SIZE bytes, page-aligned and pre-faulted. + * Returns the base pointer (NULL on failure) and writes the size to *out_bytes. */ +uint8_t *arena_alloc(uint32_t num_slots, uint64_t *out_bytes); +void arena_free(uint8_t *base); + +#endif /* TORRENT_PEER_ARENA_H */ diff --git a/src/connection.c b/src/connection.c new file mode 100644 index 0000000..c29714a --- /dev/null +++ b/src/connection.c @@ -0,0 +1,338 @@ +/* + * connection.c - Per-peer connection lifecycle and socket I/O. + * + * A connection is created, driven, and destroyed entirely on its owning loop + * thread, so nothing here needs locking. Reads are pulled into the loop's + * shared recv staging buffer and pushed through proto_feed(); writes are + * batched through a fixed per-connection outgoing buffer. The transport vtable + * abstracts the byte stream (TCP today). + */ +#include "engine_internal.h" + +#include +#include +#include + +#include + +/* ---- outgoing buffer ------------------------------------------------- */ + +void outbuf_compact(outbuf *o) { + size_t pending = o->tail - o->head; + if (o->head > 0) { + memmove(o->buf, o->buf + o->head, pending); + o->head = 0; + o->tail = pending; + } +} + +void outbuf_append(outbuf *o, const void *data, size_t n) { + if (OUTBUF_CAP - o->tail < n) outbuf_compact(o); + size_t space = OUTBUF_CAP - o->tail; + if (n > space) n = space; /* gated by callers; never expected to trip */ + memcpy(o->buf + o->tail, data, n); + o->tail += n; +} + +/* ---- epoll interest -------------------------------------------------- */ + +static void want_events(conn *c, uint32_t ev) { + if (ev == c->cur_events) return; + struct epoll_event e; + memset(&e, 0, sizeof e); + e.events = ev; + e.data.ptr = c; + epoll_ctl(c->lp->epfd, EPOLL_CTL_MOD, c->tr.fd, &e); + c->cur_events = ev; +} + +/* ---- transport selection + fallback ladder --------------------------- */ + +/* Build the ordered list of (transport, encryption) combinations to try for a + * peer. With fallback off it is just the configured combo. With fallback on we + * try the preferred transport first, MSE before plaintext (unless RC4 is + * required, in which case only RC4), then the other transport -- so a peer that + * rejects one combination is reached over another. */ +static void build_ladder(const engine_config *cfg, conn *c) { + if (!cfg->fallback) { + c->v_utp[0] = (uint8_t)(cfg->utp ? 1 : 0); + c->v_enc[0] = (uint8_t)cfg->encryption; + c->nvariants = 1; + return; + } + uint8_t encs[2]; int ne = 0; + if (cfg->encryption == 2) encs[ne++] = 2; /* require RC4 */ + else if (cfg->encryption == 1) { encs[ne++] = 1; encs[ne++] = 0; } + else encs[ne++] = 0; /* plaintext only */ + + uint8_t tps[2]; int nt = 0; + tps[nt++] = (uint8_t)(cfg->utp ? 1 : 0); + tps[nt++] = (uint8_t)(cfg->utp ? 0 : 1); + + int n = 0; + for (int t = 0; t < nt; t++) + for (int e = 0; e < ne; e++) { c->v_utp[n] = tps[t]; c->v_enc[n] = encs[e]; n++; } + c->nvariants = (uint8_t)n; +} + +/* A terminal peer failure must release every scheduler resource that peer + * claimed. Otherwise a dead connection can leave the torrent with whole pieces + * marked requested and loop-wide in-flight credit consumed, which is most + * visible near the end as a few blocks stuck at 0 B/s until restart. */ +static void conn_release_pending(conn *c) { + if (!c || !c->tor) return; + + if (c->rd.state == RS_PIECE_BODY) { + slot_ring_push(&c->lp->free_ring, c->rd.cur_slot); + if (c->rd.cur_piece < c->tor->num_pieces) + c->tor->requested[c->rd.cur_piece] = 0; + c->rd.state = RS_LEN; + } + + uint32_t removed = 0; + if (c->inflight.slots) { + uint32_t cap = c->inflight.mask + 1; + for (uint32_t i = 0; i < cap; i++) { + if (c->inflight.slots[i].key == REQ_EMPTY) continue; + uint32_t piece = c->inflight.slots[i].piece; + if (piece < c->tor->num_pieces) c->tor->requested[piece] = 0; + c->inflight.slots[i].key = REQ_EMPTY; + removed++; + } + c->inflight.count = 0; + } + + for (uint32_t i = 0; i < c->requeue_count; i++) { + uint32_t piece = c->requeue[i].piece; + if (piece < c->tor->num_pieces) c->tor->requested[piece] = 0; + } + c->requeue_count = 0; + + if (c->have_cur_piece && c->cur_piece < c->tor->num_pieces) + c->tor->requested[c->cur_piece] = 0; + c->have_cur_piece = 0; + c->cur_block_off = 0; + + if (removed) { + uint32_t out = atomic_load_explicit(&c->outstanding, + memory_order_relaxed); + if (out > removed) + atomic_fetch_sub_explicit(&c->outstanding, removed, + memory_order_relaxed); + else + atomic_store_explicit(&c->outstanding, 0, memory_order_relaxed); + + c->lp->outstanding = c->lp->outstanding > removed + ? c->lp->outstanding - removed : 0; + } +} + +/* Open the transport for the current ladder variant, reset per-attempt parse + * state, and register the new fd with epoll. Returns 0 on success, -1 if the + * variant could not be opened (caller advances to the next one). */ +static int conn_start_variant(conn *c) { + engine *e = c->lp->eng; + uint8_t use_utp = c->v_utp[c->variant]; + uint8_t enc = c->v_enc[c->variant]; + uint32_t recvbuf = e->cfg.recv_buffer_bytes; + int connecting = 0; + int rc; + + if (use_utp) { + transport inner; + int ic = 0; + if (transport_utp_connect(&inner, c->ip, c->port, recvbuf, &ic) != 0) + return -1; + if (enc) + rc = transport_mse_wrap(&c->tr, &inner, c->tor->info_hash, + enc == 2, &connecting); + else { c->tr = inner; connecting = ic; rc = 0; } + } else if (enc) { + rc = transport_mse_connect(&c->tr, c->ip, c->port, recvbuf, + c->tor->info_hash, enc == 2, &connecting); + } else { + rc = transport_tcp_connect(&c->tr, c->ip, c->port, recvbuf, &connecting); + } + if (rc != 0) return -1; + + /* Fresh attempt: clear parse/send state so a half-spoken prior transport + * leaves nothing behind. (Fallback only happens pre-handshake, so there is + * no in-flight request state to unwind.) */ + uint8_t *other = c->rd.other; size_t ocap = c->rd.other_cap; + memset(&c->rd, 0, sizeof c->rd); + c->rd.other = other; c->rd.other_cap = ocap; c->rd.state = RS_HANDSHAKE; + c->out.head = c->out.tail = 0; + c->unchoked = 0; c->have_cur_piece = 0; c->cur_block_off = 0; + c->bt_established = 0; c->fast_enabled = 0; c->ext_enabled = 0; + c->requeue_count = 0; + if (c->have_bits) memset(c->have_bits, 0, c->tor->bf_bytes); + if (c->allowed_fast_bits) memset(c->allowed_fast_bits, 0, c->tor->bf_bytes); + atomic_store_explicit(&c->outstanding, 0, memory_order_relaxed); + atomic_store_explicit(&c->aerror, PEER_OK, memory_order_relaxed); + c->connect_started_ns = peer_now_ns(); + c->connecting = connecting; + + if (connecting) { + atomic_store_explicit(&c->astate, PEER_STATE_CONNECTING, memory_order_relaxed); + c->cur_events = EPOLLOUT; + } else { + proto_queue_handshake(c); + atomic_store_explicit(&c->astate, PEER_STATE_HANDSHAKE, memory_order_relaxed); + c->cur_events = EPOLLIN | EPOLLOUT; + } + + struct epoll_event ev; + memset(&ev, 0, sizeof ev); + ev.events = c->cur_events; + ev.data.ptr = c; + if (epoll_ctl(c->lp->epfd, EPOLL_CTL_ADD, c->tr.fd, &ev) != 0) { + if (c->tr.close) c->tr.close(&c->tr); + return -1; + } + return 0; +} + +void conn_fail_or_fallback(conn *c, peer_error e) { + /* Once we've spoken BitTorrent the transport works; failures are terminal. + * Before that, a failure may just mean the peer wanted a different + * transport/encryption, so walk the ladder. */ + if (!c->bt_established) { + if (c->tr.close) c->tr.close(&c->tr); + while (c->variant + 1 < c->nvariants) { + c->variant++; + if (conn_start_variant(c) == 0) return; /* retrying next combo */ + } + } + conn_release_pending(c); + conn_set_error(c, e); + c->dead = 1; +} + +/* ---- lifecycle ------------------------------------------------------- */ + +conn *conn_create(loop *lp, torrent *tor, const char *ip, uint16_t port) { + conn *c = calloc(1, sizeof *c); + if (!c) return NULL; + c->lp = lp; + c->tor = tor; + strncpy(c->ip, ip, sizeof c->ip - 1); + c->port = port; + c->request_timeout_ns = (uint64_t)lp->eng->cfg.request_timeout_ms * 1000000ull; + c->connect_started_ns = peer_now_ns(); + + atomic_init(&c->astate, PEER_STATE_IDLE); + atomic_init(&c->aerror, PEER_OK); + atomic_init(&c->outstanding, 0); + atomic_init(&c->bytes_received, 0); + atomic_init(&c->blocks_received, 0); + + c->have_bits = calloc(tor->bf_bytes, 1); + c->allowed_fast_bits = calloc(tor->bf_bytes, 1); + c->rd.other = malloc(MAX_OTHER_MSG); + if (!c->have_bits || !c->allowed_fast_bits || !c->rd.other) goto fail; + c->rd.other_cap = MAX_OTHER_MSG; + c->rd.state = RS_HANDSHAKE; + + uint32_t cap = lp->eng->cfg.max_pipeline; + if (reqtab_init(&c->inflight, cap * 2) != 0) goto fail; + c->requeue_cap = cap; + c->requeue = malloc((size_t)c->requeue_cap * sizeof(*c->requeue)); + if (!c->requeue) goto fail; + + /* Build the fallback ladder and open the first combo that succeeds. */ + build_ladder(&lp->eng->cfg, c); + c->variant = 0; + while (conn_start_variant(c) != 0) { + c->variant++; + if (c->variant >= c->nvariants) goto fail; + } + + /* Link into the loop's connection list. A torrent's connections are found + * by filtering this list on conn->tor (the torrent is pinned to this loop), + * so conn->next belongs solely to the loop list. */ + c->next = lp->conns; + lp->conns = c; + tor->conn_count++; + return c; + +fail: + free(c->have_bits); + free(c->allowed_fast_bits); + free(c->rd.other); + free(c->requeue); + if (c->inflight.slots) reqtab_free(&c->inflight); + free(c); + return NULL; +} + +void conn_destroy(conn *c) { + if (!c) return; + if (c->tr.close) c->tr.close(&c->tr); + free(c->have_bits); + free(c->allowed_fast_bits); + free(c->rd.other); + free(c->requeue); + reqtab_free(&c->inflight); + free(c); +} + +/* ---- I/O ------------------------------------------------------------- */ + +int conn_flush(conn *c) { + if (c->dead) return 0; + outbuf *o = &c->out; + while (o->head < o->tail) { + ssize_t n = c->tr.send(&c->tr, o->buf + o->head, o->tail - o->head); + if (n > 0) { o->head += (size_t)n; continue; } + if (n < 0 && errno == EINTR) continue; + if (n < 0 && (errno == EAGAIN || errno == EWOULDBLOCK)) break; + conn_fail_or_fallback(c, PEER_ERR_IO); + return -1; + } + if (o->head == o->tail) { o->head = o->tail = 0; } + return 0; +} + +/* Advance the transport handshake (TCP connect, then any MSE exchange). Called + * on read or write readiness while the connection is still connecting. */ +void conn_drive_handshake(conn *c) { + uint32_t want = EPOLLOUT; + int r = c->tr.handshake(&c->tr, &want); + if (r < 0) { conn_fail_or_fallback(c, PEER_ERR_CONNECT); return; } + if (r == 0) { want_events(c, want ? want : EPOLLOUT); return; } + /* Stream is ready: start the BitTorrent handshake over it. */ + c->connecting = 0; + proto_queue_handshake(c); + atomic_store_explicit(&c->astate, PEER_STATE_HANDSHAKE, memory_order_relaxed); + want_events(c, EPOLLIN | EPOLLOUT); + conn_flush(c); +} + +void conn_on_readable(conn *c) { + loop *lp = c->lp; + for (;;) { + ssize_t n = c->tr.recv(&c->tr, lp->recvbuf, lp->recvbuf_cap); + if (n > 0) { + if (proto_feed(c, lp->recvbuf, (size_t)n) < 0) { + int err = atomic_load_explicit(&c->aerror, memory_order_relaxed); + conn_fail_or_fallback(c, (peer_error)err); + return; + } + if ((size_t)n < lp->recvbuf_cap) break; /* drained for now */ + continue; + } + if (n == 0) { conn_fail_or_fallback(c, PEER_ERR_CLOSED); return; } + if (errno == EINTR) continue; + if (errno == EAGAIN || errno == EWOULDBLOCK) break; + conn_fail_or_fallback(c, PEER_ERR_IO); + return; + } +} + +/* Recompute and apply epoll interest after a tick. While the transport is still + * handshaking, its interest is owned by conn_drive_handshake(); leave it be. */ +void conn_update_interest(conn *c) { + if (c->dead || c->connecting) return; + uint32_t want = EPOLLIN | (outbuf_pending(&c->out) ? EPOLLOUT : 0u); + want_events(c, want); +} diff --git a/src/crypto.c b/src/crypto.c new file mode 100644 index 0000000..1639d75 --- /dev/null +++ b/src/crypto.c @@ -0,0 +1,300 @@ +/* + * crypto.c - SHA-1, RC4, and 768-bit Diffie-Hellman for MSE. See crypto.h. + * + * The bignum is a fixed-width (24 x 32-bit limb) schoolbook implementation: + * a modular exponentiation runs once per connection during the handshake, so + * clarity beats cleverness here. Limbs are little-endian; byte I/O is + * big-endian to match the wire format. + */ +#include "crypto.h" + +#include +#include + +#if defined(__linux__) +#include +#endif +#include +#include + +/* ===================== SHA-1 ========================================== */ + +static inline uint32_t rol32(uint32_t v, int s) { + return (v << s) | (v >> (32 - s)); +} + +void sha1_init(sha1_ctx *c) { + c->h[0] = 0x67452301; c->h[1] = 0xEFCDAB89; c->h[2] = 0x98BADCFE; + c->h[3] = 0x10325476; c->h[4] = 0xC3D2E1F0; + c->len = 0; + c->buf_len = 0; +} + +static void sha1_block(sha1_ctx *c, const uint8_t *p) { + uint32_t w[80]; + for (int i = 0; i < 16; i++) + w[i] = ((uint32_t)p[i*4] << 24) | ((uint32_t)p[i*4+1] << 16) | + ((uint32_t)p[i*4+2] << 8) | (uint32_t)p[i*4+3]; + for (int i = 16; i < 80; i++) + w[i] = rol32(w[i-3] ^ w[i-8] ^ w[i-14] ^ w[i-16], 1); + + uint32_t a = c->h[0], b = c->h[1], d = c->h[2], e = c->h[3], f = c->h[4]; + for (int i = 0; i < 80; i++) { + uint32_t t, k; + if (i < 20) { t = (b & d) | (~b & e); k = 0x5A827999; } + else if (i < 40) { t = b ^ d ^ e; k = 0x6ED9EBA1; } + else if (i < 60) { t = (b & d) | (b & e) | (d & e); k = 0x8F1BBCDC; } + else { t = b ^ d ^ e; k = 0xCA62C1D6; } + uint32_t tmp = rol32(a, 5) + t + f + k + w[i]; + f = e; e = d; d = rol32(b, 30); b = a; a = tmp; + } + c->h[0] += a; c->h[1] += b; c->h[2] += d; c->h[3] += e; c->h[4] += f; +} + +void sha1_update(sha1_ctx *c, const void *data, size_t len) { + const uint8_t *p = data; + c->len += len; + while (len > 0) { + size_t take = 64 - c->buf_len; + if (take > len) take = len; + memcpy(c->buf + c->buf_len, p, take); + c->buf_len += take; + p += take; + len -= take; + if (c->buf_len == 64) { sha1_block(c, c->buf); c->buf_len = 0; } + } +} + +void sha1_final(sha1_ctx *c, uint8_t out[20]) { + uint64_t bits = c->len * 8; + uint8_t pad = 0x80; + sha1_update(c, &pad, 1); + uint8_t zero = 0; + while (c->buf_len != 56) sha1_update(c, &zero, 1); + uint8_t lenb[8]; + for (int i = 0; i < 8; i++) lenb[i] = (uint8_t)(bits >> (56 - 8*i)); + sha1_update(c, lenb, 8); + for (int i = 0; i < 5; i++) { + out[i*4] = (uint8_t)(c->h[i] >> 24); + out[i*4+1] = (uint8_t)(c->h[i] >> 16); + out[i*4+2] = (uint8_t)(c->h[i] >> 8); + out[i*4+3] = (uint8_t)c->h[i]; + } +} + +void sha1_concat(uint8_t out[20], const void *a, size_t na, + const void *b, size_t nb, const void *c, size_t nc) { + sha1_ctx ctx; + sha1_init(&ctx); + if (na) sha1_update(&ctx, a, na); + if (nb) sha1_update(&ctx, b, nb); + if (nc) sha1_update(&ctx, c, nc); + sha1_final(&ctx, out); +} + +/* ===================== RC4 ============================================ */ + +void rc4_init(rc4_ctx *c, const void *key, size_t keylen) { + const uint8_t *k = key; + for (int i = 0; i < 256; i++) c->s[i] = (uint8_t)i; + uint8_t j = 0; + for (int i = 0; i < 256; i++) { + j = (uint8_t)(j + c->s[i] + k[i % keylen]); + uint8_t t = c->s[i]; c->s[i] = c->s[j]; c->s[j] = t; + } + c->i = c->j = 0; +} + +void rc4_process(rc4_ctx *c, const void *in, void *out, size_t len) { + const uint8_t *ip = in; + uint8_t *op = out; + uint8_t i = c->i, j = c->j; + for (size_t n = 0; n < len; n++) { + i = (uint8_t)(i + 1); + j = (uint8_t)(j + c->s[i]); + uint8_t t = c->s[i]; c->s[i] = c->s[j]; c->s[j] = t; + op[n] = ip[n] ^ c->s[(uint8_t)(c->s[i] + c->s[j])]; + } + c->i = i; c->j = j; +} + +void rc4_skip(rc4_ctx *c, size_t n) { + uint8_t i = c->i, j = c->j; + for (size_t k = 0; k < n; k++) { + i = (uint8_t)(i + 1); + j = (uint8_t)(j + c->s[i]); + uint8_t t = c->s[i]; c->s[i] = c->s[j]; c->s[j] = t; + } + c->i = i; c->j = j; +} + +/* ===================== bignum (24 x uint32) =========================== */ + +#define NL 24 /* limbs for a 768-bit number */ + +/* The MSE prime P (768-bit MODP group), big-endian. */ +static const uint8_t MSE_P_BE[MSE_DH_LEN] = { + 0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xC9,0x0F,0xDA,0xA2,0x21,0x68,0xC2,0x34, + 0xC4,0xC6,0x62,0x8B,0x80,0xDC,0x1C,0xD1,0x29,0x02,0x4E,0x08,0x8A,0x67,0xCC,0x74, + 0x02,0x0B,0xBE,0xA6,0x3B,0x13,0x9B,0x22,0x51,0x4A,0x08,0x79,0x8E,0x34,0x04,0xDD, + 0xEF,0x95,0x19,0xB3,0xCD,0x3A,0x43,0x1B,0x30,0x2B,0x0A,0x6D,0xF2,0x5F,0x14,0x37, + 0x4F,0xE1,0x35,0x6D,0x6D,0x51,0xC2,0x45,0xE4,0x85,0xB5,0x76,0x62,0x5E,0x7E,0xC6, + 0xF4,0x4C,0x42,0xE9,0xA6,0x3A,0x36,0x21,0x00,0x00,0x00,0x00,0x00,0x09,0x05,0x63 +}; + +static void be_to_limbs(const uint8_t *be, size_t nbytes, uint32_t out[NL]) { + memset(out, 0, NL * sizeof(uint32_t)); + /* be is big-endian; limb 0 is least significant. */ + for (size_t i = 0; i < nbytes; i++) { + size_t byte_from_end = nbytes - 1 - i; /* 0 = LSB */ + size_t limb = byte_from_end / 4; + size_t sh = (byte_from_end % 4) * 8; + if (limb < NL) out[limb] |= (uint32_t)be[i] << sh; + } +} + +static void limbs_to_be(const uint32_t in[NL], uint8_t *be, size_t nbytes) { + for (size_t i = 0; i < nbytes; i++) { + size_t byte_from_end = nbytes - 1 - i; + size_t limb = byte_from_end / 4; + size_t sh = (byte_from_end % 4) * 8; + be[i] = (limb < NL) ? (uint8_t)(in[limb] >> sh) : 0; + } +} + +/* Compare a[n] vs b[n]: -1, 0, 1. */ +static int bn_cmp(const uint32_t *a, const uint32_t *b, int n) { + for (int i = n - 1; i >= 0; i--) + if (a[i] != b[i]) return a[i] < b[i] ? -1 : 1; + return 0; +} + +/* prod[2*NL] = a[NL] * b[NL] (schoolbook). */ +static void bn_mul(const uint32_t *a, const uint32_t *b, uint32_t *prod) { + memset(prod, 0, 2 * NL * sizeof(uint32_t)); + for (int i = 0; i < NL; i++) { + uint64_t carry = 0; + for (int j = 0; j < NL; j++) { + uint64_t cur = (uint64_t)a[i] * b[j] + prod[i + j] + carry; + prod[i + j] = (uint32_t)cur; + carry = cur >> 32; + } + prod[i + NL] += (uint32_t)carry; + } +} + +/* rem[NL] = x[2*NL] mod m[NL], bitwise long division. */ +static void bn_mod(const uint32_t *x, const uint32_t *m, uint32_t *rem_out) { + uint32_t rem[NL + 1]; + memset(rem, 0, sizeof rem); + for (int bit = 2 * NL * 32 - 1; bit >= 0; bit--) { + /* rem <<= 1 */ + uint32_t carry = 0; + for (int i = 0; i < NL + 1; i++) { + uint32_t nc = rem[i] >> 31; + rem[i] = (rem[i] << 1) | carry; + carry = nc; + } + /* bring in bit `bit` of x */ + rem[0] |= (x[bit >> 5] >> (bit & 31)) & 1u; + /* if rem >= m (m has implicit 0 top limb), subtract */ + int ge; + if (rem[NL] != 0) ge = 1; + else ge = bn_cmp(rem, m, NL) >= 0; + if (ge) { + uint64_t borrow = 0; + for (int i = 0; i < NL; i++) { + uint64_t cur = (uint64_t)rem[i] - m[i] - borrow; + rem[i] = (uint32_t)cur; + borrow = (cur >> 32) & 1; + } + rem[NL] -= (uint32_t)borrow; + } + } + memcpy(rem_out, rem, NL * sizeof(uint32_t)); +} + +static void bn_modmul(const uint32_t *a, const uint32_t *b, const uint32_t *m, + uint32_t *out) { + uint32_t prod[2 * NL]; + bn_mul(a, b, prod); + bn_mod(prod, m, out); +} + +/* out = base^exp mod m, exp given big-endian. */ +static void bn_modexp(const uint32_t *base, const uint8_t *exp, size_t explen, + const uint32_t *m, uint32_t *out) { + uint32_t res[NL]; memset(res, 0, sizeof res); res[0] = 1; + uint32_t b[NL]; + { + /* b = base mod m (base < m already for our use, but normalise anyway) */ + uint32_t wide[2 * NL]; memset(wide, 0, sizeof wide); + memcpy(wide, base, NL * sizeof(uint32_t)); + bn_mod(wide, m, b); + } + for (size_t i = 0; i < explen; i++) { + for (int bit = 7; bit >= 0; bit--) { + uint32_t sq[NL]; + bn_modmul(res, res, m, sq); + memcpy(res, sq, sizeof res); + if ((exp[i] >> bit) & 1) { + uint32_t mul[NL]; + bn_modmul(res, b, m, mul); + memcpy(res, mul, sizeof res); + } + } + } + memcpy(out, res, NL * sizeof(uint32_t)); +} + +/* ===================== DH ============================================= */ + +int crypto_random(void *buf, size_t len) { +#if defined(__linux__) + { + uint8_t *p = buf; + size_t got = 0; + while (got < len) { + ssize_t r = getrandom(p + got, len - got, 0); + if (r > 0) { got += (size_t)r; continue; } + if (r < 0 && errno == EINTR) continue; + break; + } + if (got == len) return 0; + } +#endif + int fd = open("/dev/urandom", O_RDONLY); + if (fd < 0) return -1; + uint8_t *p = buf; + size_t got = 0; + while (got < len) { + ssize_t r = read(fd, p + got, len - got); + if (r > 0) { got += (size_t)r; continue; } + if (r < 0 && errno == EINTR) continue; + close(fd); + return -1; + } + close(fd); + return 0; +} + +int dh_generate(uint8_t priv_out[20], uint8_t pub_out[MSE_DH_LEN]) { + if (crypto_random(priv_out, 20) != 0) return -1; + priv_out[0] |= 0x01; /* ensure non-zero exponent */ + + uint32_t P[NL], g[NL], Ya[NL]; + be_to_limbs(MSE_P_BE, MSE_DH_LEN, P); + memset(g, 0, sizeof g); g[0] = 2; + bn_modexp(g, priv_out, 20, P, Ya); + limbs_to_be(Ya, pub_out, MSE_DH_LEN); + return 0; +} + +void dh_shared(const uint8_t priv[20], const uint8_t peer_pub[MSE_DH_LEN], + uint8_t secret_out[MSE_DH_LEN]) { + uint32_t P[NL], Yb[NL], S[NL]; + be_to_limbs(MSE_P_BE, MSE_DH_LEN, P); + be_to_limbs(peer_pub, MSE_DH_LEN, Yb); + bn_modexp(Yb, priv, 20, P, S); + limbs_to_be(S, secret_out, MSE_DH_LEN); +} diff --git a/src/crypto.h b/src/crypto.h new file mode 100644 index 0000000..d27630f --- /dev/null +++ b/src/crypto.h @@ -0,0 +1,62 @@ +/* + * crypto.h - Minimal crypto primitives for MSE (Message Stream Encryption). + * + * Just enough to implement the BitTorrent PE/MSE handshake from scratch (no + * external deps): SHA-1 (for the HASH() construction), RC4 with the mandated + * 1024-byte keystream discard, and 768-bit Diffie-Hellman over the well-known + * MSE prime with g=2. None of this is meant for general-purpose security; RC4 + * and a 768-bit DH group are what the BitTorrent MSE spec mandates. + */ +#ifndef TORRENT_CRYPTO_H +#define TORRENT_CRYPTO_H + +#include +#include + +/* ---- SHA-1 ----------------------------------------------------------- */ +typedef struct { + uint32_t h[5]; + uint64_t len; /* total bytes hashed */ + uint8_t buf[64]; + size_t buf_len; +} sha1_ctx; + +void sha1_init(sha1_ctx *c); +void sha1_update(sha1_ctx *c, const void *data, size_t len); +void sha1_final(sha1_ctx *c, uint8_t out[20]); + +/* One-shot helper: SHA-1 over up to three concatenated chunks (any may be NULL + * with len 0). Covers the SHA1('tag' || S || SKEY) patterns MSE uses. */ +void sha1_concat(uint8_t out[20], + const void *a, size_t na, + const void *b, size_t nb, + const void *c, size_t nc); + +/* ---- RC4 ------------------------------------------------------------- */ +typedef struct { + uint8_t s[256]; + uint8_t i, j; +} rc4_ctx; + +void rc4_init(rc4_ctx *c, const void *key, size_t keylen); +/* XOR `len` keystream bytes into out (out may equal in for in-place). */ +void rc4_process(rc4_ctx *c, const void *in, void *out, size_t len); +/* Advance the keystream by `n` bytes, discarding output (RC4-drop-N). */ +void rc4_skip(rc4_ctx *c, size_t n); + +/* ---- Diffie-Hellman (MSE: 768-bit MODP, g=2) ------------------------- */ +#define MSE_DH_LEN 96 /* 768 bits */ + +/* Generate a private exponent and the public key Ya = 2^Xa mod P. + * priv_out receives the 20-byte (160-bit) private exponent; pub_out the + * 96-byte big-endian public key. Returns 0 on success, -1 on RNG failure. */ +int dh_generate(uint8_t priv_out[20], uint8_t pub_out[MSE_DH_LEN]); + +/* Compute the shared secret S = Yb^Xa mod P (96-byte big-endian). */ +void dh_shared(const uint8_t priv[20], const uint8_t peer_pub[MSE_DH_LEN], + uint8_t secret_out[MSE_DH_LEN]); + +/* CSPRNG bytes (getrandom/urandom). Returns 0 on success, -1 on failure. */ +int crypto_random(void *buf, size_t len); + +#endif /* TORRENT_CRYPTO_H */ diff --git a/src/engine.c b/src/engine.c new file mode 100644 index 0000000..21f6216 --- /dev/null +++ b/src/engine.c @@ -0,0 +1,571 @@ +/* + * engine.c - Public engine ABI, object lifecycle, the torrent registry, and the + * control-plane command fan-out. + * + * The engine owns a fixed pool of event loops (loop.c). Torrents are pinned to + * the least-loaded loop at registration time ("affinity"); all of a torrent's + * connections then live on that one loop, which keeps the per-torrent piece + * state lock-free. Control calls (add peer, set priorities) are turned into + * commands posted to the owning loop; the data plane (poll/release) talks to the + * per-loop arenas and rings directly. + */ +#include "engine_internal.h" +#include "arena.h" + +#include +#include +#include +#include +#include + +#include +#include + +#define DEFAULT_SLOTS 4096u /* 64 MiB arena per loop */ +#define DEFAULT_PIPELINE 2048u /* per-connection outstanding cap */ +#define DEFAULT_TIMEOUT_MS 30000u /* re-request a block after this long */ +#define DEFAULT_CONNECT_MS 10000u /* drop a peer stuck connecting/handshaking */ +#define DEFAULT_MAX_LOOPS 8u +#define RECV_STAGING_CAP (256u * 1024u) + +/* ---- shared helpers -------------------------------------------------- */ + +/* Update the limit (control plane). burst is one second of credit, floored at + * 1 MiB so a small limit can still admit whole blocks. */ +void rate_set(rate_limiter *rl, uint64_t bytes_per_sec) { + pthread_mutex_lock(&rl->lock); + atomic_store_explicit(&rl->rate_bps, bytes_per_sec, memory_order_relaxed); + rl->burst = bytes_per_sec > (1u << 20) ? bytes_per_sec : (1u << 20); + rl->last_ns = peer_now_ns(); + if (rl->tokens > (double)rl->burst) rl->tokens = (double)rl->burst; + pthread_mutex_unlock(&rl->lock); +} + +/* Token-bucket throttle. Refills lazily: tokens accrue at rate_bps since the + * last call, capped at burst. Returns false without consuming when starved. */ +bool rate_try_consume(rate_limiter *rl, uint32_t bytes) { + if (atomic_load_explicit(&rl->rate_bps, memory_order_relaxed) == 0) + return true; /* unlimited; no lock on the hot path */ + pthread_mutex_lock(&rl->lock); + bool ok = true; + if (atomic_load_explicit(&rl->rate_bps, memory_order_relaxed) == 0) { + pthread_mutex_unlock(&rl->lock); /* became unlimited */ + return true; + } + uint64_t now = peer_now_ns(); + if (rl->last_ns == 0) rl->last_ns = now; + double accrued = (double)(now - rl->last_ns) * 1e-9 * + (double)atomic_load_explicit(&rl->rate_bps, + memory_order_relaxed); + rl->last_ns = now; + rl->tokens += accrued; + if (rl->tokens > (double)rl->burst) rl->tokens = (double)rl->burst; + if (rl->tokens >= (double)bytes) + rl->tokens -= (double)bytes; + else + ok = false; + pthread_mutex_unlock(&rl->lock); + return ok; +} + +uint64_t peer_now_ns(void) { + struct timespec ts; + clock_gettime(CLOCK_MONOTONIC, &ts); + return (uint64_t)ts.tv_sec * 1000000000ull + (uint64_t)ts.tv_nsec; +} + +uint64_t torrent_piece_len(const torrent *t, uint32_t piece) { + if (piece + 1 == t->num_pieces) { + uint64_t before = (uint64_t)piece * t->piece_length; + return t->total_size - before; + } + return t->piece_length; +} + +torrent *engine_find_torrent(engine *e, uint32_t id) { + for (torrent *t = e->torrents; t; t = t->enext) + if (t->id == id) return t; + return NULL; +} + +void loop_post(loop *lp, cmd *c) { + pthread_mutex_lock(&lp->cmd_lock); + c->next = NULL; + if (lp->cmd_tail) lp->cmd_tail->next = c; else lp->cmd_head = c; + lp->cmd_tail = c; + pthread_mutex_unlock(&lp->cmd_lock); + uint64_t one = 1; + ssize_t w = write(lp->cmd_efd, &one, sizeof one); + (void)w; +} + +/* ---- lifecycle ------------------------------------------------------- */ + +static int loop_init(engine *e, loop *lp, uint32_t index) { + lp->eng = e; + lp->index = (int)index; + lp->epfd = -1; + lp->cmd_efd = -1; + atomic_init(&lp->stop, 0); + atomic_init(&lp->want_release_wake, 0); + pthread_mutex_init(&lp->cmd_lock, NULL); + + lp->num_slots = e->cfg.slots_per_loop; + lp->arena = arena_alloc(lp->num_slots, &lp->arena_bytes); + if (!lp->arena) return -1; + if (slot_ring_init(&lp->free_ring, lp->num_slots) != 0) return -1; + if (desc_ring_init(&lp->ready_ring, lp->num_slots) != 0) return -1; + for (uint32_t s = 0; s < lp->num_slots; s++) slot_ring_push(&lp->free_ring, s); + + lp->recvbuf_cap = RECV_STAGING_CAP; + lp->recvbuf = malloc(lp->recvbuf_cap); + if (!lp->recvbuf) return -1; + + lp->epfd = epoll_create1(0); + lp->cmd_efd = eventfd(0, EFD_NONBLOCK | EFD_CLOEXEC); + if (lp->epfd < 0 || lp->cmd_efd < 0) return -1; + + struct epoll_event ev; + memset(&ev, 0, sizeof ev); + ev.events = EPOLLIN; + ev.data.ptr = NULL; /* NULL = the command/credit eventfd */ + epoll_ctl(lp->epfd, EPOLL_CTL_ADD, lp->cmd_efd, &ev); + return 0; +} + +engine *engine_create(const engine_config *cfg) { + engine *e = calloc(1, sizeof *e); + if (!e) return NULL; + pthread_mutex_init(&e->lock, NULL); + pthread_mutex_init(&e->dl_limit.lock, NULL); /* rate_bps 0 => unlimited */ + e->ready_efd = -1; + if (cfg) e->cfg = *cfg; + + if (e->cfg.loop_count == 0) { + long nc = sysconf(_SC_NPROCESSORS_ONLN); + if (nc < 1) nc = 1; + if (nc > (long)DEFAULT_MAX_LOOPS) nc = DEFAULT_MAX_LOOPS; + e->cfg.loop_count = (uint32_t)nc; + } + if (e->cfg.slots_per_loop == 0) e->cfg.slots_per_loop = DEFAULT_SLOTS; + if (e->cfg.max_pipeline == 0) e->cfg.max_pipeline = DEFAULT_PIPELINE; + if (e->cfg.max_pipeline > e->cfg.slots_per_loop) + e->cfg.max_pipeline = e->cfg.slots_per_loop; + if (e->cfg.request_timeout_ms == 0) e->cfg.request_timeout_ms = DEFAULT_TIMEOUT_MS; + if (e->cfg.connect_timeout_ms == 0) e->cfg.connect_timeout_ms = DEFAULT_CONNECT_MS; + /* cfg.fallback defaults to 0 (single attempt) -- left as-is. */ + e->nloops = e->cfg.loop_count; + + e->ready_efd = eventfd(0, EFD_NONBLOCK | EFD_CLOEXEC); + if (e->ready_efd < 0) goto fail; + + /* loop embeds rings whose head/tail are _Alignas(64), so loop is + * over-aligned; calloc only guarantees max_align_t. Allocate the array with + * the real alignment (sizeof(loop) is a multiple of it). */ + size_t align = _Alignof(loop); + if (align < sizeof(void *)) align = sizeof(void *); + e->loops = aligned_alloc(align, (size_t)e->nloops * sizeof(loop)); + if (!e->loops) goto fail; + memset(e->loops, 0, (size_t)e->nloops * sizeof(loop)); + + for (uint32_t i = 0; i < e->nloops; i++) + if (loop_init(e, &e->loops[i], i) != 0) goto fail; + + for (uint32_t i = 0; i < e->nloops; i++) { + if (pthread_create(&e->loops[i].thread, NULL, loop_run, &e->loops[i]) != 0) + goto fail; + e->loops[i].thread_started = 1; + } + return e; + +fail: + engine_destroy(e); + return NULL; +} + +void engine_destroy(engine *e) { + if (!e) return; + + if (e->loops) { + for (uint32_t i = 0; i < e->nloops; i++) { + loop *lp = &e->loops[i]; + if (lp->thread_started) { + atomic_store_explicit(&lp->stop, 1, memory_order_relaxed); + if (lp->cmd_efd >= 0) { + uint64_t one = 1; + ssize_t w = write(lp->cmd_efd, &one, sizeof one); + (void)w; + } + } + } + for (uint32_t i = 0; i < e->nloops; i++) { + loop *lp = &e->loops[i]; + if (lp->thread_started) { + pthread_join(lp->thread, NULL); + lp->thread_started = 0; + } + } + /* Threads are gone: tear down loop-owned state without contention. */ + for (uint32_t i = 0; i < e->nloops; i++) { + loop *lp = &e->loops[i]; + conn *c = lp->conns; + while (c) { conn *nx = c->next; conn_destroy(c); c = nx; } + cmd *cm = lp->cmd_head; + while (cm) { + cmd *nx = cm->next; + if (cm->kind == CMD_SET_PRIORITIES) free(cm->pri); + free(cm); + cm = nx; + } + pthread_mutex_destroy(&lp->cmd_lock); + if (lp->cmd_efd >= 0) close(lp->cmd_efd); + if (lp->epfd >= 0) close(lp->epfd); + free(lp->recvbuf); + desc_ring_free(&lp->ready_ring); + slot_ring_free(&lp->free_ring); + if (lp->arena) arena_free(lp->arena); + } + free(e->loops); + } + + torrent *t = e->torrents; + while (t) { + torrent *nx = t->enext; + free(t->priority); + free(t->requested); + if (t->recv_bits) { + for (uint32_t p = 0; p < t->num_pieces; p++) free(t->recv_bits[p]); + free(t->recv_bits); + } + free(t); + t = nx; + } + + if (e->ready_efd >= 0) close(e->ready_efd); + pthread_mutex_destroy(&e->lock); + free(e); +} + +/* ---- control plane --------------------------------------------------- */ + +int32_t engine_add_torrent(engine *e, const uint8_t info_hash[20], + const uint8_t peer_id[20], uint64_t piece_length, + uint64_t total_size, uint32_t num_pieces) { + if (!e || num_pieces == 0 || piece_length == 0 || total_size == 0) return -1; + + torrent *t = calloc(1, sizeof *t); + if (!t) return -1; + t->eng = e; + memcpy(t->info_hash, info_hash, 20); + memcpy(t->peer_id, peer_id, 20); + t->piece_length = piece_length; + t->total_size = total_size; + t->num_pieces = num_pieces; + t->bpp = (uint32_t)((piece_length + PEER_BLOCK_SIZE - 1) / PEER_BLOCK_SIZE); + t->bf_bytes = (num_pieces + 7) / 8; + t->priority = calloc(num_pieces, 1); + t->requested = calloc(num_pieces, 1); + t->recv_bits = calloc(num_pieces, sizeof(*t->recv_bits)); + if (!t->priority || !t->requested || !t->recv_bits) { + free(t->priority); + free(t->requested); + free(t->recv_bits); + free(t); + return -1; + } + + pthread_mutex_lock(&e->lock); + loop *best = &e->loops[0]; + for (uint32_t i = 1; i < e->nloops; i++) + if (e->loops[i].load < best->load) best = &e->loops[i]; + t->lp = best; + best->load++; + t->id = e->next_torrent_id++; + t->enext = e->torrents; + e->torrents = t; + t->next = best->tors; /* bookkeeping; loop thread never walks it */ + best->tors = t; + pthread_mutex_unlock(&e->lock); + + return (int32_t)t->id; +} + +static torrent *find_locked(engine *e, uint32_t id) { + pthread_mutex_lock(&e->lock); + torrent *t = engine_find_torrent(e, id); + pthread_mutex_unlock(&e->lock); + return t; +} + +int engine_add_peer(engine *e, uint32_t torrent_id, const char *ip, uint16_t port) { + if (!e || !ip) return -1; + torrent *t = find_locked(e, torrent_id); + if (!t) return -1; + + cmd *c = calloc(1, sizeof *c); + if (!c) return -1; + c->kind = CMD_ADD_PEER; + c->tor = t; + strncpy(c->ip, ip, sizeof c->ip - 1); + c->port = port; + loop_post(t->lp, c); + return 0; +} + +int engine_set_priorities(engine *e, uint32_t torrent_id, + const uint8_t *priorities, uint32_t count) { + if (!e || !priorities) return -1; + torrent *t = find_locked(e, torrent_id); + if (!t || count != t->num_pieces) return -1; + + uint8_t *copy = malloc(count); + if (!copy) return -1; + memcpy(copy, priorities, count); + + cmd *c = calloc(1, sizeof *c); + if (!c) { free(copy); return -1; } + c->kind = CMD_SET_PRIORITIES; + c->tor = t; + c->pri = copy; + c->pri_count = count; + loop_post(t->lp, c); + return 0; +} + +int engine_set_priority(engine *e, uint32_t torrent_id, uint32_t piece, + uint8_t value) { + if (!e) return -1; + torrent *t = find_locked(e, torrent_id); + if (!t || piece >= t->num_pieces) return -1; + + cmd *c = calloc(1, sizeof *c); + if (!c) return -1; + c->kind = CMD_SET_PRIORITY; + c->tor = t; + c->piece = piece; + c->value = value; + loop_post(t->lp, c); + return 0; +} + +int engine_request_piece(engine *e, uint32_t torrent_id, uint32_t piece) { + if (!e) return -1; + torrent *t = find_locked(e, torrent_id); + if (!t || piece >= t->num_pieces) return -1; + + cmd *c = calloc(1, sizeof *c); + if (!c) return -1; + c->kind = CMD_REQUEST_PIECE; + c->tor = t; + c->piece = piece; + loop_post(t->lp, c); + return 0; +} + +void engine_set_download_rate(engine *e, uint64_t bytes_per_sec) { + if (e) rate_set(&e->dl_limit, bytes_per_sec); +} + +/* ---- data plane ------------------------------------------------------ */ + +uint32_t engine_poll_ready(engine *e, engine_block *out, uint32_t max) { + uint32_t n = 0; + for (uint32_t i = 0; i < e->nloops && n < max; i++) { + desc_ring *r = &e->loops[i].ready_ring; + while (n < max && desc_ring_pop(r, &out[n])) n++; + } + return n; +} + +void engine_release_slot(engine *e, uint32_t loop, uint32_t slot) { + if (!e || loop >= e->nloops) return; + slot_ring_push(&e->loops[loop].free_ring, slot); + /* If the loop parked out of credit, wake it so the returned slot becomes a + * request immediately. Coalesced: only the first release after starvation + * pays the eventfd write. */ + if (atomic_exchange_explicit(&e->loops[loop].want_release_wake, 0, + memory_order_relaxed)) { + uint64_t one = 1; + ssize_t w = write(e->loops[loop].cmd_efd, &one, sizeof one); + (void)w; + } +} + +int engine_wait(engine *e, int timeout_ms) { + struct pollfd pfd = { .fd = e->ready_efd, .events = POLLIN, .revents = 0 }; + int r = poll(&pfd, 1, timeout_ms); + if (r < 0) return -1; + if (r == 0) return 0; + uint64_t drain; + ssize_t rd = read(e->ready_efd, &drain, sizeof drain); + (void)rd; + return 1; +} + +void *engine_arena_base(engine *e, uint32_t loop) { + if (!e || loop >= e->nloops) return NULL; + return e->loops[loop].arena; +} + +uint64_t engine_arena_bytes(engine *e, uint32_t loop) { + if (!e || loop >= e->nloops) return 0; + return e->loops[loop].arena_bytes; +} + +uint32_t engine_loop_count(engine *e) { return e ? e->nloops : 0; } + +void engine_torrent_status(engine *e, uint32_t torrent_id, torrent_status *out) { + memset(out, 0, sizeof *out); + out->state = PEER_STATE_IDLE; + if (!e) return; + torrent *t = find_locked(e, torrent_id); + if (!t) return; + + loop *lp = t->lp; + int any_running = 0, max_state = PEER_STATE_IDLE, err = PEER_OK; + uint32_t peers = 0, connected = 0, failed = 0, outst = 0, ptarget = 0; + uint64_t bytes = 0, blocks = 0; + double rate = 0.0, rttmin = 0.0; + + for (conn *c = lp->conns; c; c = c->next) { + if (c->tor != t) continue; + peers++; + int st = atomic_load_explicit(&c->astate, memory_order_relaxed); + int er = atomic_load_explicit(&c->aerror, memory_order_relaxed); + if (er != PEER_OK) err = er; + if (st == PEER_STATE_ERROR) failed++; + if (st == PEER_STATE_RUNNING || st == PEER_STATE_CHOKED) connected++; + if (st == PEER_STATE_RUNNING) any_running = 1; + if (st > max_state) max_state = st; + bytes += atomic_load_explicit(&c->bytes_received, memory_order_relaxed); + blocks += atomic_load_explicit(&c->blocks_received, memory_order_relaxed); + outst += atomic_load_explicit(&c->outstanding, memory_order_relaxed); + ptarget += c->pipeline_target; + rate += c->rate_bps; + double rm = (double)c->rtt_min_ns / 1e6; + if (c->rtt_min_ns && (rttmin == 0.0 || rm < rttmin)) rttmin = rm; + } + + out->state = peers ? (any_running ? PEER_STATE_RUNNING : max_state) + : PEER_STATE_IDLE; + out->error = err; + out->bytes_received = bytes; + out->blocks_received = blocks; + out->peers = peers; + out->peers_connected = connected; + out->peers_failed = failed; + out->outstanding = outst; + out->free_slots = slot_ring_count(&lp->free_ring); + out->pipeline_target = ptarget; + out->rate_bps = rate; + out->rtt_min_ms = rttmin; +} + +static const char *peer_state_name(int state) { + switch (state) { + case PEER_STATE_IDLE: return "idle"; + case PEER_STATE_CONNECTING: return "connecting"; + case PEER_STATE_HANDSHAKE: return "handshake"; + case PEER_STATE_CHOKED: return "choked"; + case PEER_STATE_RUNNING: return "running"; + case PEER_STATE_STOPPED: return "stopped"; + case PEER_STATE_ERROR: return "error"; + default: return "?"; + } +} + +void engine_dump_torrent(engine *e, uint32_t torrent_id, FILE *out) { + if (!e || !out) return; + torrent *t = find_locked(e, torrent_id); + if (!t) { fprintf(out, "engine: torrent %u not found\n", torrent_id); return; } + loop *lp = t->lp; + + /* Tally availability (connected peers advertising the piece) and in-flight + * block requests per piece in a single pass over the loop's connections, + * then walk the wanted pieces once. Avoids an O(pieces * conns) scan. */ + uint32_t *avail = calloc(t->num_pieces, sizeof *avail); + uint32_t *inflight = calloc(t->num_pieces, sizeof *inflight); + if (!avail || !inflight) { + free(avail); free(inflight); + fprintf(out, "engine: out of memory rendering dump for torrent %u\n", + torrent_id); + return; + } + + fprintf(out, "=== engine dump: torrent %u (%u pieces, %llu B/piece) ===\n", + torrent_id, t->num_pieces, (unsigned long long)t->piece_length); + fprintf(out, "loop %d: outstanding=%u free_slots=%u\n", + lp->index, lp->outstanding, slot_ring_count(&lp->free_ring)); + + fprintf(out, "connections:\n"); + uint32_t conns = 0, connected = 0; + for (conn *c = lp->conns; c; c = c->next) { + if (c->tor != t) continue; + conns++; + int st = atomic_load_explicit(&c->astate, memory_order_relaxed); + bool up = (st == PEER_STATE_RUNNING || st == PEER_STATE_CHOKED) && !c->dead; + if (up) connected++; + + uint32_t have = 0; + if (c->have_bits) + for (uint32_t i = 0; i < t->num_pieces; i++) { + if (!have_bit(c->have_bits, i)) continue; + have++; + if (up) avail[i]++; + } + if (c->inflight.slots) { + uint32_t cap = c->inflight.mask + 1; + for (uint32_t i = 0; i < cap; i++) { + if (c->inflight.slots[i].key == REQ_EMPTY) continue; + uint32_t p = c->inflight.slots[i].piece; + if (p < t->num_pieces) inflight[p]++; + } + } + + uint32_t out_reqs = atomic_load_explicit(&c->outstanding, + memory_order_relaxed); + uint64_t rx = atomic_load_explicit(&c->bytes_received, + memory_order_relaxed); + fprintf(out, + " %s:%u state=%s%s unchoked=%d cur_piece=%s out=%u " + "requeue=%u rate=%.1fKB/s have=%u/%u rx=%lluB\n", + c->ip, c->port, peer_state_name(st), c->dead ? "(dead)" : "", + c->unchoked, c->have_cur_piece ? "" : "-", + out_reqs, c->requeue_count, c->rate_bps / 1024.0, + have, t->num_pieces, (unsigned long long)rx); + if (c->have_cur_piece) + fprintf(out, " (working piece %u)\n", c->cur_piece); + } + fprintf(out, "peers: %u total, %u connected\n", conns, connected); + + /* Per-piece breakdown of everything still wanted. At the tail of a download + * this is the handful of pieces that refuse to finish. */ + uint32_t wanted = 0, claimed = 0, starved = 0; + for (uint32_t i = 0; i < t->num_pieces; i++) { + if (t->priority[i] == 0) continue; + wanted++; + if (t->requested[i]) claimed++; + if (avail[i] == 0) starved++; + } + fprintf(out, + "wanted pieces: %u (claimed=%u, no-connected-peer-has-it=%u)\n", + wanted, claimed, starved); + + uint32_t shown = 0; + const uint32_t limit = 512; + for (uint32_t i = 0; i < t->num_pieces; i++) { + if (t->priority[i] == 0) continue; + if (shown++ >= limit) continue; + fprintf(out, + " piece %u pri=%u claimed=%u avail=%u inflight=%u%s\n", + i, t->priority[i], t->requested[i], avail[i], inflight[i], + (t->requested[i] && inflight[i] == 0) + ? " <-- claimed but no requests in flight" + : (avail[i] == 0 ? " <-- no connected peer has it" : "")); + } + if (wanted > limit) + fprintf(out, " ... %u more wanted pieces not shown\n", wanted - limit); + + free(avail); + free(inflight); +} diff --git a/src/engine_internal.h b/src/engine_internal.h new file mode 100644 index 0000000..98d058b --- /dev/null +++ b/src/engine_internal.h @@ -0,0 +1,300 @@ +/* + * engine_internal.h - Shared internals for the engine modules + * (engine.c, loop.c, connection.c, proto.c, scheduler.c). Not public. + * + * Ownership rules (these are what make the hot path lock-free): + * - A loop thread is the sole owner of its connections, their reader/outbuf, + * its arena, and the piece state of the torrents pinned to it. + * - Cross-thread interaction is via each loop's command queue (mutex-guarded, + * low rate) + an engine-wide ready eventfd. + * - SPSC rings: ready_ring (loop -> consumer), free_ring (consumer -> loop). + */ +#ifndef TORRENT_ENGINE_INTERNAL_H +#define TORRENT_ENGINE_INTERNAL_H + +#include +#include +#include +#include + +#include "../include/engine.h" +#include "ring.h" +#include "transport.h" + +/* BitTorrent peer message ids. */ +enum { + MSG_CHOKE = 0, MSG_UNCHOKE = 1, MSG_INTERESTED = 2, MSG_NOT_INTERESTED = 3, + MSG_HAVE = 4, MSG_BITFIELD = 5, MSG_REQUEST = 6, MSG_PIECE = 7, + MSG_CANCEL = 8, MSG_PORT = 9, MSG_SUGGEST = 13, MSG_HAVE_ALL = 14, + MSG_HAVE_NONE = 15, MSG_REJECT = 16, MSG_ALLOWED_FAST = 17, + MSG_EXTENDED = 20 +}; + +#define EXT_LT_DONTHAVE 1u + +#define HANDSHAKE_LEN 68 +#define PIECE_HDR_LEN 8 +#define MAX_OTHER_MSG (1u << 20) +#define OUTBUF_CAP (1u << 16) +#define REQ_EMPTY UINT64_MAX + +/* ---- wire reader (push model: bytes are fed in, not pulled) ---------- */ +typedef enum { + RS_HANDSHAKE = 0, RS_LEN, RS_ID, RS_PIECE_HDR, RS_PIECE_BODY, + RS_PIECE_DROP, RS_OTHER +} read_state; + +typedef struct { + read_state state; + uint8_t hs[HANDSHAKE_LEN]; size_t hs_got; + uint8_t lenb[4]; size_t len_got; + uint32_t msg_len; + uint8_t msg_id; size_t id_got; + uint8_t phdr[PIECE_HDR_LEN]; size_t phdr_got; + uint32_t cur_piece, cur_begin, body_len, body_got, cur_slot; + uint8_t *cur_slot_ptr; + uint8_t *other; size_t other_cap, other_total, other_got; +} reader; + +typedef struct { uint8_t buf[OUTBUF_CAP]; size_t head, tail; } outbuf; + +/* ---- in-flight request table (per connection, loop-thread-only) ------ */ +typedef struct { + uint64_t key; /* global block id, or REQ_EMPTY */ + uint32_t piece, begin, len; + uint64_t issue_ns; +} req_entry; + +typedef struct { + req_entry *slots; + uint32_t mask, count; +} reqtab; + +typedef struct { uint32_t piece, begin, len; } block_req; + +typedef struct conn conn; +typedef struct torrent torrent; +typedef struct loop loop; + +/* ---- a single peer connection (owned by one loop) -------------------- */ +struct conn { + loop *lp; + torrent *tor; + transport tr; + int connecting; + int unchoked; + int dead; /* errored/closed: skip scheduling, keep + * around for status until teardown */ + int bt_established; /* a valid BitTorrent handshake was received; + * past this point failures are terminal */ + int fast_enabled; /* peer also advertised BEP-6 Fast Ext */ + int ext_enabled; /* peer also advertised BEP-10 */ + uint8_t variant, nvariants; /* current/total transport-fallback combos */ + uint8_t v_utp[4], v_enc[4]; /* the fallback ladder */ + uint32_t cur_events; /* current epoll interest */ + + reader rd; + outbuf out; + uint8_t *have_bits; /* this peer's availability bitfield */ + uint8_t *allowed_fast_bits; /* BEP-6 pieces requestable while choked */ + + /* per-connection scheduler cursor */ + int have_cur_piece; + uint32_t cur_piece; + uint64_t cur_piece_len, cur_block_off; + + /* in-flight tracking + re-queue */ + reqtab inflight; + block_req *requeue; uint32_t requeue_count, requeue_cap; + uint64_t request_timeout_ns, last_timeout_scan_ns; + + /* connect/handshake deadline (loop-thread) */ + uint64_t connect_started_ns; + + /* rate / RTT (loop-thread) */ + uint64_t last_sample_ns, last_sample_bytes, rtt_min_ns; + double rate_bps; + uint32_t pipeline_target; + + /* status (loop writes, consumer reads) */ + atomic_int astate, aerror; + atomic_uint outstanding; /* this conn's in-flight count */ + atomic_uint_fast64_t bytes_received, blocks_received; + + char ip[64]; + uint16_t port; + conn *next; /* loop's connection list */ +}; + +/* ---- a torrent (pinned to one loop) ---------------------------------- */ +struct torrent { + engine *eng; + loop *lp; + uint32_t id; + + uint8_t info_hash[20]; + uint8_t peer_id[20]; + uint64_t piece_length, total_size; + uint32_t num_pieces, bpp, bf_bytes; + + uint8_t *priority; /* per piece, 0 = skip (loop-thread-only) */ + uint8_t *requested; /* per piece, 1 = claimed (shared across conns) */ + uint8_t **recv_bits; /* [num_pieces] received-block bitmap, lazily + * allocated; lets endgame skip blocks already in */ + int endgame; /* loop-thread: no unclaimed wanted piece remains, so + * idle peers may race blocks of claimed pieces */ + + conn *conns; uint32_t conn_count; + torrent *next; /* loop's torrent list */ + torrent *enext; /* engine's global registry list */ +}; + +/* ---- cross-thread commands delivered to a loop ----------------------- */ +typedef enum { + CMD_ADD_PEER, CMD_SET_PRIORITIES, CMD_SET_PRIORITY, CMD_REQUEST_PIECE, CMD_STOP +} cmd_kind; + +typedef struct cmd { + cmd_kind kind; + torrent *tor; + char ip[64]; + uint16_t port; + uint8_t *pri; /* CMD_SET_PRIORITIES: heap copy, loop frees */ + uint32_t pri_count; + uint32_t piece; + uint8_t value; + struct cmd *next; +} cmd; + +/* ---- an event loop (one OS thread) ----------------------------------- */ +struct loop { + engine *eng; + int index; + pthread_t thread; + int thread_started; + int epfd; + int cmd_efd; /* wake for command-queue / credit return */ + + uint8_t *arena; uint64_t arena_bytes; uint32_t num_slots; + slot_ring free_ring; /* consumer -> loop */ + desc_ring ready_ring; /* loop -> consumer */ + uint8_t *recvbuf; size_t recvbuf_cap; /* shared recv staging (one conn at + * a time on this thread) */ + + uint32_t outstanding; /* loop-wide in-flight (credit accounting)*/ + atomic_int want_release_wake; /* coalesced credit-return wake */ + + pthread_mutex_t cmd_lock; + cmd *cmd_head, *cmd_tail; + + conn *conns; + torrent *tors; + uint32_t load; /* torrents assigned (for balancing) */ + + atomic_int stop; + uint64_t last_timeout_ns; +}; + +/* ---- the engine ------------------------------------------------------ */ +/* Engine-wide download throttle (leech-side). A token bucket over requested + * bytes; gating request issuance bounds the receive rate. rate_bps==0 means + * unlimited. Refilled lazily from peer_now_ns(); guarded by its own lock since + * every loop thread consumes from it. */ +typedef struct { + pthread_mutex_t lock; + _Atomic uint64_t rate_bps; /* 0 => unlimited (read lock-free on hot path) */ + double tokens; /* bytes currently available */ + uint64_t last_ns; + uint64_t burst; /* token cap */ +} rate_limiter; + +void rate_set(rate_limiter *rl, uint64_t bytes_per_sec); + +/* Consume `bytes` if available; returns false (without consuming) when the + * bucket is empty so the caller can defer issuing. Always true when unlimited. */ +bool rate_try_consume(rate_limiter *rl, uint32_t bytes); + +struct engine { + engine_config cfg; + loop *loops; uint32_t nloops; + int ready_efd; /* shared: any loop signals, consumer waits */ + pthread_mutex_t lock; /* guards torrent registry + assignment */ + torrent *torrents; /* global registry (by id) */ + uint32_t next_torrent_id; + rate_limiter dl_limit; /* engine-wide download throttle */ +}; + +/* ---- loop.c ---------------------------------------------------------- */ +void *loop_run(void *arg); /* reactor thread entry point */ + +/* ---- helpers (engine.c) ---------------------------------------------- */ +uint64_t peer_now_ns(void); +uint64_t torrent_piece_len(const torrent *t, uint32_t piece); +torrent *engine_find_torrent(engine *e, uint32_t id); +void loop_post(loop *lp, cmd *c); /* enqueue + wake (thread-safe) */ + +/* ---- reqtab.c -------------------------------------------------------- */ +int reqtab_init(reqtab *t, uint32_t capacity); +void reqtab_free(reqtab *t); +int reqtab_insert(reqtab *t, uint64_t key, uint32_t piece, uint32_t begin, + uint32_t len, uint64_t issue_ns); +int reqtab_take(reqtab *t, uint64_t key, req_entry *out); +int reqtab_has(const reqtab *t, uint64_t key); +uint32_t reqtab_take_piece(reqtab *t, uint32_t piece); + +static inline uint64_t block_key(const torrent *t, uint32_t piece, uint32_t begin) { + return (uint64_t)piece * t->bpp + begin / PEER_BLOCK_SIZE; +} + +/* ---- availability bitfield (MSB-first) ------------------------------- */ +static inline int have_bit(const uint8_t *bf, uint32_t i) { + return (bf[i >> 3] >> (7 - (i & 7))) & 1; +} +static inline void set_have_bit(uint8_t *bf, uint32_t i) { + bf[i >> 3] |= (uint8_t)(0x80u >> (i & 7)); +} +static inline void clear_have_bit(uint8_t *bf, uint32_t i) { + bf[i >> 3] &= (uint8_t)~(0x80u >> (i & 7)); +} +static inline void set_all_have_bits(uint8_t *bf, uint32_t nbits) { + uint32_t nbytes = (nbits + 7) / 8; + for (uint32_t i = 0; i < nbytes; i++) bf[i] = 0xff; + if ((nbits & 7) != 0) bf[nbytes - 1] &= (uint8_t)(0xffu << (8 - (nbits & 7))); +} + +/* ---- outbuf ---------------------------------------------------------- */ +static inline size_t outbuf_pending(const outbuf *o) { return o->tail - o->head; } +static inline size_t outbuf_space(const outbuf *o) { return OUTBUF_CAP - o->tail; } +void outbuf_compact(outbuf *o); +void outbuf_append(outbuf *o, const void *data, size_t n); + +/* ---- proto.c (push-model parser) ------------------------------------- */ +void proto_queue_handshake(conn *c); +void proto_queue_msg(conn *c, uint8_t id); +void proto_queue_request(conn *c, uint32_t piece, uint32_t begin, uint32_t length); +void proto_queue_cancel(conn *c, uint32_t piece, uint32_t begin, uint32_t length); +/* Feed received bytes to the parser. Returns 0 ok, -1 fatal (sets conn error). + * Pushes completed blocks to the loop ready_ring and signals the engine. */ +int proto_feed(conn *c, const uint8_t *data, size_t len); +void conn_set_error(conn *c, peer_error e); + +/* ---- scheduler.c ----------------------------------------------------- */ +void scheduler_tick(conn *c); +void scheduler_check_timeouts(conn *c); +/* Record a delivered block so endgame won't re-request it; clear a piece's + * record when it is re-armed for download (hash failure). Loop-thread only. */ +void torrent_mark_received(torrent *t, uint32_t piece, uint32_t begin); +void torrent_reset_received(torrent *t, uint32_t piece); + +/* ---- connection.c ---------------------------------------------------- */ +conn *conn_create(loop *lp, torrent *tor, const char *ip, uint16_t port); +void conn_destroy(conn *c); +void conn_drive_handshake(conn *c); /* advance transport handshake (connect/MSE) */ +void conn_on_readable(conn *c); /* recv -> proto_feed */ +/* Pre-handshake failure: retry the next transport in the fallback ladder, or + * mark the connection failed if none remain (or the handshake was established). */ +void conn_fail_or_fallback(conn *c, peer_error e); +int conn_flush(conn *c); /* drain outbuf via transport */ +void conn_update_interest(conn *c); /* recompute epoll interest after a tick */ + +#endif /* TORRENT_ENGINE_INTERNAL_H */ diff --git a/src/loop.c b/src/loop.c new file mode 100644 index 0000000..6e80393 --- /dev/null +++ b/src/loop.c @@ -0,0 +1,154 @@ +/* + * loop.c - The reactor: one OS thread per event loop. + * + * The loop is the sole owner of its connections, its arena/rings, and the piece + * state of the torrents pinned to it, so the entire hot path (recv -> parse -> + * handoff -> schedule) runs lock-free. The only cross-thread inputs are the + * command queue (mutex-guarded, low rate) and credit-return wakes; both arrive + * via cmd_efd. Completed blocks leave through the loop's ready_ring and the + * engine's shared ready eventfd. + */ +#include "engine_internal.h" + +#include +#include +#include +#include + +#include + +#define MAX_EVENTS 64 +#define RUNNING_TICK_MS 2 /* poll cadence so freed slots become requests */ +#define TIMEOUT_SCAN_NS (250ull * 1000000ull) /* how often to scan for stalls */ + +/* EWMA download rate per connection (loop-thread only; status reads racily). */ +static void update_rate(conn *c) { + uint64_t now = peer_now_ns(); + uint64_t dt = now - c->last_sample_ns; + if (c->last_sample_ns == 0) { c->last_sample_ns = now; return; } + if (dt < 100000000ull) return; /* sample at ~10 Hz */ + uint64_t bytes = atomic_load_explicit(&c->bytes_received, memory_order_relaxed); + double inst = (double)(bytes - c->last_sample_bytes) * 1e9 / (double)dt; + c->rate_bps = c->rate_bps * 0.6 + inst * 0.4; + c->last_sample_ns = now; + c->last_sample_bytes = bytes; +} + +/* Drain and apply the command queue. conn_create / priority updates touch + * loop-owned state, so they must run on this thread. */ +static void process_commands(loop *lp) { + pthread_mutex_lock(&lp->cmd_lock); + cmd *list = lp->cmd_head; + lp->cmd_head = lp->cmd_tail = NULL; + pthread_mutex_unlock(&lp->cmd_lock); + + while (list) { + cmd *c = list; + list = list->next; + switch (c->kind) { + case CMD_ADD_PEER: + conn_create(lp, c->tor, c->ip, c->port); + break; + case CMD_SET_PRIORITIES: + if (c->pri_count == c->tor->num_pieces) + memcpy(c->tor->priority, c->pri, c->pri_count); + free(c->pri); + break; + case CMD_SET_PRIORITY: + if (c->piece < c->tor->num_pieces) c->tor->priority[c->piece] = c->value; + break; + case CMD_REQUEST_PIECE: + if (c->piece < c->tor->num_pieces) { + c->tor->requested[c->piece] = 0; + /* Re-armed after a hash failure: forget which blocks we had so + * endgame re-requests the whole piece. */ + torrent_reset_received(c->tor, c->piece); + } + break; + case CMD_STOP: + break; /* teardown is driven by the atomic stop flag */ + } + free(c); + } +} + +void *loop_run(void *arg) { + loop *lp = arg; + lp->last_timeout_ns = peer_now_ns(); + + struct epoll_event evs[MAX_EVENTS]; + while (!atomic_load_explicit(&lp->stop, memory_order_relaxed)) { + int connecting = 0, has_conn = 0; + for (conn *c = lp->conns; c; c = c->next) { + if (c->dead) continue; + has_conn = 1; + if (c->connecting) { connecting = 1; break; } + } + int timeout = has_conn ? (connecting ? 1000 : RUNNING_TICK_MS) : 1000; + + int n = epoll_wait(lp->epfd, evs, MAX_EVENTS, timeout); + if (n < 0) { + if (errno == EINTR) continue; + break; + } + + for (int i = 0; i < n; i++) { + void *ptr = evs[i].data.ptr; + if (ptr == NULL) { /* cmd_efd: commands and/or credit-return wake */ + uint64_t drain; + ssize_t r = read(lp->cmd_efd, &drain, sizeof drain); + (void)r; + continue; + } + conn *c = ptr; + if (c->dead) continue; + uint32_t e = evs[i].events; + if (c->connecting) { + /* Let the transport handshake decide success/failure (it checks + * SO_ERROR); EPOLLHUP can accompany a perfectly good connect. */ + if (e & (EPOLLIN | EPOLLOUT | EPOLLERR | EPOLLHUP)) + conn_drive_handshake(c); + continue; + } + if (e & (EPOLLHUP | EPOLLERR)) { + conn_fail_or_fallback(c, PEER_ERR_CLOSED); + continue; + } + if (e & EPOLLIN) conn_on_readable(c); + if ((e & EPOLLOUT) && !c->dead) conn_flush(c); + } + + process_commands(lp); + + uint64_t now = peer_now_ns(); + int scan = (now - lp->last_timeout_ns >= TIMEOUT_SCAN_NS); + if (scan) lp->last_timeout_ns = now; + + uint64_t connect_timeout_ns = + (uint64_t)lp->eng->cfg.connect_timeout_ms * 1000000ull; + for (conn *c = lp->conns; c; c = c->next) { + if (c->dead) continue; + if (c->tr.pump) c->tr.pump(&c->tr); /* µTP retransmit / delayed ack */ + if (scan) { + scheduler_check_timeouts(c); + /* Reclaim a peer stuck connecting or mid-handshake: it never + * answered, so it occupies a slot without ever delivering. */ + int st = atomic_load_explicit(&c->astate, memory_order_relaxed); + if ((st == PEER_STATE_CONNECTING || st == PEER_STATE_HANDSHAKE) && + now - c->connect_started_ns > connect_timeout_ns) { + conn_fail_or_fallback(c, PEER_ERR_CONNECT); + continue; + } + } + scheduler_tick(c); + conn_flush(c); + conn_update_interest(c); + update_rate(c); + } + } + + for (conn *c = lp->conns; c; c = c->next) + if (atomic_load_explicit(&c->astate, memory_order_relaxed) != PEER_STATE_ERROR) + atomic_store_explicit(&c->astate, PEER_STATE_STOPPED, memory_order_relaxed); + return NULL; +} diff --git a/src/peer_compat.c b/src/peer_compat.c new file mode 100644 index 0000000..44895f2 --- /dev/null +++ b/src/peer_compat.c @@ -0,0 +1,126 @@ +/* + * peer_compat.c - Legacy single-peer peer_* ABI implemented as a thin wrapper + * over the multi-peer engine. A peer_handle is a private engine configured with + * exactly one loop, holding one torrent with one connection. The semantics match + * the old standalone peer, so existing callers and tests keep working unchanged. + */ +#include "../include/peer.h" +#include "../include/engine.h" + +#include +#include + +struct peer_handle { + engine *eng; + uint32_t tid; + uint32_t num_pieces; +}; + +peer_handle *peer_create(const peer_config *cfg) { + if (!cfg || cfg->num_pieces == 0 || cfg->piece_length == 0 || + cfg->total_size == 0) + return NULL; + + peer_handle *h = calloc(1, sizeof *h); + if (!h) return NULL; + + engine_config ec; + memset(&ec, 0, sizeof ec); + ec.loop_count = 1; /* single loop = single peer */ + ec.slots_per_loop = cfg->num_slots; /* 0 => engine default */ + ec.max_pipeline = cfg->max_pipeline; + ec.request_timeout_ms = cfg->request_timeout_ms; + ec.recv_buffer_bytes = cfg->recv_buffer_bytes; + + h->eng = engine_create(&ec); + if (!h->eng) { free(h); return NULL; } + + int32_t id = engine_add_torrent(h->eng, cfg->info_hash, cfg->peer_id, + cfg->piece_length, cfg->total_size, + cfg->num_pieces); + if (id < 0) { engine_destroy(h->eng); free(h); return NULL; } + h->tid = (uint32_t)id; + h->num_pieces = cfg->num_pieces; + return h; +} + +int peer_start(peer_handle *h, const char *ip, uint16_t port) { + if (!h || !ip) return -1; + return engine_add_peer(h->eng, h->tid, ip, port); +} + +int peer_set_priorities(peer_handle *h, const uint8_t *priorities, uint32_t count) { + if (!h) return -1; + return engine_set_priorities(h->eng, h->tid, priorities, count); +} + +int peer_set_priority(peer_handle *h, uint32_t piece_index, uint8_t priority) { + if (!h) return -1; + return engine_set_priority(h->eng, h->tid, piece_index, priority); +} + +int peer_request_piece(peer_handle *h, uint32_t piece_index) { + if (!h) return -1; + return engine_request_piece(h->eng, h->tid, piece_index); +} + +void peer_stop(peer_handle *h) { + /* The engine has no per-torrent stop; teardown happens in peer_destroy. + * Connections idle harmlessly until then. */ + (void)h; +} + +void peer_destroy(peer_handle *h) { + if (!h) return; + if (h->eng) engine_destroy(h->eng); + free(h); +} + +void *peer_arena_base(const peer_handle *h) { + return engine_arena_base(h->eng, 0); +} + +uint64_t peer_arena_bytes(const peer_handle *h) { + return engine_arena_bytes(h->eng, 0); +} + +uint32_t peer_poll_ready(peer_handle *h, block_desc *out, uint32_t max) { + uint32_t total = 0; + engine_block tmp[256]; + while (total < max) { + uint32_t want = max - total; + if (want > 256) want = 256; + uint32_t n = engine_poll_ready(h->eng, tmp, want); + for (uint32_t i = 0; i < n; i++) { + out[total].piece = tmp[i].piece; + out[total].begin = tmp[i].begin; + out[total].len = tmp[i].len; + out[total].slot = tmp[i].slot; /* single loop => loop index 0 */ + total++; + } + if (n < want) break; + } + return total; +} + +void peer_release_slot(peer_handle *h, uint32_t slot) { + engine_release_slot(h->eng, 0, slot); +} + +int peer_wait(peer_handle *h, int timeout_ms) { + return engine_wait(h->eng, timeout_ms); +} + +void peer_get_status(const peer_handle *h, peer_status *out) { + torrent_status ts; + engine_torrent_status(h->eng, h->tid, &ts); + out->state = ts.state; + out->error = ts.error; + out->bytes_received = ts.bytes_received; + out->blocks_received = ts.blocks_received; + out->outstanding = ts.outstanding; + out->free_slots = ts.free_slots; + out->pipeline_target = ts.pipeline_target; + out->rate_bps = ts.rate_bps; + out->rtt_min_ms = ts.rtt_min_ms; +} diff --git a/src/proto.c b/src/proto.c new file mode 100644 index 0000000..3515c25 --- /dev/null +++ b/src/proto.c @@ -0,0 +1,385 @@ +/* + * proto.c - BitTorrent peer wire protocol: handshake, message framing, and the + * incremental push-model parser. + * + * The loop thread does the recv() and feeds the bytes here via proto_feed(); + * the parser is a resumable state machine that never blocks and never calls + * recv() itself. A piece's payload is memcpy'd out of the fed buffer straight + * into its destination arena slot, so the only copy on the hot path is that one + * (unavoidable, since the bytes already live in the loop's recv staging buffer). + * + * Completed blocks are pushed to the owning loop's ready_ring and the engine's + * shared ready eventfd is signalled so the single consumer wakes. + */ +#include "engine_internal.h" + +#include +#include +#include + +/* pstrlen (19) followed by "BitTorrent protocol"; exactly 20 bytes, no NUL. */ +static const uint8_t BT_PROTOCOL[20] = { + 19, 'B','i','t','T','o','r','r','e','n','t',' ','p','r','o','t','o','c','o','l' +}; + +static inline uint32_t be32(const uint8_t *p) { + return ((uint32_t)p[0] << 24) | ((uint32_t)p[1] << 16) | + ((uint32_t)p[2] << 8) | (uint32_t)p[3]; +} +static inline void put_be32(uint8_t *p, uint32_t v) { + p[0] = (uint8_t)(v >> 24); p[1] = (uint8_t)(v >> 16); + p[2] = (uint8_t)(v >> 8); p[3] = (uint8_t)v; +} + +/* ---- outgoing message construction --------------------------------- */ + +void proto_queue_handshake(conn *c) { + uint8_t hs[HANDSHAKE_LEN]; + memcpy(hs, BT_PROTOCOL, 20); + memset(hs + 20, 0, 8); /* reserved */ + hs[25] |= 0x10; /* BEP-10 extension protocol */ + hs[27] |= 0x04; /* BEP-6 Fast Extension */ + memcpy(hs + 28, c->tor->info_hash, 20); + memcpy(hs + 48, c->tor->peer_id, 20); + outbuf_append(&c->out, hs, sizeof hs); +} + +void proto_queue_msg(conn *c, uint8_t id) { + uint8_t m[5]; + put_be32(m, 1); + m[4] = id; + outbuf_append(&c->out, m, sizeof m); +} + +void proto_queue_request(conn *c, uint32_t piece, uint32_t begin, uint32_t length) { + uint8_t m[17]; + put_be32(m, 13); + m[4] = MSG_REQUEST; + put_be32(m + 5, piece); + put_be32(m + 9, begin); + put_be32(m + 13, length); + outbuf_append(&c->out, m, sizeof m); +} + +void proto_queue_cancel(conn *c, uint32_t piece, uint32_t begin, uint32_t length) { + uint8_t m[17]; + put_be32(m, 13); + m[4] = MSG_CANCEL; + put_be32(m + 5, piece); + put_be32(m + 9, begin); + put_be32(m + 13, length); + outbuf_append(&c->out, m, sizeof m); +} + +static void proto_queue_have_none(conn *c) { + proto_queue_msg(c, MSG_HAVE_NONE); +} + +static void proto_queue_ext_handshake(conn *c) { + char payload[160]; + int n = snprintf(payload, sizeof payload, + "d1:md11:lt_donthavei%uee1:reqqi%ue1:v14:torrent-peer/0ee", + EXT_LT_DONTHAVE, c->lp->eng->cfg.max_pipeline); + if (n <= 0 || (size_t)n >= sizeof payload) return; + + uint8_t hdr[6]; + put_be32(hdr, (uint32_t)n + 2); + hdr[4] = MSG_EXTENDED; + hdr[5] = 0; /* extended handshake */ + outbuf_append(&c->out, hdr, sizeof hdr); + outbuf_append(&c->out, payload, (size_t)n); +} + +/* ---- fed-buffer reader --------------------------------------------- */ + +/* Copy up to (need - *got) bytes from the feed cursor [*p, end) into dst. + * Advances the cursor. Returns 1 once *got reaches need, else 0 (need more). */ +static inline int feed_take(const uint8_t **p, const uint8_t *end, void *dst, + size_t need, size_t *got) { + size_t avail = (size_t)(end - *p); + size_t n = need - *got; + if (n > avail) n = avail; + memcpy((uint8_t *)dst + *got, *p, n); + *p += n; + *got += n; + return *got >= need; +} + +static void signal_ready(conn *c) { + uint64_t one = 1; + ssize_t w = write(c->lp->eng->ready_efd, &one, sizeof one); + (void)w; /* eventfd write only fails on overflow; harmless */ +} + +/* Endgame races a block across several peers; once one copy lands, cancel the + * others so we don't pay to receive the same block twice. No-op outside endgame + * (no two connections share an in-flight block then), so the hot path is clean. */ +static void cancel_redundant_copies(conn *src, uint32_t piece, uint32_t begin, + uint32_t len) { + torrent *t = src->tor; + if (!t->endgame) return; + uint64_t key = block_key(t, piece, begin); + for (conn *o = src->lp->conns; o; o = o->next) { + if (o == src || o->tor != t || o->dead) continue; + req_entry e; + if (reqtab_take(&o->inflight, key, &e)) { + proto_queue_cancel(o, piece, begin, len); + atomic_fetch_sub_explicit(&o->outstanding, 1, memory_order_relaxed); + o->lp->outstanding--; + } + } +} + +static void drop_inflight_piece(conn *c, uint32_t piece) { + uint32_t removed = reqtab_take_piece(&c->inflight, piece); + if (removed == 0) return; + atomic_fetch_sub_explicit(&c->outstanding, removed, memory_order_relaxed); + c->lp->outstanding -= removed; + c->tor->requested[piece] = 0; + if (c->have_cur_piece && c->cur_piece == piece) c->have_cur_piece = 0; +} + +static void handle_reject(conn *c, const uint8_t *payload, size_t len) { + if (!c->fast_enabled || len < 12) { + conn_set_error(c, PEER_ERR_PROTOCOL); + return; + } + + uint32_t piece = be32(payload); + uint32_t begin = be32(payload + 4); + if (piece >= c->tor->num_pieces) { + conn_set_error(c, PEER_ERR_PROTOCOL); + return; + } + + req_entry e; + if (reqtab_take(&c->inflight, block_key(c->tor, piece, begin), &e)) { + atomic_fetch_sub_explicit(&c->outstanding, 1, memory_order_relaxed); + c->lp->outstanding--; + c->tor->requested[piece] = 0; + if (c->have_cur_piece && c->cur_piece == piece) c->have_cur_piece = 0; + } +} + +static void handle_extended(conn *c, const uint8_t *payload, size_t len) { + if (!c->ext_enabled || len < 1) return; + + uint8_t ext_id = payload[0]; + if (ext_id == 0) return; /* handshake: no fields needed */ + if (ext_id != EXT_LT_DONTHAVE || len < 5) return; + + uint32_t piece = be32(payload + 1); + if (piece >= c->tor->num_pieces) return; + clear_have_bit(c->have_bits, piece); + clear_have_bit(c->allowed_fast_bits, piece); + drop_inflight_piece(c, piece); +} + +/* A fully-received non-piece control message (payload after the id). have_bits + * is loop-thread-only, so the selector reads it from the same thread. */ +static void handle_control(conn *c, uint8_t id, const uint8_t *payload, size_t len) { + switch (id) { + case MSG_UNCHOKE: + c->unchoked = 1; + atomic_store_explicit(&c->astate, PEER_STATE_RUNNING, memory_order_relaxed); + break; + case MSG_CHOKE: + c->unchoked = 0; + atomic_store_explicit(&c->astate, PEER_STATE_CHOKED, memory_order_relaxed); + break; + case MSG_BITFIELD: { + size_t n = len < c->tor->bf_bytes ? len : c->tor->bf_bytes; + memcpy(c->have_bits, payload, n); + break; + } + case MSG_HAVE: + if (len >= 4) { + uint32_t idx = be32(payload); + if (idx < c->tor->num_pieces) set_have_bit(c->have_bits, idx); + } + break; + case MSG_SUGGEST: + if (!c->fast_enabled || len < 4) conn_set_error(c, PEER_ERR_PROTOCOL); + break; + case MSG_HAVE_ALL: + if (!c->fast_enabled) { + conn_set_error(c, PEER_ERR_PROTOCOL); + break; + } + set_all_have_bits(c->have_bits, c->tor->num_pieces); + break; + case MSG_HAVE_NONE: + if (!c->fast_enabled) { + conn_set_error(c, PEER_ERR_PROTOCOL); + break; + } + memset(c->have_bits, 0, c->tor->bf_bytes); + memset(c->allowed_fast_bits, 0, c->tor->bf_bytes); + break; + case MSG_REJECT: + handle_reject(c, payload, len); + break; + case MSG_ALLOWED_FAST: + if (!c->fast_enabled || len < 4) { + conn_set_error(c, PEER_ERR_PROTOCOL); + break; + } + { + uint32_t idx = be32(payload); + if (idx < c->tor->num_pieces) set_have_bit(c->allowed_fast_bits, idx); + } + break; + case MSG_EXTENDED: + handle_extended(c, payload, len); + break; + default: + /* not-interested/port/fast-extension/etc: ignored by a leech. */ + break; + } +} + +void conn_set_error(conn *c, peer_error e) { + atomic_store_explicit(&c->aerror, (int)e, memory_order_relaxed); + atomic_store_explicit(&c->astate, PEER_STATE_ERROR, memory_order_relaxed); +} + +/* ---- the parser ----------------------------------------------------- */ + +int proto_feed(conn *c, const uint8_t *data, size_t len) { + reader *r = &c->rd; + torrent *tor = c->tor; + loop *lp = c->lp; + const uint8_t *p = data; + const uint8_t *end = data + len; + int pushed = 0; + + while (p < end) { + switch (r->state) { + case RS_HANDSHAKE: + if (!feed_take(&p, end, r->hs, HANDSHAKE_LEN, &r->hs_got)) goto out; + if (memcmp(r->hs, BT_PROTOCOL, 20) != 0 || + memcmp(r->hs + 28, tor->info_hash, 20) != 0) { + conn_set_error(c, PEER_ERR_HANDSHAKE); + return -1; + } + c->fast_enabled = (r->hs[27] & 0x04) != 0; + c->ext_enabled = (r->hs[25] & 0x10) != 0; + if (c->fast_enabled) proto_queue_have_none(c); + if (c->ext_enabled) proto_queue_ext_handshake(c); + /* Express interest; wait for unchoke or Allowed Fast. */ + c->bt_established = 1; + proto_queue_msg(c, MSG_INTERESTED); + atomic_store_explicit(&c->astate, PEER_STATE_CHOKED, memory_order_relaxed); + r->state = RS_LEN; + r->len_got = 0; + break; + + case RS_LEN: + if (!feed_take(&p, end, r->lenb, 4, &r->len_got)) goto out; + r->msg_len = be32(r->lenb); + r->len_got = 0; + if (r->msg_len == 0) break; /* keep-alive */ + r->id_got = 0; + r->state = RS_ID; + break; + + case RS_ID: + if (!feed_take(&p, end, &r->msg_id, 1, &r->id_got)) goto out; + if (r->msg_id == MSG_PIECE) { + if (r->msg_len < 1 + PIECE_HDR_LEN || + r->msg_len - 1 - PIECE_HDR_LEN > PEER_BLOCK_SIZE) { + conn_set_error(c, PEER_ERR_PROTOCOL); + return -1; + } + r->body_len = r->msg_len - 1 - PIECE_HDR_LEN; + r->phdr_got = 0; + r->state = RS_PIECE_HDR; + } else { + uint32_t payload = r->msg_len - 1; + if (payload == 0) { + handle_control(c, r->msg_id, NULL, 0); + if (atomic_load_explicit(&c->aerror, memory_order_relaxed) != PEER_OK) + return -1; + r->state = RS_LEN; + } else { + if (payload > r->other_cap) { + conn_set_error(c, PEER_ERR_PROTOCOL); + return -1; + } + r->other_total = payload; + r->other_got = 0; + r->state = RS_OTHER; + } + } + break; + + case RS_PIECE_HDR: { + if (!feed_take(&p, end, r->phdr, PIECE_HDR_LEN, &r->phdr_got)) goto out; + r->cur_piece = be32(r->phdr); + r->cur_begin = be32(r->phdr + 4); + /* C2: only accept a block we actually requested; otherwise drain and + * drop it without consuming a slot or disturbing credit. */ + req_entry e; + if (!reqtab_take(&c->inflight, block_key(tor, r->cur_piece, r->cur_begin), &e)) { + r->other_total = r->body_len; /* <= block size <= other_cap */ + r->other_got = 0; + r->state = RS_PIECE_DROP; + break; + } + uint64_t rtt = peer_now_ns() - e.issue_ns; + if (c->rtt_min_ns == 0 || rtt < c->rtt_min_ns) c->rtt_min_ns = rtt; + atomic_fetch_sub_explicit(&c->outstanding, 1, memory_order_relaxed); + lp->outstanding--; + + if (!slot_ring_pop(&lp->free_ring, &r->cur_slot)) { + /* Flow control guarantees a free slot here. */ + conn_set_error(c, PEER_ERR_PROTOCOL); + return -1; + } + r->cur_slot_ptr = lp->arena + (uint64_t)r->cur_slot * PEER_BLOCK_SIZE; + r->body_got = 0; + r->state = RS_PIECE_BODY; + break; + } + + case RS_PIECE_BODY: { + size_t avail = (size_t)(end - p); + size_t n = r->body_len - r->body_got; + if (n > avail) n = avail; + memcpy(r->cur_slot_ptr + r->body_got, p, n); + p += n; + r->body_got += n; + if (r->body_got < r->body_len) goto out; + engine_block d = { tor->id, r->cur_piece, r->cur_begin, r->body_len, + (uint32_t)lp->index, r->cur_slot }; + /* ready_ring capacity >= num_slots, so this cannot fail. */ + desc_ring_push(&lp->ready_ring, d); + torrent_mark_received(tor, r->cur_piece, r->cur_begin); + cancel_redundant_copies(c, r->cur_piece, r->cur_begin, r->body_len); + atomic_fetch_add_explicit(&c->bytes_received, r->body_len, + memory_order_relaxed); + atomic_fetch_add_explicit(&c->blocks_received, 1, memory_order_relaxed); + pushed = 1; + r->state = RS_LEN; + break; + } + + case RS_PIECE_DROP: + if (!feed_take(&p, end, r->other, r->other_total, &r->other_got)) goto out; + r->state = RS_LEN; + break; + + case RS_OTHER: + if (!feed_take(&p, end, r->other, r->other_total, &r->other_got)) goto out; + handle_control(c, r->msg_id, r->other, r->other_total); + if (atomic_load_explicit(&c->aerror, memory_order_relaxed) != PEER_OK) + return -1; + r->state = RS_LEN; + break; + } + } + +out: + if (pushed) signal_ready(c); + return 0; +} diff --git a/src/reqtab.c b/src/reqtab.c new file mode 100644 index 0000000..cfaac3e --- /dev/null +++ b/src/reqtab.c @@ -0,0 +1,112 @@ +/* + * reqtab.c - Open-addressing hash set of outstanding block requests. + * + * Net-thread-only, so no locking. Linear probing with canonical backward-shift + * deletion (Knuth Algorithm R) keeps the table tombstone-free. The caller keeps + * the load factor <= 0.5 (capacity = 2 * max_pipeline), so probe chains stay + * short and insert never fails. + */ +#include "engine_internal.h" +#include "ring.h" /* peer_next_pow2 */ + +#include + +static inline uint32_t mix64(uint64_t k) { + k ^= k >> 33; k *= 0xff51afd7ed558ccdULL; + k ^= k >> 33; k *= 0xc4ceb9fe1a85ec53ULL; + k ^= k >> 33; + return (uint32_t)k; +} + +int reqtab_init(reqtab *t, uint32_t capacity) { + capacity = peer_next_pow2(capacity < 16 ? 16 : capacity); + t->slots = malloc((size_t)capacity * sizeof(*t->slots)); + if (!t->slots) return -1; + for (uint32_t i = 0; i < capacity; i++) t->slots[i].key = REQ_EMPTY; + t->mask = capacity - 1; + t->count = 0; + return 0; +} + +void reqtab_free(reqtab *t) { + free(t->slots); + t->slots = NULL; +} + +int reqtab_insert(reqtab *t, uint64_t key, uint32_t piece, uint32_t begin, + uint32_t len, uint64_t issue_ns) { + uint32_t mask = t->mask; + uint32_t i = mix64(key) & mask; + int inserted = 0; + while (t->slots[i].key != REQ_EMPTY) { + if (t->slots[i].key == key) break; /* re-issue: overwrite in place */ + i = (i + 1) & mask; + } + if (t->slots[i].key == REQ_EMPTY) { + t->count++; + inserted = 1; + } + t->slots[i].key = key; + t->slots[i].piece = piece; + t->slots[i].begin = begin; + t->slots[i].len = len; + t->slots[i].issue_ns = issue_ns; + return inserted; +} + +int reqtab_has(const reqtab *t, uint64_t key) { + if (!t->slots) return 0; + uint32_t mask = t->mask; + uint32_t i = mix64(key) & mask; + while (t->slots[i].key != REQ_EMPTY) { + if (t->slots[i].key == key) return 1; + i = (i + 1) & mask; + } + return 0; +} + +int reqtab_take(reqtab *t, uint64_t key, req_entry *out) { + uint32_t mask = t->mask; + uint32_t i = mix64(key) & mask; + while (t->slots[i].key != REQ_EMPTY) { + if (t->slots[i].key == key) { + *out = t->slots[i]; + /* backward-shift deletion to fill the gap at i */ + uint32_t j = i; + for (;;) { + t->slots[i].key = REQ_EMPTY; + uint32_t k; + do { + j = (j + 1) & mask; + if (t->slots[j].key == REQ_EMPTY) { t->count--; return 1; } + k = mix64(t->slots[j].key) & mask; + /* keep advancing while slot j must stay (k in (i, j]) */ + } while ((i <= j) ? (i < k && k <= j) : (i < k || k <= j)); + t->slots[i] = t->slots[j]; + i = j; + } + } + i = (i + 1) & mask; + } + return 0; +} + +uint32_t reqtab_take_piece(reqtab *t, uint32_t piece) { + uint32_t removed = 0; + if (!t || !t->slots) return 0; + + for (;;) { + uint64_t key = REQ_EMPTY; + uint32_t cap = t->mask + 1; + for (uint32_t i = 0; i < cap; i++) { + if (t->slots[i].key != REQ_EMPTY && t->slots[i].piece == piece) { + key = t->slots[i].key; + break; + } + } + if (key == REQ_EMPTY) break; + req_entry ignored; + if (reqtab_take(t, key, &ignored)) removed++; + } + return removed; +} diff --git a/src/ring.c b/src/ring.c new file mode 100644 index 0000000..9dd35ba --- /dev/null +++ b/src/ring.c @@ -0,0 +1,34 @@ +/* ring.c - allocation/teardown for the SPSC rings (hot path is in ring.h). */ +#include "ring.h" + +#include + +int slot_ring_init(slot_ring *r, uint32_t capacity) { + capacity = peer_next_pow2(capacity); + r->buf = malloc((size_t)capacity * sizeof(*r->buf)); + if (!r->buf) return -1; + r->mask = capacity - 1; + atomic_init(&r->head, 0); + atomic_init(&r->tail, 0); + return 0; +} + +void slot_ring_free(slot_ring *r) { + free(r->buf); + r->buf = NULL; +} + +int desc_ring_init(desc_ring *r, uint32_t capacity) { + capacity = peer_next_pow2(capacity); + r->buf = malloc((size_t)capacity * sizeof(*r->buf)); + if (!r->buf) return -1; + r->mask = capacity - 1; + atomic_init(&r->head, 0); + atomic_init(&r->tail, 0); + return 0; +} + +void desc_ring_free(desc_ring *r) { + free(r->buf); + r->buf = NULL; +} diff --git a/src/ring.h b/src/ring.h new file mode 100644 index 0000000..2def2e0 --- /dev/null +++ b/src/ring.h @@ -0,0 +1,111 @@ +/* + * ring.h - Bounded single-producer/single-consumer lock-free rings. + * + * Two specializations are provided: `slot_ring` (uint32 slot indices, used for + * the free list) and `desc_ring` (engine_block, used for completed blocks). + * + * The hot-path push/pop are static-inline. Head/tail are free-running counters + * on separate cache lines; the index into the buffer is (counter & mask). This + * disambiguates full vs empty without a wasted slot. Correct for capacities up + * to 2^31. A single thread must own each end (SPSC). + * + * Memory ordering: the producer release-stores its counter after writing the + * payload; the consumer acquire-loads it before reading the payload. This pairs + * to guarantee the payload write is visible before the index advance. + */ +#ifndef TORRENT_PEER_RING_H +#define TORRENT_PEER_RING_H + +#include +#include +#include + +#include "../include/engine.h" + +#define PEER_CACHELINE 64 + +/* ---- slot_ring: uint32 elements ------------------------------------- */ + +typedef struct { + uint32_t *buf; + uint32_t mask; /* capacity - 1 (capacity is a power of two) */ + char _pad0[PEER_CACHELINE - sizeof(uint32_t *) - sizeof(uint32_t)]; + _Alignas(PEER_CACHELINE) atomic_uint_fast32_t head; /* consumer cursor */ + _Alignas(PEER_CACHELINE) atomic_uint_fast32_t tail; /* producer cursor */ +} slot_ring; + +/* capacity must be a power of two. Returns 0 on success, -1 on OOM. */ +int slot_ring_init(slot_ring *r, uint32_t capacity); +void slot_ring_free(slot_ring *r); + +static inline int slot_ring_push(slot_ring *r, uint32_t v) { + uint_fast32_t t = atomic_load_explicit(&r->tail, memory_order_relaxed); + uint_fast32_t h = atomic_load_explicit(&r->head, memory_order_acquire); + if ((uint32_t)(t - h) > r->mask) return 0; /* full */ + r->buf[t & r->mask] = v; + atomic_store_explicit(&r->tail, t + 1, memory_order_release); + return 1; +} + +static inline int slot_ring_pop(slot_ring *r, uint32_t *out) { + uint_fast32_t h = atomic_load_explicit(&r->head, memory_order_relaxed); + uint_fast32_t t = atomic_load_explicit(&r->tail, memory_order_acquire); + if (h == t) return 0; /* empty */ + *out = r->buf[h & r->mask]; + atomic_store_explicit(&r->head, h + 1, memory_order_release); + return 1; +} + +static inline uint32_t slot_ring_count(const slot_ring *r) { + uint_fast32_t t = atomic_load_explicit(&r->tail, memory_order_acquire); + uint_fast32_t h = atomic_load_explicit(&r->head, memory_order_acquire); + return (uint32_t)(t - h); +} + +/* ---- desc_ring: engine_block elements ------------------------------- */ + +typedef struct { + engine_block *buf; + uint32_t mask; + char _pad0[PEER_CACHELINE - sizeof(engine_block *) - sizeof(uint32_t)]; + _Alignas(PEER_CACHELINE) atomic_uint_fast32_t head; + _Alignas(PEER_CACHELINE) atomic_uint_fast32_t tail; +} desc_ring; + +int desc_ring_init(desc_ring *r, uint32_t capacity); +void desc_ring_free(desc_ring *r); + +static inline int desc_ring_push(desc_ring *r, engine_block v) { + uint_fast32_t t = atomic_load_explicit(&r->tail, memory_order_relaxed); + uint_fast32_t h = atomic_load_explicit(&r->head, memory_order_acquire); + if ((uint32_t)(t - h) > r->mask) return 0; /* full */ + r->buf[t & r->mask] = v; + atomic_store_explicit(&r->tail, t + 1, memory_order_release); + return 1; +} + +static inline int desc_ring_pop(desc_ring *r, engine_block *out) { + uint_fast32_t h = atomic_load_explicit(&r->head, memory_order_relaxed); + uint_fast32_t t = atomic_load_explicit(&r->tail, memory_order_acquire); + if (h == t) return 0; /* empty */ + *out = r->buf[h & r->mask]; + atomic_store_explicit(&r->head, h + 1, memory_order_release); + return 1; +} + +static inline uint32_t desc_ring_count(const desc_ring *r) { + uint_fast32_t t = atomic_load_explicit(&r->tail, memory_order_acquire); + uint_fast32_t h = atomic_load_explicit(&r->head, memory_order_acquire); + return (uint32_t)(t - h); +} + +/* Smallest power of two >= n (>= 1). */ +static inline uint32_t peer_next_pow2(uint32_t n) { + if (n < 2) return 1; + n--; + n |= n >> 1; n |= n >> 2; n |= n >> 4; + n |= n >> 8; n |= n >> 16; + return n + 1; +} + +#endif /* TORRENT_PEER_RING_H */ diff --git a/src/scheduler.c b/src/scheduler.c new file mode 100644 index 0000000..c98b345 --- /dev/null +++ b/src/scheduler.c @@ -0,0 +1,232 @@ +/* + * scheduler.c - Per-connection request pipeline, priority-driven piece + * selection, adaptive depth, and request-timeout handling. + * + * Selection: among pieces this peer has (conn->have_bits) and that the torrent + * has not already claimed (tor->requested, shared by every connection on the + * loop -> free cross-peer dedup), pick the highest harness-assigned priority, + * ties toward the lowest index. Priority 0 = skip. Re-evaluated at every piece + * boundary so the harness can drive any scheme (sequential, rarest-first, + * deadline ramps) live. + * + * Depth (P1d): the per-connection in-flight target tracks the bandwidth-delay + * product, target ~= BDP_HEADROOM * rate * min_rtt / block_size, clamped to + * [MIN, max_pipeline]. The *minimum* observed RTT is used, not an average, so + * the estimate reflects the unloaded path instead of feeding back our own + * queueing delay. + * + * Flow control: the credit invariant is loop-wide -- loop->outstanding (the sum + * of all connections' in-flight requests on this loop) must stay <= the number + * of free arena slots, so every arriving block is guaranteed a slot from the + * shared free_ring. + */ +#include "engine_internal.h" + +#include + +#define REQUEST_BYTES 17u /* len(4)+id(1)+index(4)+begin(4)+length(4) */ +#define MIN_PIPELINE 32u /* floor: covers our own refill cadence */ +#define BOOTSTRAP_PIPELINE 64u /* depth before we have rate/RTT samples */ +#define BDP_HEADROOM 2.0 /* keep the pipe a bit over the BDP estimate */ + +static uint32_t piece_blocks(const torrent *t, uint32_t piece) { + uint64_t plen = torrent_piece_len(t, piece); + return (uint32_t)((plen + PEER_BLOCK_SIZE - 1) / PEER_BLOCK_SIZE); +} + +static uint32_t block_len_of(const torrent *t, uint32_t piece, uint32_t block) { + uint64_t plen = torrent_piece_len(t, piece); + uint64_t off = (uint64_t)block * PEER_BLOCK_SIZE; + uint64_t rem = plen - off; + return rem < PEER_BLOCK_SIZE ? (uint32_t)rem : PEER_BLOCK_SIZE; +} + +static int block_received(const torrent *t, uint32_t piece, uint32_t block) { + const uint8_t *bm = t->recv_bits[piece]; + return bm && (bm[block >> 3] & (uint8_t)(1u << (block & 7))); +} + +void torrent_mark_received(torrent *t, uint32_t piece, uint32_t begin) { + if (piece >= t->num_pieces || begin % PEER_BLOCK_SIZE != 0) return; + uint32_t block = begin / PEER_BLOCK_SIZE; + if (block >= t->bpp) return; + if (!t->recv_bits[piece]) { + t->recv_bits[piece] = calloc((t->bpp + 7) / 8, 1); + if (!t->recv_bits[piece]) return; /* endgame just stays less precise */ + } + t->recv_bits[piece][block >> 3] |= (uint8_t)(1u << (block & 7)); +} + +void torrent_reset_received(torrent *t, uint32_t piece) { + if (piece >= t->num_pieces) return; + free(t->recv_bits[piece]); + t->recv_bits[piece] = NULL; +} + +/* Pick the highest-priority unclaimed piece this peer can request. Also reports, + * via *endgame, whether *any* wanted piece anywhere is still unclaimed — when + * none are, an otherwise-idle peer is allowed to race blocks of claimed pieces + * (BitTorrent endgame), which is what frees a tail piece held by a stalled peer. */ +static long select_next_piece(conn *c, int allowed_only, int *endgame) { + torrent *t = c->tor; + long best = -1; + int best_pri = 0; + int any_unclaimed = 0; + for (uint32_t i = 0; i < t->num_pieces; i++) { + int pri = t->priority[i]; + if (pri == 0) continue; /* complete / not wanted */ + if (!t->requested[i]) any_unclaimed = 1; + if (pri <= best_pri) continue; + if (t->requested[i]) continue; + if (!have_bit(c->have_bits, i)) continue; + if (allowed_only && !have_bit(c->allowed_fast_bits, i)) continue; + best = i; + best_pri = pri; + } + if (best >= 0) t->requested[best] = 1; + t->endgame = !any_unclaimed; + if (endgame) *endgame = t->endgame; + return best; +} + +/* Endgame fallback: find a still-needed block of any wanted piece this peer has + * that this connection has not already requested. Skips blocks already received + * (so we don't re-download a near-complete piece) and blocks in this peer's own + * in-flight set. Duplicate copies across peers are reaped by CANCEL on delivery. */ +static int select_endgame_block(conn *c, int allowed_only, uint32_t *piece_out, + uint32_t *begin_out, uint32_t *len_out) { + torrent *t = c->tor; + for (uint32_t i = 0; i < t->num_pieces; i++) { + if (t->priority[i] == 0) continue; + if (!have_bit(c->have_bits, i)) continue; + if (allowed_only && !have_bit(c->allowed_fast_bits, i)) continue; + uint32_t nb = piece_blocks(t, i); + for (uint32_t b = 0; b < nb; b++) { + if (block_received(t, i, b)) continue; + uint32_t begin = b * PEER_BLOCK_SIZE; + if (reqtab_has(&c->inflight, block_key(t, i, begin))) continue; + *piece_out = i; + *begin_out = begin; + *len_out = block_len_of(t, i, b); + return 1; + } + } + return 0; +} + +static uint32_t compute_target(conn *c) { + uint32_t cap = c->lp->eng->cfg.max_pipeline; + uint32_t target; + if (c->rtt_min_ns == 0 || c->rate_bps <= 0.0) { + target = BOOTSTRAP_PIPELINE; + } else { + double bdp = c->rate_bps * ((double)c->rtt_min_ns / 1e9) + / (double)PEER_BLOCK_SIZE; + double t = bdp * BDP_HEADROOM + 1.0; + target = (t < MIN_PIPELINE) ? MIN_PIPELINE : (uint32_t)t; + } + return target > cap ? cap : target; +} + +static void issue_block(conn *c, uint32_t piece, uint32_t begin, uint32_t len) { + proto_queue_request(c, piece, begin, len); + if (reqtab_insert(&c->inflight, block_key(c->tor, piece, begin), piece, + begin, len, peer_now_ns())) { + atomic_fetch_add_explicit(&c->outstanding, 1, memory_order_relaxed); + c->lp->outstanding++; + } +} + +void scheduler_tick(conn *c) { + if (c->dead) return; + loop *lp = c->lp; + int allowed_only = !c->unchoked; + + uint32_t target = compute_target(c); + c->pipeline_target = target; + + for (;;) { + uint32_t out = atomic_load_explicit(&c->outstanding, memory_order_relaxed); + if (out >= target) break; + + /* Loop-wide credit: never have more in flight than free slots. */ + uint32_t free_cnt = slot_ring_count(&lp->free_ring); + if (lp->outstanding >= free_cnt) { + /* Credit-limited: arm a wake so a returned slot refills at once. */ + atomic_store_explicit(&lp->want_release_wake, 1, memory_order_relaxed); + break; + } + if (outbuf_space(&c->out) < REQUEST_BYTES) break; + + /* Re-issue timed-out blocks before requesting fresh ones. */ + if (c->requeue_count > 0) { + block_req r = c->requeue[c->requeue_count - 1]; + /* Download throttle: defer if the bucket can't cover this block. */ + if (!rate_try_consume(&lp->eng->dl_limit, r.len)) break; + c->requeue_count--; + issue_block(c, r.piece, r.begin, r.len); + continue; + } + + if (!c->have_cur_piece) { + int endgame = 0; + long pick = select_next_piece(c, allowed_only, &endgame); + if (pick < 0) { + /* No unclaimed piece to start. In endgame, race a still-needed + * block of an already-claimed piece instead of going idle. */ + uint32_t ep, eb, el; + if (!endgame || + !select_endgame_block(c, allowed_only, &ep, &eb, &el)) + break; + if (!rate_try_consume(&lp->eng->dl_limit, el)) break; + issue_block(c, ep, eb, el); + continue; + } + c->cur_piece = (uint32_t)pick; + c->cur_piece_len = torrent_piece_len(c->tor, c->cur_piece); + c->cur_block_off = 0; + c->have_cur_piece = (c->cur_piece_len > 0); + if (!c->have_cur_piece) continue; + } + + uint32_t begin = (uint32_t)c->cur_block_off; + uint64_t rem = c->cur_piece_len - c->cur_block_off; + uint32_t len = (rem < PEER_BLOCK_SIZE) ? (uint32_t)rem : PEER_BLOCK_SIZE; + /* Download throttle: defer (keep piece/offset) if out of credit. */ + if (!rate_try_consume(&lp->eng->dl_limit, len)) break; + issue_block(c, c->cur_piece, begin, len); + c->cur_block_off += len; + if (c->cur_block_off >= c->cur_piece_len) c->have_cur_piece = 0; + } +} + +void scheduler_check_timeouts(conn *c) { + if (c->dead) return; + reqtab *t = &c->inflight; + if (t->count == 0) return; + + uint64_t now = peer_now_ns(); + uint64_t timeout = c->request_timeout_ns; + uint32_t cap = t->mask + 1; + uint32_t start = c->requeue_count; + + /* Phase 1: capture expired requests into the re-queue (table untouched). */ + for (uint32_t i = 0; i < cap && c->requeue_count < c->requeue_cap; i++) { + if (t->slots[i].key == REQ_EMPTY) continue; + if (now - t->slots[i].issue_ns < timeout) continue; + c->requeue[c->requeue_count].piece = t->slots[i].piece; + c->requeue[c->requeue_count].begin = t->slots[i].begin; + c->requeue[c->requeue_count].len = t->slots[i].len; + c->requeue_count++; + } + + /* Phase 2: remove the captured ones and drop their outstanding credit. */ + for (uint32_t r = start; r < c->requeue_count; r++) { + uint64_t key = block_key(c->tor, c->requeue[r].piece, c->requeue[r].begin); + req_entry e; + if (reqtab_take(t, key, &e)) { + atomic_fetch_sub_explicit(&c->outstanding, 1, memory_order_relaxed); + c->lp->outstanding--; + } + } +} diff --git a/src/transport.c b/src/transport.c new file mode 100644 index 0000000..84274a7 --- /dev/null +++ b/src/transport.c @@ -0,0 +1,87 @@ +/* transport.c - TCP implementation of the transport vtable. */ +#include "transport.h" + +#include +#include +#include +#include + +#include +#include +#include +#include +#include + +static ssize_t tcp_recv(transport *t, void *buf, size_t n) { + return recv(t->fd, buf, n, 0); +} + +static ssize_t tcp_send(transport *t, const void *buf, size_t n) { + return send(t->fd, buf, n, MSG_NOSIGNAL); +} + +static void tcp_close(transport *t) { + if (t->fd >= 0) { + close(t->fd); + t->fd = -1; + } +} + +static int tcp_handshake(transport *t, uint32_t *want_events) { + int r = transport_tcp_check_connected(t); + if (r == 0 && want_events) *want_events = EPOLLOUT; + return r; +} + +int transport_tcp_connect(transport *t, const char *ip, uint16_t port, + uint32_t recv_buffer_bytes, int *connecting) { + memset(t, 0, sizeof *t); + t->fd = -1; + + /* IPv6 literal if it contains a colon; otherwise IPv4. */ + int family = strchr(ip, ':') ? AF_INET6 : AF_INET; + int fd = socket(family, SOCK_STREAM | SOCK_NONBLOCK, 0); + if (fd < 0) return -1; + + int one = 1; + setsockopt(fd, IPPROTO_TCP, TCP_NODELAY, &one, sizeof one); + if (recv_buffer_bytes > 0) { + int rcv = (int)recv_buffer_bytes; + setsockopt(fd, SOL_SOCKET, SO_RCVBUF, &rcv, sizeof rcv); + } + + struct sockaddr_storage ss; + socklen_t slen; + memset(&ss, 0, sizeof ss); + if (family == AF_INET) { + struct sockaddr_in *sa = (struct sockaddr_in *)&ss; + sa->sin_family = AF_INET; + sa->sin_port = htons(port); + if (inet_pton(AF_INET, ip, &sa->sin_addr) != 1) { close(fd); return -1; } + slen = sizeof *sa; + } else { + struct sockaddr_in6 *sa = (struct sockaddr_in6 *)&ss; + sa->sin6_family = AF_INET6; + sa->sin6_port = htons(port); + if (inet_pton(AF_INET6, ip, &sa->sin6_addr) != 1) { close(fd); return -1; } + slen = sizeof *sa; + } + + int cr = connect(fd, (struct sockaddr *)&ss, slen); + if (cr != 0 && errno != EINPROGRESS) { close(fd); return -1; } + + t->fd = fd; + t->recv = tcp_recv; + t->send = tcp_send; + t->close = tcp_close; + t->handshake = tcp_handshake; + *connecting = (cr != 0); /* EINPROGRESS */ + return 0; +} + +int transport_tcp_check_connected(transport *t) { + int err = 0; + socklen_t l = sizeof err; + if (getsockopt(t->fd, SOL_SOCKET, SO_ERROR, &err, &l) != 0) return -1; + return err == 0 ? 1 : -1; +} diff --git a/src/transport.h b/src/transport.h new file mode 100644 index 0000000..9e31087 --- /dev/null +++ b/src/transport.h @@ -0,0 +1,73 @@ +/* + * transport.h - Byte-stream transport abstraction. + * + * The loop drives recv/send through this vtable, so alternative transports + * (io_uring, MSE encryption wrapping an inner transport, µTP over UDP) can slot + * in without touching the protocol parser. Only the TCP transport exists today. + * + * recv/send semantics match nonblocking sockets: + * recv: >0 bytes, 0 = peer closed, -1 = error (errno EAGAIN means try later) + * send: >=0 bytes accepted, -1 = error (errno EAGAIN means buffer full) + * + * A transport may need a multi-step async handshake before its byte stream is + * usable (TCP connect completion; the MSE crypto exchange). While `*connecting` + * is set by the connect call, the loop drives handshake() on read/write + * readiness until it reports the stream is ready; only then does BitTorrent + * protocol traffic flow through recv/send. + */ +#ifndef TORRENT_TRANSPORT_H +#define TORRENT_TRANSPORT_H + +#include +#include +#include + +typedef struct transport { + ssize_t (*recv)(struct transport *t, void *buf, size_t n); + ssize_t (*send)(struct transport *t, const void *buf, size_t n); + void (*close)(struct transport *t); + /* Advance the handshake. Returns 1 ready, 0 in progress, -1 failed. + * On 0, *want_events is set to the epoll interest the handshake needs + * next (EPOLLIN and/or EPOLLOUT). NULL if the transport needs no + * handshake beyond connect completion. */ + int (*handshake)(struct transport *t, uint32_t *want_events); + /* Optional: called each loop tick to service time-based work (µTP + * retransmission, delayed ACKs). NULL if the transport needs no timers. */ + void (*pump)(struct transport *t); + int fd; /* pollable fd for the loop's epoll set (-1 if none) */ + void *ctx; /* transport-private state */ +} transport; + +/* TCP transport: nonblocking connect to ip:port. ip may be IPv4 or IPv6 + * literal. Returns 0 and fills *t on success (connection may still be in + * progress); negative on immediate failure. `connecting` is set to 1 while the + * TCP handshake is still completing. */ +int transport_tcp_connect(transport *t, const char *ip, uint16_t port, + uint32_t recv_buffer_bytes, int *connecting); + +/* For a TCP transport in the connecting state, check whether connect() + * finished. Returns 1 connected, 0 still connecting, -1 failed. */ +int transport_tcp_check_connected(transport *t); + +/* MSE (Message Stream Encryption) transport over an inner TCP connection. + * Performs the initiator-side PE handshake keyed by the torrent info_hash, + * offering RC4 + plaintext (require_rc4 forces RC4-only). Same return contract + * as transport_tcp_connect; the encryption handshake runs during handshake(). */ +int transport_mse_connect(transport *t, const char *ip, uint16_t port, + uint32_t recv_buffer_bytes, + const uint8_t info_hash[20], int require_rc4, + int *connecting); + +/* Wrap an already-opened inner transport (TCP or µTP) in MSE. Takes ownership + * of *inner. Same return contract as the connect helpers. */ +int transport_mse_wrap(transport *t, const transport *inner, + const uint8_t info_hash[20], int require_rc4, + int *connecting); + +/* µTP (BEP-29) transport over UDP. Establishes the µTP connection during + * handshake() and provides a reliable, ordered byte stream via recv/send. + * Same return contract as transport_tcp_connect. */ +int transport_utp_connect(transport *t, const char *ip, uint16_t port, + uint32_t recv_buffer_bytes, int *connecting); + +#endif /* TORRENT_TRANSPORT_H */ diff --git a/src/transport_mse.c b/src/transport_mse.c new file mode 100644 index 0000000..fd6b5db --- /dev/null +++ b/src/transport_mse.c @@ -0,0 +1,393 @@ +/* + * transport_mse.c - MSE / PE (Message Stream Encryption) transport, initiator + * side, wrapping an inner TCP transport. + * + * Handshake (BEP-8 / Azureus MSE), A = us (initiator), B = peer: + * 1. A -> B : Ya = g^Xa mod P (96B) || PadA + * 2. B -> A : Yb (96B) || PadB + * 3. A computes S = Yb^Xa mod P, then SKEY = info_hash + * 4. A -> B : HASH('req1',S) || (HASH('req2',SKEY) xor HASH('req3',S)) || + * ENCRYPT(VC || crypto_provide || len(PadC) || PadC || + * len(IA) || IA) [keyA = HASH('keyA',S,SKEY)] + * 5. B -> A : ENCRYPT(VC || crypto_select || len(PadD) || PadD) + * followed by the (encrypted, if RC4 selected) payload + * [keyB = HASH('keyB',S,SKEY)] + * Both RC4 keystreams discard their first 1024 bytes. We send empty PadA/PadC + * and empty IA (the BitTorrent handshake flows afterwards over the now-ready + * stream), and we synchronise on B's encrypted VC to skip its PadB. + */ +#include "transport.h" +#include "crypto.h" + +#include +#include +#include + +#include + +#define VC_LEN 8 +#define IN_CAP 1024 +#define PAD_CAP 1024 +#define SEND_CHUNK 16384 + +enum { + MSE_CONNECT = 0, /* inner TCP still connecting */ + MSE_SEND_YA, /* sending our public key */ + MSE_RECV_YB, /* reading peer public key (96B) */ + MSE_SEND_REQ, /* sending req1/req2/req3 + enc payload */ + MSE_SYNC_VC, /* scanning for the encrypted VC */ + MSE_RECV_SELECT, /* reading crypto_select + len(PadD) */ + MSE_RECV_PADD, /* discarding PadD */ + MSE_READY +}; + +typedef struct { + transport inner; + uint8_t info_hash[20]; + int require_rc4; + + int hs_state; + uint8_t priv[20], pub[MSE_DH_LEN], secret[MSE_DH_LEN]; + rc4_ctx send_rc4, recv_rc4; + int rc4_active; + uint8_t vc_cipher[VC_LEN]; + + uint8_t out[MSE_DH_LEN + 64]; /* Ya, then the req message */ + size_t out_len, out_off; + + uint8_t yb[MSE_DH_LEN]; + size_t yb_got; + + uint8_t in[IN_CAP]; /* raw bytes buffered during sync */ + size_t in_len, in_off; + int synced; + + uint8_t selbuf[6]; /* crypto_select(4) + len(PadD)(2) */ + size_t sel_got; + uint32_t pad_len; + size_t pad_got; + uint8_t padbuf[PAD_CAP]; + + uint8_t leftover[IN_CAP]; /* decrypted app bytes past PadD */ + size_t leftover_len, leftover_off; +} mse_ctx; + +static inline uint32_t be32(const uint8_t *p) { + return ((uint32_t)p[0] << 24) | ((uint32_t)p[1] << 16) | + ((uint32_t)p[2] << 8) | (uint32_t)p[3]; +} +static inline uint16_t be16(const uint8_t *p) { + return (uint16_t)(((uint16_t)p[0] << 8) | p[1]); +} +static inline void put_be32(uint8_t *p, uint32_t v) { + p[0]=(uint8_t)(v>>24); p[1]=(uint8_t)(v>>16); p[2]=(uint8_t)(v>>8); p[3]=(uint8_t)v; +} + +/* Derive the RC4 keys from the shared secret and SKEY, drop 1024 bytes each, + * and precompute the expected encrypted VC (keyB applied to 8 zero bytes), + * which leaves recv_rc4 positioned right after the VC. */ +static void derive_keys(mse_ctx *m) { + uint8_t key[20]; + sha1_concat(key, "keyA", 4, m->secret, MSE_DH_LEN, m->info_hash, 20); + rc4_init(&m->send_rc4, key, 20); + rc4_skip(&m->send_rc4, 1024); + + sha1_concat(key, "keyB", 4, m->secret, MSE_DH_LEN, m->info_hash, 20); + rc4_init(&m->recv_rc4, key, 20); + rc4_skip(&m->recv_rc4, 1024); + + uint8_t zeros[VC_LEN] = {0}; + rc4_process(&m->recv_rc4, zeros, m->vc_cipher, VC_LEN); +} + +static void build_req(mse_ctx *m) { + uint8_t *p = m->out; + /* HASH('req1', S) */ + sha1_concat(p, "req1", 4, m->secret, MSE_DH_LEN, NULL, 0); + p += 20; + /* HASH('req2', SKEY) xor HASH('req3', S) */ + uint8_t h2[20], h3[20]; + sha1_concat(h2, "req2", 4, m->info_hash, 20, NULL, 0); + sha1_concat(h3, "req3", 4, m->secret, MSE_DH_LEN, NULL, 0); + for (int i = 0; i < 20; i++) p[i] = h2[i] ^ h3[i]; + p += 20; + /* ENCRYPT(VC(8 zeros) || crypto_provide(4) || len(PadC)=0 || len(IA)=0) */ + uint8_t payload[16]; + memset(payload, 0, VC_LEN); + put_be32(payload + VC_LEN, m->require_rc4 ? 0x02u : 0x03u); /* RC4 [+plain] */ + payload[12] = payload[13] = 0; /* len(PadC) = 0 */ + payload[14] = payload[15] = 0; /* len(IA) = 0 */ + rc4_process(&m->send_rc4, payload, p, sizeof payload); + p += sizeof payload; + m->out_len = (size_t)(p - m->out); + m->out_off = 0; +} + +/* Flush ctx->out. 1 done, 0 would-block, -1 error. */ +static int flush_out(mse_ctx *m) { + while (m->out_off < m->out_len) { + ssize_t n = m->inner.send(&m->inner, m->out + m->out_off, + m->out_len - m->out_off); + if (n > 0) { m->out_off += (size_t)n; continue; } + if (n < 0 && errno == EINTR) continue; + if (n < 0 && (errno == EAGAIN || errno == EWOULDBLOCK)) return 0; + return -1; + } + return 1; +} + +/* Read exactly `need` raw bytes into dst. 1 done, 0 would-block, -1 error. */ +static int recv_raw(mse_ctx *m, uint8_t *dst, size_t need, size_t *got, + uint32_t *want) { + while (*got < need) { + ssize_t n = m->inner.recv(&m->inner, dst + *got, need - *got); + if (n > 0) { *got += (size_t)n; continue; } + if (n == 0) return -1; + if (errno == EINTR) continue; + if (errno == EAGAIN || errno == EWOULDBLOCK) { *want = EPOLLIN; return 0; } + return -1; + } + return 1; +} + +/* Fill dst with `need` decrypted bytes, draining the in-buffer (raw, decrypted + * as consumed) first, then the socket. 1 done, 0 would-block, -1 error. */ +static int take_dec(mse_ctx *m, uint8_t *dst, size_t need, size_t *got, + uint32_t *want) { + while (*got < need && m->in_off < m->in_len) { + size_t avail = m->in_len - m->in_off; + size_t take = need - *got; + if (take > avail) take = avail; + rc4_process(&m->recv_rc4, m->in + m->in_off, dst + *got, take); + m->in_off += take; + *got += take; + } + while (*got < need) { + ssize_t n = m->inner.recv(&m->inner, dst + *got, need - *got); + if (n > 0) { + rc4_process(&m->recv_rc4, dst + *got, dst + *got, (size_t)n); + *got += (size_t)n; + continue; + } + if (n == 0) return -1; + if (errno == EINTR) continue; + if (errno == EAGAIN || errno == EWOULDBLOCK) { *want = EPOLLIN; return 0; } + return -1; + } + return 1; +} + +static int mse_handshake(transport *t, uint32_t *want) { + mse_ctx *m = t->ctx; + for (;;) { + switch (m->hs_state) { + case MSE_CONNECT: { + uint32_t w = EPOLLOUT; + int r = m->inner.handshake(&m->inner, &w); + if (r < 0) return -1; + if (r == 0) { *want = w; return 0; } + memcpy(m->out, m->pub, MSE_DH_LEN); /* Ya, PadA length 0 */ + m->out_len = MSE_DH_LEN; + m->out_off = 0; + m->hs_state = MSE_SEND_YA; + break; + } + case MSE_SEND_YA: { + int r = flush_out(m); + if (r < 0) return -1; + if (r == 0) { *want = EPOLLOUT; return 0; } + m->yb_got = 0; + m->hs_state = MSE_RECV_YB; + break; + } + case MSE_RECV_YB: { + int r = recv_raw(m, m->yb, MSE_DH_LEN, &m->yb_got, want); + if (r <= 0) return r; + dh_shared(m->priv, m->yb, m->secret); + derive_keys(m); + build_req(m); + m->hs_state = MSE_SEND_REQ; + break; + } + case MSE_SEND_REQ: { + int r = flush_out(m); + if (r < 0) return -1; + if (r == 0) { *want = EPOLLOUT; return 0; } + m->in_len = m->in_off = 0; + m->synced = 0; + m->hs_state = MSE_SYNC_VC; + break; + } + case MSE_SYNC_VC: { + while (!m->synced) { + if (m->in_len >= sizeof m->in) return -1; /* VC never appeared */ + ssize_t n = m->inner.recv(&m->inner, m->in + m->in_len, + sizeof m->in - m->in_len); + if (n > 0) { + m->in_len += (size_t)n; + for (size_t k = 0; k + VC_LEN <= m->in_len; k++) { + if (memcmp(m->in + k, m->vc_cipher, VC_LEN) == 0) { + size_t rest = m->in_len - (k + VC_LEN); + memmove(m->in, m->in + k + VC_LEN, rest); + m->in_len = rest; + m->in_off = 0; + m->synced = 1; + break; + } + } + if (!m->synced && m->in_len > VC_LEN - 1) { + /* keep only the tail that could start a match */ + size_t keep = VC_LEN - 1; + memmove(m->in, m->in + m->in_len - keep, keep); + m->in_len = keep; + } + continue; + } + if (n == 0) return -1; + if (errno == EINTR) continue; + if (errno == EAGAIN || errno == EWOULDBLOCK) { *want = EPOLLIN; return 0; } + return -1; + } + m->sel_got = 0; + m->hs_state = MSE_RECV_SELECT; + break; + } + case MSE_RECV_SELECT: { + int r = take_dec(m, m->selbuf, sizeof m->selbuf, &m->sel_got, want); + if (r <= 0) return r; + uint32_t select = be32(m->selbuf); + m->pad_len = be16(m->selbuf + 4); + if (m->pad_len > PAD_CAP) return -1; + if (select == 0x02) m->rc4_active = 1; + else if (select == 0x01 && !m->require_rc4) m->rc4_active = 0; + else return -1; /* peer chose something we didn't offer */ + m->pad_got = 0; + m->hs_state = MSE_RECV_PADD; + break; + } + case MSE_RECV_PADD: { + if (m->pad_len > 0) { + int r = take_dec(m, m->padbuf, m->pad_len, &m->pad_got, want); + if (r <= 0) return r; + } + /* Any bytes still buffered are the start of B's payload. They were + * received raw; decrypt iff RC4 was selected, and hand them to the + * first recv() calls. */ + size_t rem = m->in_len - m->in_off; + if (rem > 0) { + if (m->rc4_active) + rc4_process(&m->recv_rc4, m->in + m->in_off, m->leftover, rem); + else + memcpy(m->leftover, m->in + m->in_off, rem); + m->leftover_len = rem; + } + m->hs_state = MSE_READY; + break; + } + case MSE_READY: + return 1; + } + } +} + +static ssize_t mse_recv(transport *t, void *buf, size_t n) { + mse_ctx *m = t->ctx; + if (m->leftover_off < m->leftover_len) { + size_t avail = m->leftover_len - m->leftover_off; + size_t take = n < avail ? n : avail; + memcpy(buf, m->leftover + m->leftover_off, take); + m->leftover_off += take; + return (ssize_t)take; + } + ssize_t r = m->inner.recv(&m->inner, buf, n); + if (r > 0 && m->rc4_active) rc4_process(&m->recv_rc4, buf, buf, (size_t)r); + return r; +} + +static ssize_t mse_send(transport *t, const void *buf, size_t n) { + mse_ctx *m = t->ctx; + if (!m->rc4_active) return m->inner.send(&m->inner, buf, n); + + const uint8_t *src = buf; + uint8_t tmp[SEND_CHUNK]; + size_t sent = 0; + while (sent < n) { + size_t chunk = n - sent; + if (chunk > sizeof tmp) chunk = sizeof tmp; + /* Snapshot so we can advance the keystream by exactly what the socket + * accepts -- a stateful cipher can't tolerate a partial write. */ + rc4_ctx save = m->send_rc4; + rc4_process(&m->send_rc4, src + sent, tmp, chunk); + ssize_t s = m->inner.send(&m->inner, tmp, chunk); + if (s > 0) { + if ((size_t)s < chunk) { + m->send_rc4 = save; + rc4_skip(&m->send_rc4, (size_t)s); + sent += (size_t)s; + break; /* socket full */ + } + sent += (size_t)s; + continue; + } + m->send_rc4 = save; /* nothing went out this chunk */ + if (sent > 0) break; + return s; /* -1 with errno (EAGAIN/etc.) set by inner */ + } + return (ssize_t)sent; +} + +static void mse_close(transport *t) { + mse_ctx *m = t->ctx; + if (!m) return; + if (m->inner.close) m->inner.close(&m->inner); + t->fd = -1; + t->ctx = NULL; + free(m); +} + +/* Forward timer servicing to the inner transport (e.g. µTP retransmission). */ +static void mse_pump(transport *t) { + mse_ctx *m = t->ctx; + if (m && m->inner.pump) m->inner.pump(&m->inner); +} + +int transport_mse_wrap(transport *t, const transport *inner, + const uint8_t info_hash[20], int require_rc4, + int *connecting) { + memset(t, 0, sizeof *t); + t->fd = -1; + + mse_ctx *m = calloc(1, sizeof *m); + if (!m) return -1; + m->inner = *inner; + memcpy(m->info_hash, info_hash, 20); + m->require_rc4 = require_rc4; + if (dh_generate(m->priv, m->pub) != 0) { + if (m->inner.close) m->inner.close(&m->inner); + free(m); + return -1; + } + m->hs_state = MSE_CONNECT; + + t->ctx = m; + t->fd = m->inner.fd; + t->recv = mse_recv; + t->send = mse_send; + t->close = mse_close; + t->handshake = mse_handshake; + t->pump = mse_pump; + *connecting = 1; /* always need the MSE handshake before the stream is ready */ + return 0; +} + +int transport_mse_connect(transport *t, const char *ip, uint16_t port, + uint32_t recv_buffer_bytes, + const uint8_t info_hash[20], int require_rc4, + int *connecting) { + transport inner; + int inner_connecting = 0; + if (transport_tcp_connect(&inner, ip, port, recv_buffer_bytes, + &inner_connecting) != 0) + return -1; + return transport_mse_wrap(t, &inner, info_hash, require_rc4, connecting); +} diff --git a/src/transport_utp.c b/src/transport_utp.c new file mode 100644 index 0000000..709112b --- /dev/null +++ b/src/transport_utp.c @@ -0,0 +1,427 @@ +/* + * transport_utp.c - µTP (Micro Transport Protocol, BEP-29) over UDP. + * + * Provides a reliable, in-order byte stream so the BitTorrent protocol parser + * runs unchanged on top of it. We are always the initiator (a leech dialing + * out): we SYN, then send a low volume of control bytes (handshake, interested, + * requests) and receive a high volume of piece data. + * + * Header (v1, 20 bytes, big-endian): + * [type<<4 | 1][ext][connection_id:2][timestamp_us:4][timestamp_diff_us:4] + * [wnd_size:4][seq_nr:2][ack_nr:2] then optional extensions, then payload. + * + * Design notes / scope: + * - Receive path (the hot one): cumulative ACK, an out-of-order reorder ring, + * and a generous advertised window so the sender (libtorrent) is not flow- + * limited. We ACK promptly on every batch of received packets. + * - Send path (low volume): each ST_DATA/ST_SYN is tracked and retransmitted + * on RTO; ACKs free them. Selective-ACK extensions are parsed enough to be + * skipped (cumulative ACK + RTO covers loss). + * - Congestion control is deliberately minimal (a fixed, large window): on the + * loopback/LAN paths these tests exercise there is no loss, and our outbound + * volume is tiny, so LEDBAT ramping would add risk without changing results. + */ +#include "transport.h" +#include "crypto.h" + +#include +#include +#include +#include +#include + +#include +#include +#include +#include + +/* packet types (high nibble of byte 0) */ +enum { ST_DATA = 0, ST_FIN = 1, ST_STATE = 2, ST_RESET = 3, ST_SYN = 4 }; + +#define UTP_VER 1 +#define UTP_HDR 20 +#define UTP_MSS 1400 /* payload bytes per data packet */ +#define IN_CAP (1u << 19) /* 512 KiB in-order receive buffer */ +#define REO_SLOTS 1024 /* out-of-order reorder ring */ +#define REO_MASK (REO_SLOTS - 1) +#define OUT_SLOTS 2048 /* in-flight (unacked) outbound ring */ +#define OUT_MASK (OUT_SLOTS - 1) +#define RTO_MIN_US 500000u /* 500 ms minimum retransmit timeout */ +#define SYN_RTO_US 1000000u + +typedef struct { uint8_t *data; uint16_t len; int present; } reo_slot; +typedef struct { + uint8_t *data; uint16_t len; uint8_t type; int present; uint32_t sent_us; +} out_slot; + +typedef enum { U_INIT = 0, U_SYN_SENT, U_CONNECTED, U_RESET } utp_state; + +typedef struct { + int fd; + utp_state state; + + uint16_t conn_id_recv, conn_id_send; + uint16_t seq_nr; /* next outgoing seq to assign */ + uint16_t ack_nr; /* last in-order seq received */ + uint16_t send_base; /* oldest unacked seq */ + uint32_t peer_wnd; + uint32_t reply_micro; /* now - peer_timestamp, echoed for their LEDBAT */ + uint32_t rtt_us, rto_us; + uint32_t syn_sent_us; + + int got_fin; + uint16_t fin_seq; + int reset; + + /* in-order delivered bytes awaiting recv() */ + uint8_t *in_buf; + uint32_t in_cap, in_off, in_len; + + reo_slot reo[REO_SLOTS]; + out_slot out[OUT_SLOTS]; + uint32_t inflight_bytes; + int need_ack; +} utp_ctx; + +/* ---- time / byte helpers -------------------------------------------- */ + +static uint32_t now_us(void) { + struct timespec ts; + clock_gettime(CLOCK_MONOTONIC, &ts); + return (uint32_t)((uint64_t)ts.tv_sec * 1000000ull + ts.tv_nsec / 1000); +} +static void put16(uint8_t *p, uint16_t v) { p[0]=(uint8_t)(v>>8); p[1]=(uint8_t)v; } +static void put32(uint8_t *p, uint32_t v) { + p[0]=(uint8_t)(v>>24); p[1]=(uint8_t)(v>>16); p[2]=(uint8_t)(v>>8); p[3]=(uint8_t)v; +} +static uint16_t get16(const uint8_t *p) { return (uint16_t)((p[0]<<8)|p[1]); } +static uint32_t get32(const uint8_t *p) { + return ((uint32_t)p[0]<<24)|((uint32_t)p[1]<<16)|((uint32_t)p[2]<<8)|p[3]; +} +/* 16-bit sequence comparisons (modular). */ +static int seq_gt(uint16_t a, uint16_t b) { return (int16_t)(a - b) > 0; } + +static uint32_t in_free(const utp_ctx *u) { return u->in_cap - u->in_len; } + +/* ---- packet emission ------------------------------------------------ */ + +static void send_pkt(utp_ctx *u, uint8_t type, uint16_t seq, + const uint8_t *payload, uint16_t plen) { + uint8_t buf[UTP_HDR + UTP_MSS]; + if (plen > UTP_MSS) plen = UTP_MSS; + buf[0] = (uint8_t)((type << 4) | UTP_VER); + buf[1] = 0; /* no extensions */ + put16(buf + 2, type == ST_SYN ? u->conn_id_recv : u->conn_id_send); + put32(buf + 4, now_us()); + put32(buf + 8, u->reply_micro); + put32(buf + 12, in_free(u)); + put16(buf + 16, seq); + put16(buf + 18, u->ack_nr); + if (plen) memcpy(buf + UTP_HDR, payload, plen); + ssize_t r = send(u->fd, buf, UTP_HDR + plen, 0); + (void)r; /* UDP: drops are recovered by retransmission */ + u->need_ack = 0; +} + +static void send_state(utp_ctx *u) { send_pkt(u, ST_STATE, u->seq_nr, NULL, 0); } + +/* Queue + transmit one data packet, tracking it for retransmission. */ +static void send_data(utp_ctx *u, const uint8_t *payload, uint16_t plen) { + uint16_t seq = u->seq_nr; + out_slot *s = &u->out[seq & OUT_MASK]; + s->data = malloc(plen ? plen : 1); + if (!s->data) return; + memcpy(s->data, payload, plen); + s->len = plen; + s->type = ST_DATA; + s->present = 1; + s->sent_us = now_us(); + u->inflight_bytes += plen; + u->seq_nr++; + send_pkt(u, ST_DATA, seq, payload, plen); +} + +/* ---- inbound reassembly --------------------------------------------- */ + +static int in_append(utp_ctx *u, const uint8_t *d, uint32_t len) { + if (u->in_len + len > u->in_cap) return 0; /* no room: drop, peer resends */ + if (u->in_off + u->in_len + len > u->in_cap) { + memmove(u->in_buf, u->in_buf + u->in_off, u->in_len); + u->in_off = 0; + } + memcpy(u->in_buf + u->in_off + u->in_len, d, len); + u->in_len += len; + return 1; +} + +static void deliver_data(utp_ctx *u, uint16_t seq, const uint8_t *payload, + uint16_t plen) { + uint16_t expected = (uint16_t)(u->ack_nr + 1); + if (seq == expected) { + if (!in_append(u, payload, plen)) return; /* keep ack_nr; peer resends */ + u->ack_nr = seq; + for (;;) { + uint16_t nx = (uint16_t)(u->ack_nr + 1); + reo_slot *r = &u->reo[nx & REO_MASK]; + if (!r->present) break; + if (!in_append(u, r->data, r->len)) break; + free(r->data); + r->present = 0; + u->ack_nr = nx; + } + } else if (seq_gt(seq, expected)) { + reo_slot *r = &u->reo[seq & REO_MASK]; + if (!r->present) { + r->data = malloc(plen ? plen : 1); + if (r->data) { memcpy(r->data, payload, plen); r->len = plen; r->present = 1; } + } + } /* else duplicate: ignore */ + u->need_ack = 1; +} + +/* Free outbound packets the peer has cumulatively acked. */ +static void process_ack(utp_ctx *u, uint16_t ack) { + while (u->send_base != u->seq_nr && !seq_gt(u->send_base, ack)) { + out_slot *s = &u->out[u->send_base & OUT_MASK]; + if (s->present) { + uint32_t rtt = now_us() - s->sent_us; + u->rtt_us = u->rtt_us ? (u->rtt_us * 7 + rtt) / 8 : rtt; + u->rto_us = u->rtt_us * 2; + if (u->rto_us < RTO_MIN_US) u->rto_us = RTO_MIN_US; + free(s->data); + s->data = NULL; + s->present = 0; + u->inflight_bytes -= s->len; + } + u->send_base++; + } +} + +/* Parse and act on one received datagram. */ +static void process_datagram(utp_ctx *u, const uint8_t *buf, size_t len) { + if (len < UTP_HDR) return; + uint8_t type = buf[0] >> 4; + if ((buf[0] & 0x0f) != UTP_VER) return; + + /* Skip any extension chain to find the payload. */ + size_t off = UTP_HDR; + uint8_t ext = buf[1]; + while (ext != 0 && off + 2 <= len) { + uint8_t next = buf[off]; + uint8_t elen = buf[off + 1]; + off += 2 + elen; + ext = next; + } + if (off > len) off = len; + + u->reply_micro = now_us() - get32(buf + 4); + u->peer_wnd = get32(buf + 12); + uint16_t seq = get16(buf + 16); + uint16_t ack = get16(buf + 18); + + process_ack(u, ack); + + switch (type) { + case ST_STATE: + if (u->state == U_SYN_SENT) { + u->ack_nr = (uint16_t)(seq - 1); + u->state = U_CONNECTED; + } + break; + case ST_DATA: + if (u->state == U_SYN_SENT) { /* data implies connected */ + u->ack_nr = (uint16_t)(seq - 1); + u->state = U_CONNECTED; + } + deliver_data(u, seq, buf + off, (uint16_t)(len - off)); + break; + case ST_FIN: + u->got_fin = 1; + u->fin_seq = seq; + u->need_ack = 1; + break; + case ST_RESET: + u->reset = 1; + u->state = U_RESET; + break; + default: + break; + } +} + +/* Drain all queued datagrams. Returns -1 on a hard socket error. */ +static int utp_drain(utp_ctx *u) { + uint8_t buf[UTP_HDR + UTP_MSS + 64]; + for (;;) { + ssize_t n = recv(u->fd, buf, sizeof buf, 0); + if (n > 0) { process_datagram(u, buf, (size_t)n); continue; } + if (n == 0) return 0; + if (errno == EINTR) continue; + if (errno == EAGAIN || errno == EWOULDBLOCK) return 0; + return -1; + } +} + +/* ---- transport vtable ----------------------------------------------- */ + +static int utp_handshake(transport *t, uint32_t *want) { + utp_ctx *u = t->ctx; + if (u->state == U_INIT) { + u->seq_nr = 1; + send_pkt(u, ST_SYN, u->seq_nr, NULL, 0); /* SYN consumes seq 1 */ + /* SYN is tracked so it retransmits if lost. */ + out_slot *s = &u->out[u->seq_nr & OUT_MASK]; + s->data = NULL; s->len = 0; s->type = ST_SYN; s->present = 1; + s->sent_us = now_us(); + u->seq_nr++; + u->state = U_SYN_SENT; + u->syn_sent_us = now_us(); + *want = EPOLLIN; + return 0; + } + if (utp_drain(u) < 0) return -1; + if (u->reset) return -1; + if (u->state == U_CONNECTED) { + /* SYN is acked once we're connected. */ + out_slot *s = &u->out[1 & OUT_MASK]; + if (s->present && s->type == ST_SYN) { s->present = 0; } + u->send_base = 2; + return 1; + } + *want = EPOLLIN; + return 0; +} + +static ssize_t utp_recv(transport *t, void *buf, size_t n) { + utp_ctx *u = t->ctx; + if (utp_drain(u) < 0) { errno = EIO; return -1; } + if (u->reset) { errno = ECONNRESET; return -1; } + if (u->need_ack) send_state(u); + + if (u->in_len == 0) { + /* All in-order data delivered and peer FINed with nothing pending. */ + if (u->got_fin && !seq_gt(u->fin_seq, (uint16_t)(u->ack_nr + 1))) + return 0; + errno = EAGAIN; + return -1; + } + size_t take = n < u->in_len ? n : u->in_len; + memcpy(buf, u->in_buf + u->in_off, take); + u->in_off += take; + u->in_len -= (uint32_t)take; + if (u->in_len == 0) u->in_off = 0; + return (ssize_t)take; +} + +static ssize_t utp_send(transport *t, const void *buf, size_t n) { + utp_ctx *u = t->ctx; + if (u->reset) { errno = ECONNRESET; return -1; } + const uint8_t *p = buf; + size_t sent = 0; + while (sent < n) { + /* Window: bounded by in-flight packet slots and the peer's wnd_size. */ + uint16_t inflight = (uint16_t)(u->seq_nr - u->send_base); + if (inflight >= OUT_SLOTS - 1) break; + if (u->peer_wnd && u->inflight_bytes >= u->peer_wnd && u->inflight_bytes) + break; + size_t chunk = n - sent; + if (chunk > UTP_MSS) chunk = UTP_MSS; + send_data(u, p + sent, (uint16_t)chunk); + sent += chunk; + } + if (sent == 0) { errno = EAGAIN; return -1; } + return (ssize_t)sent; +} + +static void utp_pump(transport *t) { + utp_ctx *u = t->ctx; + uint32_t now = now_us(); + + if (u->state == U_SYN_SENT) { + if (now - u->syn_sent_us >= SYN_RTO_US) { + send_pkt(u, ST_SYN, 1, NULL, 0); + u->syn_sent_us = now; + } + return; + } + /* Retransmit timed-out unacked data. */ + uint32_t rto = u->rto_us ? u->rto_us : RTO_MIN_US; + for (uint16_t seq = u->send_base; seq != u->seq_nr; seq++) { + out_slot *s = &u->out[seq & OUT_MASK]; + if (s->present && now - s->sent_us >= rto) { + send_pkt(u, s->type, seq, s->data, s->len); + s->sent_us = now; + } + } + if (u->need_ack) send_state(u); +} + +static void utp_close(transport *t) { + utp_ctx *u = t->ctx; + if (!u) return; + if (u->state == U_CONNECTED && !u->reset) + send_pkt(u, ST_FIN, u->seq_nr, NULL, 0); + if (u->fd >= 0) close(u->fd); + for (int i = 0; i < REO_SLOTS; i++) if (u->reo[i].present) free(u->reo[i].data); + for (int i = 0; i < OUT_SLOTS; i++) if (u->out[i].present) free(u->out[i].data); + free(u->in_buf); + t->fd = -1; + t->ctx = NULL; + free(u); +} + +int transport_utp_connect(transport *t, const char *ip, uint16_t port, + uint32_t recv_buffer_bytes, int *connecting) { + memset(t, 0, sizeof *t); + t->fd = -1; + + int family = strchr(ip, ':') ? AF_INET6 : AF_INET; + int fd = socket(family, SOCK_DGRAM | SOCK_NONBLOCK, 0); + if (fd < 0) return -1; + if (recv_buffer_bytes > 0) { + int rcv = (int)recv_buffer_bytes; + setsockopt(fd, SOL_SOCKET, SO_RCVBUF, &rcv, sizeof rcv); + } + + struct sockaddr_storage ss; + socklen_t slen; + memset(&ss, 0, sizeof ss); + if (family == AF_INET) { + struct sockaddr_in *sa = (struct sockaddr_in *)&ss; + sa->sin_family = AF_INET; + sa->sin_port = htons(port); + if (inet_pton(AF_INET, ip, &sa->sin_addr) != 1) { close(fd); return -1; } + slen = sizeof *sa; + } else { + struct sockaddr_in6 *sa = (struct sockaddr_in6 *)&ss; + sa->sin6_family = AF_INET6; + sa->sin6_port = htons(port); + if (inet_pton(AF_INET6, ip, &sa->sin6_addr) != 1) { close(fd); return -1; } + slen = sizeof *sa; + } + /* connect() a UDP socket: fixes the peer, filters source, enables send()/ + * recv() and a single pollable fd. */ + if (connect(fd, (struct sockaddr *)&ss, slen) != 0) { close(fd); return -1; } + + utp_ctx *u = calloc(1, sizeof *u); + if (!u) { close(fd); return -1; } + u->in_buf = malloc(IN_CAP); + if (!u->in_buf) { free(u); close(fd); return -1; } + u->in_cap = IN_CAP; + u->fd = fd; + u->state = U_INIT; + if (crypto_random(&u->conn_id_recv, sizeof u->conn_id_recv) != 0) + u->conn_id_recv = (uint16_t)now_us(); + u->conn_id_send = (uint16_t)(u->conn_id_recv + 1); + u->rto_us = RTO_MIN_US; + + t->ctx = u; + t->fd = fd; + t->recv = utp_recv; + t->send = utp_send; + t->close = utp_close; + t->handshake = utp_handshake; + t->pump = utp_pump; + *connecting = 1; /* µTP handshake (SYN/STATE) runs during handshake() */ + return 0; +} diff --git a/tests/test_encryption.py b/tests/test_encryption.py new file mode 100644 index 0000000..e2b2aac --- /dev/null +++ b/tests/test_encryption.py @@ -0,0 +1,125 @@ +""" +MSE / PE encryption tests against a libtorrent seed forced into encrypted-only +mode (in/out_enc_policy = forced, allowed_enc_level = rc4). A plaintext engine +cannot complete the handshake with such a seed, so a successful byte-for-byte +download proves the MSE transport works end to end. + + * test_encrypted_download - engine with encryption=1 (offer RC4+plaintext) + * test_require_rc4 - engine with encryption=2 (RC4 only) + +Run with: python tests/test_encryption.py +""" +from __future__ import annotations + +import hashlib +import os +import sys +import tempfile +import time + +ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) +sys.path.insert(0, os.path.join(ROOT, "harness")) + +import libtorrent as lt # noqa: E402 + +from harness import load_metadata # noqa: E402 +from engine_ffi import Engine, EngineConfig, STATE_ERROR, ERROR_NAMES # noqa: E402 +from test_localseed import ensure_built, make_torrent, pick_listen_port # noqa: E402 + +LIB = os.path.join(ROOT, "build", "libtorrentpeer.so") + + +def start_encrypted_seed(root: str, torrent_path: str): + """A seed that REQUIRES MSE/RC4 encryption (refuses plaintext).""" + port = pick_listen_port() + ses = lt.session({ + "listen_interfaces": f"127.0.0.1:{port}", + "enable_dht": False, "enable_lsd": False, + "enable_upnp": False, "enable_natpmp": False, + "in_enc_policy": int(lt.enc_policy.forced), + "out_enc_policy": int(lt.enc_policy.forced), + "allowed_enc_level": int(lt.enc_level.rc4), + "prefer_rc4": True, + }) + h = ses.add_torrent({"ti": lt.torrent_info(torrent_path), "save_path": root, + "flags": lt.torrent_flags.seed_mode}) + deadline = time.time() + 15 + while time.time() < deadline and not h.status().is_seeding: + time.sleep(0.1) + assert h.status().is_seeding, "seed did not become ready" + return ses, h, ses.listen_port() or port + + +def _download(meta, port, encryption, timeout=60.0): + with Engine(EngineConfig(loop_count=1, slots_per_loop=1024, + max_pipeline=512, encryption=encryption), + lib_path=LIB) as eng: + tid = eng.add_torrent(meta.info_hash, b"-PC0001-" + os.urandom(12), + meta.piece_length, meta.total_size, meta.num_pieces) + eng.set_priorities(tid, [1] * meta.num_pieces) + eng.add_peer(tid, "127.0.0.1", port) + + buffers = [bytearray(meta.piece_len(i)) for i in range(meta.num_pieces)] + received = [0] * meta.num_pieces + done = bytearray(meta.num_pieces) + done_count = 0 + deadline = time.time() + timeout + while done_count < meta.num_pieces: + st = eng.status(tid) + if st.state == STATE_ERROR: + raise RuntimeError(f"engine error: {ERROR_NAMES[st.error]}") + descs = eng.poll_ready() + if not descs: + eng.wait(200) + if time.time() > deadline: + raise TimeoutError(f"stalled at {done_count}/{meta.num_pieces} " + f"(connected={st.peers_connected} " + f"failed={st.peers_failed})") + continue + for x in descs: + buf = buffers[x.piece] + buf[x.begin:x.begin + x.len] = eng.block_data(x.loop, x.slot, x.len) + eng.release(x.loop, x.slot) + received[x.piece] += x.len + if not done[x.piece] and received[x.piece] >= meta.piece_len(x.piece): + if hashlib.sha1(bytes(buf)).digest() != meta.piece_hashes[x.piece]: + raise ValueError(f"piece {x.piece} hash mismatch") + done[x.piece] = 1 + done_count += 1 + eng.set_priority(tid, x.piece, 0) + return b"".join(bytes(b) for b in buffers) + + +def test_encrypted_download(): + ensure_built() + size = 8 * 1024 * 1024 + with tempfile.TemporaryDirectory() as root: + torrent, original, _ = make_torrent(root, size) + meta = load_metadata(torrent) + ses, h, port = start_encrypted_seed(root, torrent) + try: + got = _download(meta, port, encryption=1) + finally: + ses.remove_torrent(h) + assert got == original, "decrypted bytes differ from original" + print(f"encrypted (offer RC4+plain) OK: {size/1e6:.1f} MB over MSE") + + +def test_require_rc4(): + ensure_built() + size = 8 * 1024 * 1024 + with tempfile.TemporaryDirectory() as root: + torrent, original, _ = make_torrent(root, size) + meta = load_metadata(torrent) + ses, h, port = start_encrypted_seed(root, torrent) + try: + got = _download(meta, port, encryption=2) + finally: + ses.remove_torrent(h) + assert got == original, "decrypted bytes differ from original" + print(f"encrypted (require RC4) OK: {size/1e6:.1f} MB over MSE") + + +if __name__ == "__main__": + test_encrypted_download() + test_require_rc4() diff --git a/tests/test_endgame.py b/tests/test_endgame.py new file mode 100644 index 0000000..6ca7dae --- /dev/null +++ b/tests/test_endgame.py @@ -0,0 +1,220 @@ +""" +Regression test for engine-side endgame (scheduler.c). + +The bug: a piece is claimed loop-wide (tor->requested[i]=1) by one connection, +so no other peer will request it. If that peer stays alive but stops delivering +(slow/snubbing seed), the piece never completes and healthy seeds sit idle even +though they have the data -- the classic "a few pieces never finish" tail stall. + +This test reproduces it deterministically with a raw "stalling" seed that serves +every block EXCEPT the last block of each piece (so it keeps its claims but never +finishes them, and the connection stays alive so the claim is never released). A +second, healthy seed is added afterwards. Only engine endgame -- letting an idle +peer race blocks of an already-claimed piece -- can finish the download, so a +regression here turns into a TimeoutError from drive_engine(). + +Run with: python tests/test_endgame.py +""" +from __future__ import annotations + +import os +import socket +import sys +import tempfile +import threading +import time + +ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) +sys.path.insert(0, os.path.join(ROOT, "harness")) + +from harness import load_metadata # noqa: E402 +from engine_ffi import Engine, EngineConfig # noqa: E402 +from test_localseed import ensure_built, make_torrent # noqa: E402 +from test_engine import TorrentDrive, drive_engine, make_peer_id # noqa: E402 +from test_mockpeer import _recv_exact, _msg, _piece_msg # noqa: E402 + +LIB = os.path.join(ROOT, "build", "libtorrentpeer.so") +BLOCK = 16384 + + +class StallingSeed(threading.Thread): + """Seed that has every piece but never serves the LAST block of any piece. + + It keeps reading (and ignoring) the withheld requests, so the connection + stays healthy and its piece claims are never released -- exactly the state + that wedges the tail without endgame. + """ + + def __init__(self, data, meta): + super().__init__(daemon=True) + self.data = data + self.meta = meta + self.served = 0 + self.srv = socket.socket(socket.AF_INET, socket.SOCK_STREAM) + self.srv.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1) + self.srv.bind(("127.0.0.1", 0)) + self.srv.listen(1) + self.port = self.srv.getsockname()[1] + self._stop = False + + def stop(self): + self._stop = True + try: + self.srv.close() + except OSError: + pass + + def _last_begin(self, index): + plen = self.meta.piece_len(index) + nblocks = (plen + BLOCK - 1) // BLOCK + return (nblocks - 1) * BLOCK + + def run(self): + try: + conn, _ = self.srv.accept() + except OSError: + return + conn.setsockopt(socket.IPPROTO_TCP, socket.TCP_NODELAY, 1) + hs = _recv_exact(conn, 68) + if not hs: + return + info_hash = hs[28:48] + conn.sendall(bytes([19]) + b"BitTorrent protocol" + bytes(8) + + info_hash + os.urandom(20)) + nbytes = (self.meta.num_pieces + 7) // 8 + bf = bytearray(nbytes) + for i in range(self.meta.num_pieces): + bf[i >> 3] |= 0x80 >> (i & 7) + conn.sendall(_msg(5, bytes(bf))) # bitfield: has everything + conn.sendall(_msg(1)) # unchoke + + plen = self.meta.piece_length + import struct + while not self._stop: + hdr = _recv_exact(conn, 4) + if hdr is None: + break + ln = struct.unpack(">I", hdr)[0] + if ln == 0: + continue + mid = _recv_exact(conn, 1) + if mid is None: + break + payload = _recv_exact(conn, ln - 1) if ln > 1 else b"" + if payload is None: + break + if mid[0] != 6: # only act on REQUEST + continue + index, begin, length = struct.unpack(">III", payload) + if begin == self._last_begin(index): + continue # withhold the last block forever -> piece never finishes + off = index * plen + begin + conn.sendall(_piece_msg(index, begin, self.data[off:off + length])) + self.served += 1 + + +class HealthySeed(threading.Thread): + """Plain seed that serves every requested block.""" + + def __init__(self, data, meta): + super().__init__(daemon=True) + self.data = data + self.meta = meta + self.srv = socket.socket(socket.AF_INET, socket.SOCK_STREAM) + self.srv.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1) + self.srv.bind(("127.0.0.1", 0)) + self.srv.listen(1) + self.port = self.srv.getsockname()[1] + self._stop = False + + def stop(self): + self._stop = True + try: + self.srv.close() + except OSError: + pass + + def run(self): + try: + conn, _ = self.srv.accept() + except OSError: + return + conn.setsockopt(socket.IPPROTO_TCP, socket.TCP_NODELAY, 1) + hs = _recv_exact(conn, 68) + if not hs: + return + info_hash = hs[28:48] + conn.sendall(bytes([19]) + b"BitTorrent protocol" + bytes(8) + + info_hash + os.urandom(20)) + nbytes = (self.meta.num_pieces + 7) // 8 + bf = bytearray(nbytes) + for i in range(self.meta.num_pieces): + bf[i >> 3] |= 0x80 >> (i & 7) + conn.sendall(_msg(5, bytes(bf))) + conn.sendall(_msg(1)) + + plen = self.meta.piece_length + import struct + while not self._stop: + hdr = _recv_exact(conn, 4) + if hdr is None: + break + ln = struct.unpack(">I", hdr)[0] + if ln == 0: + continue + mid = _recv_exact(conn, 1) + if mid is None: + break + payload = _recv_exact(conn, ln - 1) if ln > 1 else b"" + if payload is None: + break + if mid[0] != 6: + continue + index, begin, length = struct.unpack(">III", payload) + off = index * plen + begin + conn.sendall(_piece_msg(index, begin, self.data[off:off + length])) + + +def test_endgame_rescues_stalled_claims(): + ensure_built() + with tempfile.TemporaryDirectory() as root: + torrent, original, _ = make_torrent(root, 2 * 1024 * 1024) # 8 pieces + meta = load_metadata(torrent) + stalling = StallingSeed(original, meta) + healthy = HealthySeed(original, meta) + stalling.start() + healthy.start() + try: + with Engine(EngineConfig(loop_count=1, slots_per_loop=1024, + max_pipeline=64, + request_timeout_ms=400), lib_path=LIB) as eng: + tid = eng.add_torrent(meta.info_hash, make_peer_id(), + meta.piece_length, meta.total_size, + meta.num_pieces) + eng.set_priorities(tid, [1] * meta.num_pieces) + + # Let the stalling seed connect and claim pieces first. + eng.add_peer(tid, "127.0.0.1", stalling.port) + deadline = time.time() + 10.0 + while time.time() < deadline: + if eng.status(tid).outstanding > 0: + break + eng.wait(20) + else: + raise TimeoutError("stalling seed never issued requests") + + # Now the only way to finish the pieces it claimed is endgame. + eng.add_peer(tid, "127.0.0.1", healthy.port) + drives = {tid: TorrentDrive(meta)} + drive_engine(eng, drives, timeout=30.0) + + got = b"".join(bytes(b) for b in drives[tid].buffers) + assert got == original, "endgame download bytes differ" + finally: + stalling.stop() + healthy.stop() + print("endgame OK: healthy seed finished pieces a stalled peer had claimed") + + +if __name__ == "__main__": + test_endgame_rescues_stalled_claims() diff --git a/tests/test_engine.py b/tests/test_engine.py new file mode 100644 index 0000000..864e1d4 --- /dev/null +++ b/tests/test_engine.py @@ -0,0 +1,206 @@ +""" +Engine-level tests for the multi-peer / multi-torrent download engine +(include/engine.h, harness/engine_ffi.py): + + * test_two_torrents - one engine, two torrents downloaded concurrently; + each is pinned to a loop and verified byte-for-byte. + * test_two_peers - one torrent fed by two independent seeds; both + connections share the torrent's "requested" state, so + work is split with no duplicate requests, and the file + still verifies. + +These exercise the keystone restructure (reactor/loop-pool + transport vtable) +directly, alongside the legacy single-peer path covered by test_localseed.py. + +Run with: python tests/test_engine.py +""" +from __future__ import annotations + +import hashlib +import os +import sys +import tempfile +import time + +ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) +sys.path.insert(0, os.path.join(ROOT, "harness")) + +import libtorrent as lt # noqa: E402 + +from harness import load_metadata # noqa: E402 +from engine_ffi import Engine, EngineConfig, STATE_ERROR, ERROR_NAMES # noqa: E402 +from test_localseed import ensure_built, make_torrent, start_full_seed # noqa: E402 + +LIB = os.path.join(ROOT, "build", "libtorrentpeer.so") + + +def make_peer_id() -> bytes: + return b"-PC0001-" + os.urandom(12) + + +class TorrentDrive: + """Per-torrent reassembly + verification state.""" + + def __init__(self, meta): + self.meta = meta + self.buffers = [bytearray(meta.piece_len(i)) for i in range(meta.num_pieces)] + self.seen = [set() for _ in range(meta.num_pieces)] + self.received = [0] * meta.num_pieces + self.done = bytearray(meta.num_pieces) + self.done_count = 0 + self.order = [] + + +def drive_engine(eng: Engine, drives: dict[int, TorrentDrive], timeout=60.0): + """Pump the engine until every torrent's pieces are verified.""" + want = {tid: d.meta.num_pieces for tid, d in drives.items()} + deadline = time.time() + timeout + while any(drives[tid].done_count < want[tid] for tid in drives): + for tid, d in drives.items(): + st = eng.status(tid) + if st.state == STATE_ERROR: + raise RuntimeError(f"torrent {tid} error: {ERROR_NAMES[st.error]}") + descs = eng.poll_ready() + if not descs: + eng.wait(100) + if time.time() > deadline: + raise TimeoutError( + {tid: drives[tid].done_count for tid in drives}) + continue + for x in descs: + d = drives[x.torrent] + meta = d.meta + buf = d.buffers[x.piece] + first_copy = x.begin not in d.seen[x.piece] + if first_copy: + buf[x.begin:x.begin + x.len] = eng.block_data(x.loop, x.slot, x.len) + d.seen[x.piece].add(x.begin) + d.received[x.piece] += x.len + eng.release(x.loop, x.slot) + if (not d.done[x.piece] + and d.received[x.piece] >= meta.piece_len(x.piece)): + if hashlib.sha1(bytes(buf)).digest() != meta.piece_hashes[x.piece]: + raise ValueError(f"torrent {x.torrent} piece {x.piece} mismatch") + d.done[x.piece] = 1 + d.done_count += 1 + d.order.append(x.piece) + eng.set_priority(x.torrent, x.piece, 0) + + +def test_two_torrents(): + ensure_built() + size = 6 * 1024 * 1024 + with tempfile.TemporaryDirectory() as root_a, \ + tempfile.TemporaryDirectory() as root_b: + ta, data_a, _ = make_torrent(root_a, size) + tb, data_b, _ = make_torrent(root_b, size) + meta_a, meta_b = load_metadata(ta), load_metadata(tb) + ses_a, h_a, port_a = start_full_seed(root_a, ta) + ses_b, h_b, port_b = start_full_seed(root_b, tb) + try: + with Engine(EngineConfig(loop_count=2, slots_per_loop=1024, + max_pipeline=512), lib_path=LIB) as eng: + pid = make_peer_id() + tid_a = eng.add_torrent(meta_a.info_hash, pid, meta_a.piece_length, + meta_a.total_size, meta_a.num_pieces) + tid_b = eng.add_torrent(meta_b.info_hash, pid, meta_b.piece_length, + meta_b.total_size, meta_b.num_pieces) + eng.set_priorities(tid_a, [1] * meta_a.num_pieces) + eng.set_priorities(tid_b, [1] * meta_b.num_pieces) + eng.add_peer(tid_a, "127.0.0.1", port_a) + eng.add_peer(tid_b, "127.0.0.1", port_b) + + drives = {tid_a: TorrentDrive(meta_a), + tid_b: TorrentDrive(meta_b)} + t0 = time.time() + drive_engine(eng, drives, timeout=60.0) + dt = time.time() - t0 + + got_a = b"".join(bytes(b) for b in drives[tid_a].buffers) + got_b = b"".join(bytes(b) for b in drives[tid_b].buffers) + assert got_a == data_a, "torrent A bytes differ" + assert got_b == data_b, "torrent B bytes differ" + finally: + ses_a.remove_torrent(h_a) + ses_b.remove_torrent(h_b) + print(f"two torrents OK: 2 x {size/1e6:.1f} MB in {dt:.2f}s") + + +def test_two_peers(): + ensure_built() + size = 8 * 1024 * 1024 + with tempfile.TemporaryDirectory() as root: + torrent, original, _ = make_torrent(root, size) + meta = load_metadata(torrent) + # Two independent seeds of the same content -> two peers for one torrent. + ses1, h1, port1 = start_full_seed(root, torrent) + ses2, h2, port2 = start_full_seed(root, torrent) + try: + with Engine(EngineConfig(loop_count=1, slots_per_loop=1024, + max_pipeline=256), lib_path=LIB) as eng: + tid = eng.add_torrent(meta.info_hash, make_peer_id(), + meta.piece_length, meta.total_size, + meta.num_pieces) + eng.set_priorities(tid, [1] * meta.num_pieces) + eng.add_peer(tid, "127.0.0.1", port1) + eng.add_peer(tid, "127.0.0.1", port2) + + drives = {tid: TorrentDrive(meta)} + drive_engine(eng, drives, timeout=60.0) + + got = b"".join(bytes(b) for b in drives[tid].buffers) + assert got == original, "two-peer download bytes differ" + st = eng.status(tid) + assert st.peers == 2, f"expected 2 peers, got {st.peers}" + finally: + ses1.remove_torrent(h1) + ses2.remove_torrent(h2) + print(f"two peers OK: {size/1e6:.1f} MB via 2 connections, no duplicate blocks") + + +def test_peer_failure_releases_claims(): + ensure_built() + size = 4 * 1024 * 1024 + with tempfile.TemporaryDirectory() as root: + torrent, original, _ = make_torrent(root, size) + meta = load_metadata(torrent) + ses1, h1, port1 = start_full_seed(root, torrent) + ses2, h2, port2 = start_full_seed(root, torrent) + removed1 = False + try: + with Engine(EngineConfig(loop_count=1, slots_per_loop=64, + max_pipeline=4), lib_path=LIB) as eng: + tid = eng.add_torrent(meta.info_hash, make_peer_id(), + meta.piece_length, meta.total_size, + meta.num_pieces) + eng.set_priorities(tid, [1] * meta.num_pieces) + eng.add_peer(tid, "127.0.0.1", port1) + + deadline = time.time() + 10.0 + while time.time() < deadline: + st = eng.status(tid) + if st.outstanding > 0: + break + eng.wait(20) + else: + raise TimeoutError("first peer never issued requests") + + ses1.remove_torrent(h1) + removed1 = True + eng.add_peer(tid, "127.0.0.1", port2) + + drives = {tid: TorrentDrive(meta)} + drive_engine(eng, drives, timeout=60.0) + got = b"".join(bytes(b) for b in drives[tid].buffers) + assert got == original, "download after peer failure differs" + finally: + if not removed1: + ses1.remove_torrent(h1) + ses2.remove_torrent(h2) + print("peer failure recovery OK: dead-peer claims were released") + + +if __name__ == "__main__": + test_two_torrents() + test_two_peers() + test_peer_failure_releases_claims() diff --git a/tests/test_localseed.py b/tests/test_localseed.py new file mode 100644 index 0000000..9f5f3a6 --- /dev/null +++ b/tests/test_localseed.py @@ -0,0 +1,211 @@ +""" +End-to-end tests against a local libtorrent seed: + + * test_localseed_roundtrip - full seed, download everything, verify bytes + * test_partial_availability - seed has only some pieces; the peer must + download exactly those and skip the rest + (the bug a fixed in-order schedule hit) + * test_priority_order - pieces are selected highest-priority-first + +Run with: python tests/test_localseed.py (or) python -m pytest tests/ +""" +from __future__ import annotations + +import hashlib +import os +import socket +import subprocess +import sys +import tempfile +import time + +ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) +sys.path.insert(0, os.path.join(ROOT, "harness")) + +import libtorrent as lt # noqa: E402 + +from harness import Downloader, load_metadata # noqa: E402 + +LIB = os.path.join(ROOT, "build", "libtorrentpeer.so") +PIECE_SIZE = 256 * 1024 + + +def pick_listen_port() -> int: + """Reserve a free loopback TCP port briefly, then hand it to libtorrent.""" + with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as s: + s.bind(("127.0.0.1", 0)) + return s.getsockname()[1] + + +def ensure_built(): + if os.path.exists(LIB): + return + build = os.path.join(ROOT, "build") + subprocess.run(["cmake", "-S", ROOT, "-B", build, + "-DCMAKE_BUILD_TYPE=Release"], check=True) + subprocess.run(["cmake", "--build", build], check=True) + assert os.path.exists(LIB), "build did not produce libtorrentpeer.so" + + +def make_torrent(root: str, size: int) -> tuple[str, bytes, str]: + data = os.urandom(size) + path = os.path.join(root, "data.bin") + with open(path, "wb") as f: + f.write(data) + fs = lt.file_storage() + lt.add_files(fs, path) + t = lt.create_torrent(fs, piece_size=PIECE_SIZE) + t.set_priv(False) + lt.set_piece_hashes(t, root) + torrent_path = os.path.join(root, "test.torrent") + with open(torrent_path, "wb") as f: + f.write(lt.bencode(t.generate())) + return torrent_path, data, path + + +def _session() -> tuple[lt.session, int]: + port = pick_listen_port() + return lt.session({ + "listen_interfaces": f"127.0.0.1:{port}", + "enable_dht": False, "enable_lsd": False, + "enable_upnp": False, "enable_natpmp": False, + # Force plaintext so our minimal peer's handshake is accepted. + "in_enc_policy": int(lt.enc_policy.disabled), + "out_enc_policy": int(lt.enc_policy.disabled), + }), port + + +def start_full_seed(root: str, torrent_path: str): + ses, port = _session() + h = ses.add_torrent({"ti": lt.torrent_info(torrent_path), "save_path": root, + "flags": lt.torrent_flags.seed_mode}) + deadline = time.time() + 15 + while time.time() < deadline and not h.status().is_seeding: + time.sleep(0.1) + assert h.status().is_seeding, "seed did not become ready" + return ses, h, ses.listen_port() or port + + +def start_partial_seed(root: str, torrent_path: str, data_path: str, size: int): + """Corrupt the second half on disk so the seeder only HAS the first half.""" + with open(data_path, "r+b") as f: + f.seek(size // 2) + f.write(b"\x00" * (size - size // 2)) + ses, port = _session() + h = ses.add_torrent({"ti": lt.torrent_info(torrent_path), "save_path": root}) + deadline = time.time() + 15 + while time.time() < deadline: + state = str(h.status().state) + if "checking" not in state and "queued" not in state: + break + time.sleep(0.1) + time.sleep(0.3) + avail = {i for i, b in enumerate(h.status().pieces) if b} + return ses, h, ses.listen_port() or port, avail + + +def drive(dl: Downloader, meta, port: int, prio: bytearray, want: set, + timeout: float = 30.0): + """Drive the peer until `want` pieces are verified. Returns completion order.""" + for i in range(meta.num_pieces): + dl.buffers[i] = bytearray(meta.piece_len(i)) + dl.peer.start("127.0.0.1", port) + dl.peer.set_priorities(prio) + order, got = [], set() + deadline = time.time() + timeout + while got != want and time.time() < deadline: + st = dl.peer.status() + if st.state == 6: # STATE_ERROR + raise RuntimeError(f"peer error {st.error}") + descs = dl.peer.poll_ready() + if not descs: + dl.peer.wait(100) + continue + for x in descs: + buf = dl.buffers[x.piece] + buf[x.begin:x.begin + x.len] = dl.peer.block_data(x.slot, x.len) + dl.peer.release(x.slot) + dl.received[x.piece] += x.len + if dl.received[x.piece] >= meta.piece_len(x.piece) and x.piece not in got: + assert hashlib.sha1(bytes(buf)).digest() == meta.piece_hashes[x.piece] + got.add(x.piece) + order.append(x.piece) + dl.peer.set_priority(x.piece, 0) + return order, got + + +def test_localseed_roundtrip(): + ensure_built() + size = 8 * 1024 * 1024 + with tempfile.TemporaryDirectory() as root: + torrent, original, _ = make_torrent(root, size) + meta = load_metadata(torrent) + ses, h, port = start_full_seed(root, torrent) + try: + dl = Downloader(meta, num_slots=1024, max_pipeline=512, lib_path=LIB) + try: + t0 = time.time() + got = dl.download("127.0.0.1", port, timeout=60.0) + dt = time.time() - t0 + finally: + dl.close() + finally: + ses.remove_torrent(h) + assert got == original, "downloaded bytes differ from original" + print(f"roundtrip OK: {size/1e6:.1f} MB in {dt:.2f}s " + f"({size/1e6/dt:.0f} MB/s)") + + +def test_partial_availability(): + ensure_built() + size = 8 * 1024 * 1024 + with tempfile.TemporaryDirectory() as root: + torrent, _, data_path = make_torrent(root, size) + meta = load_metadata(torrent) + ses, h, port, avail = start_partial_seed(root, torrent, data_path, size) + try: + dl = Downloader(meta, num_slots=512, max_pipeline=256, lib_path=LIB) + try: + prio = bytearray([1] * meta.num_pieces) # want everything + _, got = drive(dl, meta, port, prio, want=avail, timeout=30.0) + # peer must idle, not stall requesting missing pieces + time.sleep(0.3) + assert dl.peer.status().outstanding == 0 + finally: + dl.close() + finally: + ses.remove_torrent(h) + assert 0 < len(avail) < meta.num_pieces, "test needs a partial seed" + assert got == avail, (sorted(got), sorted(avail)) + print(f"partial OK: downloaded {len(got)}/{meta.num_pieces} available " + f"pieces, skipped {meta.num_pieces - len(avail)} missing") + + +def test_priority_order(): + ensure_built() + size = 2 * 1024 * 1024 # 8 pieces + with tempfile.TemporaryDirectory() as root: + torrent, _, _ = make_torrent(root, size) + meta = load_metadata(torrent) + ses, h, port = start_full_seed(root, torrent) + try: + # pipeline=1 makes selection order observable; piece i gets priority + # i+1, so we expect strictly descending completion order. + dl = Downloader(meta, num_slots=4, max_pipeline=1, lib_path=LIB) + try: + prio = bytearray([i + 1 for i in range(meta.num_pieces)]) + want = set(range(meta.num_pieces)) + order, _ = drive(dl, meta, port, prio, want=want, timeout=30.0) + finally: + dl.close() + finally: + ses.remove_torrent(h) + expected = list(range(meta.num_pieces - 1, -1, -1)) + assert order == expected, order + print(f"priority order OK: {order}") + + +if __name__ == "__main__": + test_localseed_roundtrip() + test_partial_availability() + test_priority_order() diff --git a/tests/test_mockpeer.py b/tests/test_mockpeer.py new file mode 100644 index 0000000..b8ce8df --- /dev/null +++ b/tests/test_mockpeer.py @@ -0,0 +1,362 @@ +""" +Deterministic tests for the P0 reliability fixes, using a tiny raw-socket mock +BitTorrent peer (no libtorrent quirks in the loop): + + * test_request_timeout_recovery - the mock silently drops the first request + for block (0,0); the peer must time out, re-request it, and still complete + (validates C1). + * test_unsolicited_block_ignored - the mock injects a duplicate/unsolicited + block; the peer must drop it without corrupting the download or its credit + accounting (validates C2). + +Run with: python tests/test_mockpeer.py (or) python -m pytest tests/ +""" +from __future__ import annotations + +import os +import socket +import struct +import subprocess +import sys +import threading +import time + +ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) +sys.path.insert(0, os.path.join(ROOT, "harness")) + +import libtorrent as lt # noqa: E402 + +from harness import Downloader, load_metadata # noqa: E402 + +LIB = os.path.join(ROOT, "build", "libtorrentpeer.so") +BLOCK = 16384 + + +def ensure_built(): + if os.path.exists(LIB): + return + build = os.path.join(ROOT, "build") + subprocess.run(["cmake", "-S", ROOT, "-B", build, + "-DCMAKE_BUILD_TYPE=Release"], check=True) + subprocess.run(["cmake", "--build", build], check=True) + + +def make_torrent(root: str, size: int, piece: int): + data = os.urandom(size) + path = os.path.join(root, "data.bin") + with open(path, "wb") as f: + f.write(data) + fs = lt.file_storage() + lt.add_files(fs, path) + t = lt.create_torrent(fs, piece_size=piece) + t.set_priv(False) + lt.set_piece_hashes(t, root) + tp = os.path.join(root, "m.torrent") + with open(tp, "wb") as f: + f.write(lt.bencode(t.generate())) + return tp, data + + +def _recv_exact(conn, n): + buf = b"" + while len(buf) < n: + try: + chunk = conn.recv(n - len(buf)) + except OSError: + return None + if not chunk: + return None + buf += chunk + return buf + + +def _msg(mid, payload=b""): + return struct.pack(">I", 1 + len(payload)) + bytes([mid]) + payload + + +def _ext_msg(ext_id, payload=b""): + return struct.pack(">I", 2 + len(payload)) + bytes([20, ext_id]) + payload + + +def _piece_msg(index, begin, data): + return (struct.pack(">I", 9 + len(data)) + bytes([7]) + + struct.pack(">II", index, begin) + data) + + +class MockPeer(threading.Thread): + """A seed that has every piece, with optional misbehavior for the test.""" + + def __init__(self, data, meta, *, drop_first=False, inject_unsolicited=False): + super().__init__(daemon=True) + self.data = data + self.meta = meta + self.drop_first = drop_first + self.inject_unsolicited = inject_unsolicited + self.srv = socket.socket(socket.AF_INET, socket.SOCK_STREAM) + self.srv.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1) + self.srv.bind(("127.0.0.1", 0)) + self.srv.listen(1) + self.port = self.srv.getsockname()[1] + self._stop = False + + def stop(self): + self._stop = True + try: + self.srv.close() + except OSError: + pass + + def run(self): + try: + conn, _ = self.srv.accept() + except OSError: + return + conn.setsockopt(socket.IPPROTO_TCP, socket.TCP_NODELAY, 1) + hs = _recv_exact(conn, 68) + if not hs: + return + info_hash = hs[28:48] + conn.sendall(bytes([19]) + b"BitTorrent protocol" + bytes(8) + + info_hash + os.urandom(20)) + + nbytes = (self.meta.num_pieces + 7) // 8 + bf = bytearray(nbytes) + for i in range(self.meta.num_pieces): + bf[i >> 3] |= 0x80 >> (i & 7) + conn.sendall(_msg(5, bytes(bf))) # bitfield: has everything + conn.sendall(_msg(1)) # unchoke + + plen = self.meta.piece_length + dropped = injected = False + served = 0 + while not self._stop: + hdr = _recv_exact(conn, 4) + if hdr is None: + break + ln = struct.unpack(">I", hdr)[0] + if ln == 0: + continue + mid = _recv_exact(conn, 1) + if mid is None: + break + payload = _recv_exact(conn, ln - 1) if ln > 1 else b"" + if payload is None: + break + if mid[0] != 6: # only act on requests + continue + index, begin, length = struct.unpack(">III", payload) + if self.drop_first and not dropped and (index, begin) == (0, 0): + dropped = True # silently drop -> force a client timeout+retry + continue + off = index * plen + begin + conn.sendall(_piece_msg(index, begin, self.data[off:off + length])) + served += 1 + if self.inject_unsolicited and not injected and served >= 4: + injected = True + # Duplicate of (0,0), already delivered -> now unsolicited. + conn.sendall(_piece_msg(0, 0, self.data[0:BLOCK])) + + +class FastMockPeer(MockPeer): + """Seed that uses BEP-6 Fast messages instead of a v1 bitfield.""" + + def __init__(self, data, meta, *, unchoke=True, allowed_fast=False): + super().__init__(data, meta) + self.unchoke = unchoke + self.allowed_fast = allowed_fast + + def run(self): + try: + conn, _ = self.srv.accept() + except OSError: + return + conn.setsockopt(socket.IPPROTO_TCP, socket.TCP_NODELAY, 1) + hs = _recv_exact(conn, 68) + if not hs: + return + info_hash = hs[28:48] + reserved = bytearray(8) + reserved[7] |= 0x04 + conn.sendall(bytes([19]) + b"BitTorrent protocol" + bytes(reserved) + + info_hash + os.urandom(20)) + + conn.sendall(_msg(14)) # HAVE_ALL + if self.allowed_fast: + for i in range(self.meta.num_pieces): + conn.sendall(_msg(17, struct.pack(">I", i))) + if self.unchoke: + conn.sendall(_msg(1)) + + plen = self.meta.piece_length + while not self._stop: + hdr = _recv_exact(conn, 4) + if hdr is None: + break + ln = struct.unpack(">I", hdr)[0] + if ln == 0: + continue + mid = _recv_exact(conn, 1) + if mid is None: + break + payload = _recv_exact(conn, ln - 1) if ln > 1 else b"" + if payload is None: + break + if mid[0] != 6: + continue + index, begin, length = struct.unpack(">III", payload) + off = index * plen + begin + conn.sendall(_piece_msg(index, begin, self.data[off:off + length])) + + +class DontHaveMockPeer(MockPeer): + """Peer that advertises all pieces, then revokes one through BEP-54.""" + + def __init__(self, data, meta, revoked_piece=0): + super().__init__(data, meta) + self.revoked_piece = revoked_piece + self.requests = [] + self.saw_ext_handshake = False + + def run(self): + try: + conn, _ = self.srv.accept() + except OSError: + return + conn.setsockopt(socket.IPPROTO_TCP, socket.TCP_NODELAY, 1) + hs = _recv_exact(conn, 68) + if not hs: + return + info_hash = hs[28:48] + reserved = bytearray(8) + reserved[5] |= 0x10 + reserved[7] |= 0x04 + conn.sendall(bytes([19]) + b"BitTorrent protocol" + bytes(reserved) + + info_hash + os.urandom(20)) + + conn.sendall(_msg(14)) # HAVE_ALL + conn.sendall(_ext_msg(0, b"d1:md11:lt_donthavei1eee")) + conn.sendall(_ext_msg(1, struct.pack(">I", self.revoked_piece))) + conn.sendall(_msg(1)) + + plen = self.meta.piece_length + while not self._stop: + hdr = _recv_exact(conn, 4) + if hdr is None: + break + ln = struct.unpack(">I", hdr)[0] + if ln == 0: + continue + mid = _recv_exact(conn, 1) + if mid is None: + break + payload = _recv_exact(conn, ln - 1) if ln > 1 else b"" + if payload is None: + break + if mid[0] == 20 and payload[:1] == b"\x00" and b"lt_donthave" in payload: + self.saw_ext_handshake = True + if mid[0] != 6: + continue + index, begin, length = struct.unpack(">III", payload) + self.requests.append((index, begin, length)) + if index == self.revoked_piece: + continue + off = index * plen + begin + conn.sendall(_piece_msg(index, begin, self.data[off:off + length])) + + +def _run(drop_first=False, inject_unsolicited=False): + ensure_built() + import tempfile + with tempfile.TemporaryDirectory() as root: + tp, data = make_torrent(root, 2 * 1024 * 1024, 256 * 1024) # 8 pieces + meta = load_metadata(tp) + mock = MockPeer(data, meta, drop_first=drop_first, + inject_unsolicited=inject_unsolicited) + mock.start() + dl = Downloader(meta, num_slots=256, max_pipeline=64, + request_timeout_ms=600, lib_path=LIB) + try: + got = dl.download("127.0.0.1", mock.port, timeout=15.0) + st = dl.peer.status() + finally: + dl.close() + mock.stop() + assert got == data, "download did not reconstruct the original bytes" + assert st.outstanding == 0, f"credit leak: outstanding={st.outstanding}" + return st + + +def _run_fast(*, unchoke=True, allowed_fast=False): + ensure_built() + import tempfile + with tempfile.TemporaryDirectory() as root: + tp, data = make_torrent(root, 512 * 1024, 256 * 1024) # 2 pieces + meta = load_metadata(tp) + mock = FastMockPeer(data, meta, unchoke=unchoke, + allowed_fast=allowed_fast) + mock.start() + dl = Downloader(meta, num_slots=64, max_pipeline=16, + request_timeout_ms=600, lib_path=LIB) + try: + got = dl.download("127.0.0.1", mock.port, timeout=15.0) + finally: + dl.close() + mock.stop() + assert got == data, "fast-extension download did not reconstruct bytes" + + +def _run_donthave(): + ensure_built() + import tempfile + with tempfile.TemporaryDirectory() as root: + tp, data = make_torrent(root, 512 * 1024, 256 * 1024) # 2 pieces + meta = load_metadata(tp) + mock = DontHaveMockPeer(data, meta, revoked_piece=0) + mock.start() + dl = Downloader(meta, num_slots=64, max_pipeline=16, + request_timeout_ms=600, lib_path=LIB) + try: + try: + dl.download("127.0.0.1", mock.port, pieces=[0], timeout=2.0, + progress_every=10.0) + raise AssertionError("revoked piece unexpectedly downloaded") + except TimeoutError: + pass + finally: + dl.close() + mock.stop() + assert mock.saw_ext_handshake, "client did not advertise lt_donthave" + assert not mock.requests, f"requested revoked piece: {mock.requests[:4]}" + + +def test_request_timeout_recovery(): + _run(drop_first=True) + print("C1 OK: recovered from a silently dropped request via timeout") + + +def test_unsolicited_block_ignored(): + _run(inject_unsolicited=True) + print("C2 OK: ignored an unsolicited block, download intact") + + +def test_fast_have_all(): + _run_fast(unchoke=True) + print("BEP-6 OK: HAVE_ALL populated peer availability") + + +def test_allowed_fast_while_choked(): + _run_fast(unchoke=False, allowed_fast=True) + print("BEP-6 OK: ALLOWED_FAST pieces downloaded while choked") + + +def test_ltep_donthave_receiver(): + _run_donthave() + print("BEP-10/54 OK: LT extension handshake + lt_donthave receiver") + + +if __name__ == "__main__": + test_request_timeout_recovery() + test_unsolicited_block_ignored() + test_fast_have_all() + test_allowed_fast_while_choked() + test_ltep_donthave_receiver() diff --git a/tests/test_swarm_endgame.py b/tests/test_swarm_endgame.py new file mode 100644 index 0000000..82b7f8c --- /dev/null +++ b/tests/test_swarm_endgame.py @@ -0,0 +1,89 @@ +""" +Regression tests for swarm_download's endgame helpers. + +Run with: python tests/test_swarm_endgame.py +""" +from __future__ import annotations + +import hashlib +import os +import sys + +ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) +sys.path.insert(0, os.path.join(ROOT, "harness")) + +from engine_ffi import BLOCK_SIZE # noqa: E402 +from swarm_download import EndgameController, PieceAssembler # noqa: E402 + + +class FakeMeta: + def __init__(self, payloads: list[bytes]): + self.payloads = payloads + self.num_pieces = len(payloads) + self.piece_length = max(len(p) for p in payloads) + self.total_size = sum(len(p) for p in payloads) + self.piece_hashes = [hashlib.sha1(p).digest() for p in payloads] + + def piece_len(self, piece: int) -> int: + return len(self.payloads[piece]) + + +class FakeEngine: + def __init__(self): + self.priorities = [] + self.rearms = [] + + def set_priority(self, tid: int, piece: int, priority: int) -> None: + self.priorities.append((tid, piece, priority)) + + def request_piece(self, tid: int, piece: int) -> None: + self.rearms.append((tid, piece)) + + +def test_piece_assembler_ignores_duplicate_blocks(): + payload = (b"a" * BLOCK_SIZE) + b"tail" + meta = FakeMeta([payload]) + done = bytearray(meta.num_pieces) + asm = PieceAssembler(meta, done) + + first = payload[:BLOCK_SIZE] + tail = payload[BLOCK_SIZE:] + + added, complete = asm.add_block(0, 0, first) + assert added + assert not complete + assert asm.received[0] == len(first) + + added, complete = asm.add_block(0, 0, first) + assert not added + assert not complete + assert asm.received[0] == len(first) + + added, complete = asm.add_block(0, BLOCK_SIZE, tail) + assert added + assert complete + assert hashlib.sha1(asm.piece_bytes(0)).digest() == meta.piece_hashes[0] + + +def test_endgame_rearms_only_unfinished_pieces_on_interval(): + meta = FakeMeta([b"a", b"b", b"c", b"d"]) + done = bytearray([0, 1, 0, 0]) + eng = FakeEngine() + ctl = EndgameController(meta, min_pieces=3, peer_factor=2.0, interval=3.0) + + ctl.maybe_rearm(eng, 7, done, done_count=1, connected=2, now=10.0) + assert eng.priorities == [(7, 0, 255), (7, 2, 255), (7, 3, 255)] + assert eng.rearms == [(7, 0), (7, 2), (7, 3)] + + ctl.maybe_rearm(eng, 7, done, done_count=1, connected=2, now=11.0) + assert eng.rearms == [(7, 0), (7, 2), (7, 3)] + + done[2] = 1 + ctl.maybe_rearm(eng, 7, done, done_count=2, connected=2, now=13.0) + assert eng.rearms[-2:] == [(7, 0), (7, 3)] + + +if __name__ == "__main__": + test_piece_assembler_ignores_duplicate_blocks() + test_endgame_rearms_only_unfinished_pieces_on_interval() + print("swarm endgame OK") diff --git a/tests/test_tracker_ffi_dht.py b/tests/test_tracker_ffi_dht.py new file mode 100644 index 0000000..4977fc5 --- /dev/null +++ b/tests/test_tracker_ffi_dht.py @@ -0,0 +1,93 @@ +""" +Smoke tests for the torrent-tracker DHT ctypes bindings. + +Run with: python tests/test_tracker_ffi_dht.py +""" +from __future__ import annotations + +import ctypes as C +import os +import sys + +ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) +sys.path.insert(0, os.path.join(ROOT, "harness")) + +from tracker_ffi import ( # noqa: E402 + DHTClient, + DHTMessage, + DHT_MSG_QUERY, + DHT_MSG_RESPONSE, + DHT_QUERY_GET_PEERS, + TRACKER_ADDR_IPV4, + TRACKER_ADDR_IPV6, + TRACKER_OK, + TrackerPeer, +) + + +def _parse(client: DHTClient, raw: bytes) -> DHTMessage: + msg = DHTMessage() + raw_buf = C.create_string_buffer(raw, len(raw)) + rc = client.lib.dht_parse_message(raw_buf, len(raw), C.byref(msg)) + assert rc == TRACKER_OK + return msg + + +def test_get_peers_query_roundtrips_through_tracker_library(): + client = DHTClient(bootstrap=()) + client.node_id = b"abcdefghij0123456789" + tx = b"aa" + info_hash = bytes(range(20)) + + msg = _parse(client, client._get_peers_packet(info_hash, tx)) + + assert msg.type == DHT_MSG_QUERY + assert msg.query == DHT_QUERY_GET_PEERS + assert bytes(msg.transaction[:msg.transaction_len]) == tx + assert bytes(msg.id) == client.node_id + assert bytes(msg.info_hash) == info_hash + assert msg.want_ipv4 == 1 + assert msg.want_ipv6 == 1 + + +def test_peers_response_parses_to_endpoint_tuples(): + client = DHTClient(bootstrap=()) + tx = b"bb" + node_id = b"mnopqrstuvwxyz123456" + token = b"tok" + peers = (TrackerPeer * 2)() + + peers[0].family = TRACKER_ADDR_IPV4 + for i, b in enumerate((8, 8, 8, 8)): + peers[0].addr[i] = b + peers[0].port = 51413 + + peers[1].family = TRACKER_ADDR_IPV6 + peers[1].addr[15] = 2 + peers[1].port = 51414 + + buf = C.create_string_buffer(1024) + written = C.c_size_t() + tx_buf = C.create_string_buffer(tx, len(tx)) + id_buf = C.create_string_buffer(node_id, len(node_id)) + token_buf = C.create_string_buffer(token, len(token)) + rc = client.lib.dht_write_peers_response( + tx_buf, len(tx), id_buf, token_buf, len(token), peers, 2, + buf, C.sizeof(buf), C.byref(written)) + assert rc == TRACKER_OK + + msg = _parse(client, buf.raw[:written.value]) + + assert msg.type == DHT_MSG_RESPONSE + assert bytes(msg.transaction[:msg.transaction_len]) == tx + assert bytes(msg.id) == node_id + assert bytes(msg.token[:msg.token_len]) == token + assert msg.peer_count == 2 + assert client._peer_endpoint(msg.peers[0]) == ("8.8.8.8", 51413) + assert client._peer_endpoint(msg.peers[1]) == ("::2", 51414) + + +if __name__ == "__main__": + test_get_peers_query_roundtrips_through_tracker_library() + test_peers_response_parses_to_endpoint_tuples() + print("tracker DHT ffi OK") diff --git a/tests/test_utp.py b/tests/test_utp.py new file mode 100644 index 0000000..d956f32 --- /dev/null +++ b/tests/test_utp.py @@ -0,0 +1,159 @@ +""" +µTP (BEP-29) tests against a libtorrent seed with TCP disabled, so the only way +to reach it is over µTP/UDP. A successful byte-for-byte download proves the µTP +transport works end to end. + + * test_utp_download - plaintext µTP (engine utp=1) + * test_utp_tcp_fallback - engine prefers TCP, then falls back to µTP + * test_utp_encrypted - MSE over µTP (engine utp=1, encryption=1) against a + seed that is both µTP-only and encryption-forced + +Run with: python tests/test_utp.py +""" +from __future__ import annotations + +import hashlib +import os +import sys +import tempfile +import time + +ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) +sys.path.insert(0, os.path.join(ROOT, "harness")) + +import libtorrent as lt # noqa: E402 + +from harness import load_metadata # noqa: E402 +from engine_ffi import Engine, EngineConfig, STATE_ERROR, ERROR_NAMES # noqa: E402 +from test_localseed import ensure_built, make_torrent, pick_listen_port # noqa: E402 + +LIB = os.path.join(ROOT, "build", "libtorrentpeer.so") + + +def start_utp_seed(root: str, torrent_path: str, encrypted: bool): + port = pick_listen_port() + settings = { + "listen_interfaces": f"127.0.0.1:{port}", + "enable_dht": False, "enable_lsd": False, + "enable_upnp": False, "enable_natpmp": False, + # µTP only: refuse TCP entirely. + "enable_outgoing_tcp": False, + "enable_incoming_tcp": False, + "enable_outgoing_utp": True, + "enable_incoming_utp": True, + } + if encrypted: + settings.update({ + "in_enc_policy": int(lt.enc_policy.forced), + "out_enc_policy": int(lt.enc_policy.forced), + "allowed_enc_level": int(lt.enc_level.rc4), + "prefer_rc4": True, + }) + else: + settings.update({ + "in_enc_policy": int(lt.enc_policy.disabled), + "out_enc_policy": int(lt.enc_policy.disabled), + }) + ses = lt.session(settings) + h = ses.add_torrent({"ti": lt.torrent_info(torrent_path), "save_path": root, + "flags": lt.torrent_flags.seed_mode}) + deadline = time.time() + 15 + while time.time() < deadline and not h.status().is_seeding: + time.sleep(0.1) + assert h.status().is_seeding, "seed did not become ready" + return ses, h, ses.listen_port() or port + + +def _download(meta, port, utp, encryption, fallback=0, + connect_timeout_ms=0, timeout=90.0): + with Engine(EngineConfig(loop_count=1, slots_per_loop=1024, max_pipeline=256, + utp=utp, encryption=encryption, + fallback=fallback, + connect_timeout_ms=connect_timeout_ms), + lib_path=LIB) as eng: + tid = eng.add_torrent(meta.info_hash, b"-PC0001-" + os.urandom(12), + meta.piece_length, meta.total_size, meta.num_pieces) + eng.set_priorities(tid, [1] * meta.num_pieces) + eng.add_peer(tid, "127.0.0.1", port) + + buffers = [bytearray(meta.piece_len(i)) for i in range(meta.num_pieces)] + received = [0] * meta.num_pieces + done = bytearray(meta.num_pieces) + done_count = 0 + deadline = time.time() + timeout + while done_count < meta.num_pieces: + st = eng.status(tid) + if st.state == STATE_ERROR: + raise RuntimeError(f"engine error: {ERROR_NAMES[st.error]}") + descs = eng.poll_ready() + if not descs: + eng.wait(200) + if time.time() > deadline: + raise TimeoutError(f"stalled at {done_count}/{meta.num_pieces} " + f"(connected={st.peers_connected} " + f"failed={st.peers_failed})") + continue + for x in descs: + buf = buffers[x.piece] + buf[x.begin:x.begin + x.len] = eng.block_data(x.loop, x.slot, x.len) + eng.release(x.loop, x.slot) + received[x.piece] += x.len + if not done[x.piece] and received[x.piece] >= meta.piece_len(x.piece): + if hashlib.sha1(bytes(buf)).digest() != meta.piece_hashes[x.piece]: + raise ValueError(f"piece {x.piece} hash mismatch") + done[x.piece] = 1 + done_count += 1 + eng.set_priority(tid, x.piece, 0) + return b"".join(bytes(b) for b in buffers) + + +def test_utp_download(): + ensure_built() + size = 8 * 1024 * 1024 + with tempfile.TemporaryDirectory() as root: + torrent, original, _ = make_torrent(root, size) + meta = load_metadata(torrent) + ses, h, port = start_utp_seed(root, torrent, encrypted=False) + try: + got = _download(meta, port, utp=1, encryption=0) + finally: + ses.remove_torrent(h) + assert got == original, "µTP download bytes differ from original" + print(f"µTP (plaintext) OK: {size/1e6:.1f} MB over UDP") + + +def test_utp_tcp_fallback(): + ensure_built() + size = 8 * 1024 * 1024 + with tempfile.TemporaryDirectory() as root: + torrent, original, _ = make_torrent(root, size) + meta = load_metadata(torrent) + ses, h, port = start_utp_seed(root, torrent, encrypted=False) + try: + got = _download(meta, port, utp=0, encryption=0, fallback=1, + connect_timeout_ms=1000) + finally: + ses.remove_torrent(h) + assert got == original, "TCP->µTP fallback bytes differ from original" + print(f"µTP fallback OK: TCP failed over to UDP for {size/1e6:.1f} MB") + + +def test_utp_encrypted(): + ensure_built() + size = 8 * 1024 * 1024 + with tempfile.TemporaryDirectory() as root: + torrent, original, _ = make_torrent(root, size) + meta = load_metadata(torrent) + ses, h, port = start_utp_seed(root, torrent, encrypted=True) + try: + got = _download(meta, port, utp=1, encryption=1) + finally: + ses.remove_torrent(h) + assert got == original, "MSE-over-µTP bytes differ from original" + print(f"µTP + MSE OK: {size/1e6:.1f} MB over encrypted UDP") + + +if __name__ == "__main__": + test_utp_download() + test_utp_tcp_fallback() + test_utp_encrypted()