mirror of
https://github.com/ruvnet/RuView
synced 2026-08-09 20:21:43 +00:00
c9fde3cba5
ADR-096 train integration. Additive — does NOT modify model.rs. The
existing WiFiDensePoseModel forward stays bit-equivalent for back-compat.
New code lives in temporal_aether.rs behind the `aether-sparse-temporal`
feature flag (which itself requires `tch-backend`).
Architecture:
tch::Tensor [T, in_dim] ──── tch nn::Linear (q/k/v projections)
↓
[T, q_heads*head_dim] etc
↓
tch_to_tensor3 (CPU, f32, 1× copy)
↓
ruvllm_sparse_attention::Tensor3
↓
AetherTemporalHead::forward()
↓
Tensor3 [T, q_heads, head_dim]
↓
tensor3_to_tch (1× copy)
↓
tch::Tensor [T, q_heads*head_dim]
↓
tch nn::Linear (output projection)
↓
tch::Tensor [T, in_dim]
Why additive rather than swapping `apply_antenna_attention` /
`apply_spatial_attention` in model.rs: those are over antenna and
spatial axes, not temporal — ADR-096 §8.1 was right that AETHER
doesn't currently HAVE a temporal-axis attention. This commit adds
that path without disturbing the others, so the §5 validation gate
can A/B the two options before flipping the production default.
Scope notes:
- B=1 prefill only this version. Multi-batch lands when §5 turns
green and we need to take perf seriously. The forward expects
`[T, in_dim]` not `[B, T, in_dim]`; documented in the file.
- Streaming step() bridge deferred — KvCache lifecycle ties to
PoseTrack per ADR-096 §8.5, which is signal-side not train-side.
- Two CPU memory copies per call (in + out). For training-rate
forwards (~100/sec at batch 16) this is negligible vs the actual
attention work; for inference-rate streaming it'd be the
bottleneck and a zero-copy path is the natural follow-up.
Build verification:
- Source compiles cleanly with cargo check on the host crate
(`-p wifi-densepose-temporal`, 21/21 tests still passing).
- The train crate's tch-backend build is environmentally blocked
on this Windows machine — torch-sys fails to link against the
system PyTorch 2.11 + MSVC 14.50 toolchain. This predates this
commit and affects all tch-bound code paths in the workspace.
CI runners with working libtorch will verify the new module
builds; the source follows the same nn::Linear / Module patterns
the existing model.rs uses.
Feature gating ensures default builds are byte-equivalent. Off by
default; enable with `--features aether-sparse-temporal`.
Co-Authored-By: claude-flow <ruv@ruv.net>
101 lines
2.8 KiB
TOML
101 lines
2.8 KiB
TOML
[package]
|
|
name = "wifi-densepose-train"
|
|
version = "0.3.0"
|
|
edition = "2021"
|
|
authors = ["rUv <ruv@ruv.net>", "WiFi-DensePose Contributors"]
|
|
license = "MIT OR Apache-2.0"
|
|
description = "Training pipeline for WiFi-DensePose pose estimation"
|
|
repository = "https://github.com/ruvnet/wifi-densepose"
|
|
documentation = "https://docs.rs/wifi-densepose-train"
|
|
keywords = ["wifi", "training", "pose-estimation", "deep-learning"]
|
|
categories = ["science", "computer-vision"]
|
|
readme = "README.md"
|
|
|
|
[[bin]]
|
|
name = "train"
|
|
path = "src/bin/train.rs"
|
|
|
|
[[bin]]
|
|
name = "verify-training"
|
|
path = "src/bin/verify_training.rs"
|
|
required-features = ["tch-backend"]
|
|
|
|
[features]
|
|
default = []
|
|
tch-backend = ["tch"]
|
|
cuda = ["tch-backend"]
|
|
# ADR-096 sparse-GQA temporal head. Pulls wifi-densepose-temporal in
|
|
# alongside tch — the new path is additive, doesn't touch the existing
|
|
# model.rs code paths, and stays opt-in until the §5 validation gate
|
|
# clears.
|
|
aether-sparse-temporal = ["tch-backend", "dep:wifi-densepose-temporal"]
|
|
|
|
[dependencies]
|
|
# Internal crates
|
|
wifi-densepose-signal = { version = "0.3.0", path = "../wifi-densepose-signal", default-features = false }
|
|
wifi-densepose-nn = { version = "0.3.0", path = "../wifi-densepose-nn" }
|
|
|
|
# Core
|
|
thiserror.workspace = true
|
|
anyhow.workspace = true
|
|
serde = { workspace = true, features = ["derive"] }
|
|
serde_json.workspace = true
|
|
|
|
# Tensor / math
|
|
ndarray.workspace = true
|
|
num-complex.workspace = true
|
|
num-traits.workspace = true
|
|
|
|
# PyTorch bindings (optional — only enabled by `tch-backend` feature)
|
|
tch = { workspace = true, optional = true }
|
|
|
|
# Graph algorithms (min-cut for optimal keypoint assignment)
|
|
petgraph.workspace = true
|
|
|
|
# ruvector integration (subpolynomial min-cut, sparse solvers, temporal compression, attention)
|
|
ruvector-mincut = { workspace = true }
|
|
ruvector-attn-mincut = { workspace = true }
|
|
ruvector-temporal-tensor = { workspace = true }
|
|
ruvector-solver = { workspace = true }
|
|
ruvector-attention = { workspace = true }
|
|
|
|
# AETHER temporal head (ADR-096). Optional + tch-gated — only meaningful
|
|
# alongside the existing tch-bound model graph.
|
|
wifi-densepose-temporal = { workspace = true, optional = true }
|
|
|
|
# Data loading
|
|
ndarray-npy.workspace = true
|
|
memmap2 = "0.9"
|
|
walkdir.workspace = true
|
|
|
|
# Serialization
|
|
csv.workspace = true
|
|
toml = "0.8"
|
|
|
|
# Logging / progress
|
|
tracing.workspace = true
|
|
tracing-subscriber.workspace = true
|
|
indicatif.workspace = true
|
|
|
|
# Async (subset of features needed by training pipeline)
|
|
tokio = { workspace = true, features = ["rt", "rt-multi-thread", "macros", "fs"] }
|
|
|
|
# Crypto (for proof hash)
|
|
sha2.workspace = true
|
|
|
|
# CLI
|
|
clap.workspace = true
|
|
|
|
# Time
|
|
chrono = { version = "0.4", features = ["serde"] }
|
|
|
|
[dev-dependencies]
|
|
criterion.workspace = true
|
|
proptest.workspace = true
|
|
tempfile = "3.10"
|
|
approx = "0.5"
|
|
|
|
[[bench]]
|
|
name = "training_bench"
|
|
harness = false
|