From 82e87c5f6bf48df79ce54e2235affc11a50ddaf9 Mon Sep 17 00:00:00 2001
From: jenya
Date: Wed, 30 Sep 2026 17:00:59 +0300
Subject: [PATCH 1/4] Engine: reject invalid input, typed errors, real CSV
parsing
Correctness
- duplicate pre-trust peers: the last value wins (same as local trust);
before, duplicates were normalized separately and then collapsed by
to_dense, so pre-trust and the scores could stop summing to 1
- Vector::to_dense adds repeated indices instead of overwriting them
- weights must be finite and non-negative; NaN/inf are rejected with the
line number instead of spinning forever
- negative weights (distrust) are rejected until their effect on the
scores is defined; the unused discount step is off the engine path
- compute() caps iterations at 10,000 (alpha = 0 on a periodic graph
used to loop forever) and guards against non-finite deltas
- flat tail check compares only the top num_leaders peers
Robustness and cleanup
- Error enum with file, line and reason instead of String errors
- CSV read with the csv crate: quotes, embedded commas, escaped quotes;
line numbers stay correct across blank lines; same speed as before
- CLI: usage message, error: ... with exit code 1, no expect/unwrap,
CSV-quoted output, quiet on a closed pipe
- wasm-bindgen tests for run(); start function renamed from main
- removed ConvergenceChecker, duplicate canonicalize, stale wrappers
and resolved todos; cargo fmt and clippy clean on native and wasm32
- crossbeam-epoch 0.9.21 (RUSTSEC-2026-0204)
- Cargo.toml license field
---
Cargo.lock | 224 +++++++++++++++++++++++++++++++++++----
Cargo.toml | 5 +
src/basic/eigentrust.rs | 119 +++++++++------------
src/basic/engine.rs | 64 +++++++----
src/basic/input.rs | 212 ++++++++++++++++++++++++++++++++++++
src/basic/localtrust.rs | 120 ++++++++++-----------
src/basic/mod.rs | 1 +
src/basic/trustvector.rs | 159 ++++++++++++---------------
src/basic/util.rs | 76 ++-----------
src/error.rs | 93 ++++++++++++++++
src/lib.rs | 63 +++++++++--
src/main.rs | 72 ++++++++-----
src/sparse/entry.rs | 8 +-
src/sparse/matrix.rs | 45 ++++----
src/sparse/util.rs | 7 +-
src/sparse/vector.rs | 24 +++--
16 files changed, 883 insertions(+), 409 deletions(-)
create mode 100644 src/basic/input.rs
create mode 100644 src/error.rs
diff --git a/Cargo.lock b/Cargo.lock
index dd6e4fd..b99bdc3 100644
--- a/Cargo.lock
+++ b/Cargo.lock
@@ -11,12 +11,45 @@ dependencies = [
"memchr",
]
+[[package]]
+name = "async-trait"
+version = "0.1.92"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "82f6aeea286b8eb4dd3431a1be1b59d290ace00f5bfd8e2a159bc2a05e2c1667"
+dependencies = [
+ "proc-macro2",
+ "quote",
+ "syn",
+]
+
+[[package]]
+name = "autocfg"
+version = "1.5.1"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "f2032f911046de80f0a198e0901378627c33f59ea0ac00e363d481118bd70a53"
+
[[package]]
name = "bumpalo"
version = "3.16.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "79296716171880943b8470b5f8d03aa55eb2e645a4874bdbb28adb49162e012c"
+[[package]]
+name = "cast"
+version = "0.3.0"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "37b2a672a2cb129a2e41c10b1224bb368f9f37a2b16b612598138befd7b37eb5"
+
+[[package]]
+name = "cc"
+version = "1.5.1"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "f360145194ee8e21db5ee7f3fcd4fe52210864c75c985dae33218202c8bbe040"
+dependencies = [
+ "find-msvc-tools",
+ "shlex",
+]
+
[[package]]
name = "cfg-if"
version = "1.0.0"
@@ -65,9 +98,9 @@ dependencies = [
[[package]]
name = "crossbeam-epoch"
-version = "0.9.18"
+version = "0.9.21"
source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "5b82ac4a3c2ca9c3460964f020e1402edd5753411d7737aa39c3714ad1b5420e"
+checksum = "dc74980687109a3b14c72fd458107bf0baa1da1a1a805e178d15501ba9b86d9d"
dependencies = [
"crossbeam-utils",
]
@@ -78,12 +111,34 @@ version = "0.8.20"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "22ec99545bb0ed0ea7bb9b8e1e9122ea386ff8a48c0922e43f36d45ab09e0e80"
+[[package]]
+name = "csv"
+version = "1.4.0"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "52cd9d68cf7efc6ddfaaee42e7288d3a99d613d4b50f76ce9827ae0c6e14f938"
+dependencies = [
+ "csv-core",
+ "itoa",
+ "ryu",
+ "serde_core",
+]
+
+[[package]]
+name = "csv-core"
+version = "0.1.13"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "704a3c26996a80471189265814dbc2c257598b96b8a7feae2d31ace646bb9782"
+dependencies = [
+ "memchr",
+]
+
[[package]]
name = "eigentrust"
version = "0.1.0"
dependencies = [
"console_error_panic_hook",
"console_log",
+ "csv",
"env_logger",
"log",
"rayon",
@@ -91,6 +146,7 @@ dependencies = [
"serde_json",
"wasm-bindgen",
"wasm-bindgen-rayon",
+ "wasm-bindgen-test",
]
[[package]]
@@ -112,6 +168,12 @@ dependencies = [
"termcolor",
]
+[[package]]
+name = "find-msvc-tools"
+version = "0.1.14"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "aedcfb3409746eddb02b9e19ebda1c3394f759a152e48ee875a0844d1b955484"
+
[[package]]
name = "futures-core"
version = "0.3.34"
@@ -182,6 +244,12 @@ version = "0.2.158"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d8adc4bb1803a324070e64a98ae98f38934d91957a99cfb3a43dcbc01bc56439"
+[[package]]
+name = "libm"
+version = "0.2.16"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "b6d2cec3eae94f9f509c767b45932f1ada8350c4bdb85af2fcab4a3c14807981"
+
[[package]]
name = "log"
version = "0.4.22"
@@ -194,12 +262,47 @@ version = "2.7.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "78ca9ab1a0babb1e7d5695e3530886289c18cf2f87ec19a575a0abdce112e3a3"
+[[package]]
+name = "minicov"
+version = "0.3.8"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "4869b6a491569605d66d3952bcdf03df789e5b536e5f0cf7758a7f08a55ae24d"
+dependencies = [
+ "cc",
+ "walkdir",
+]
+
+[[package]]
+name = "nu-ansi-term"
+version = "0.50.3"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "7957b9740744892f114936ab4a57b3f487491bbeafaf8083688b16841a4240e5"
+dependencies = [
+ "windows-sys 0.59.0",
+]
+
+[[package]]
+name = "num-traits"
+version = "0.2.19"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "071dfc062690e90b734c0b2273ce72ad0ffa95f0c74596bc250dcfd960262841"
+dependencies = [
+ "autocfg",
+ "libm",
+]
+
[[package]]
name = "once_cell"
version = "1.19.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "3fdb12b2476b595f9358c5161aa467c2438859caa136dec86c26fdd2efe17b92"
+[[package]]
+name = "oorandom"
+version = "11.1.5"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "d6790f58c7ff633d8771f42965289203411a5e5c68388703c06e14f24770b41e"
+
[[package]]
name = "pin-project-lite"
version = "0.2.17"
@@ -287,24 +390,43 @@ version = "1.0.18"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "f3cb5ba0dc43242ce17de99c180e96db90b235b8a9fdc9543c96d2209116bd9f"
+[[package]]
+name = "same-file"
+version = "1.0.6"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "93fc1dc3aaa9bfed95e02e6eadabb4baf7e3078b0bd1b4d7b6b0b68378900502"
+dependencies = [
+ "winapi-util",
+]
+
[[package]]
name = "serde"
-version = "1.0.209"
+version = "1.0.229"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba"
+dependencies = [
+ "serde_core",
+ "serde_derive",
+]
+
+[[package]]
+name = "serde_core"
+version = "1.0.229"
source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "99fce0ffe7310761ca6bf9faf5115afbc19688edd00171d81b1bb1b116c63e09"
+checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48"
dependencies = [
"serde_derive",
]
[[package]]
name = "serde_derive"
-version = "1.0.209"
+version = "1.0.229"
source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "a5831b979fd7b5439637af1752d535ff49f4860c0f341d1baeb6faf0f4242170"
+checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348"
dependencies = [
"proc-macro2",
"quote",
- "syn 2.0.76",
+ "syn",
]
[[package]]
@@ -320,21 +442,16 @@ dependencies = [
]
[[package]]
-name = "slab"
-version = "0.4.12"
+name = "shlex"
+version = "2.0.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "0c790de23124f9ab44544d7ac05d60440adc586479ce501c1d6d7da3cd8c9cf5"
+checksum = "f8fadd59c855ef2080decdef8ff161eb6661b86933c9d82e5ba29dc602a55aba"
[[package]]
-name = "syn"
-version = "2.0.76"
+name = "slab"
+version = "0.4.12"
source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "578e081a14e0cefc3279b0472138c513f37b41a08d5a3cca9b6e4e8ceb6cd525"
-dependencies = [
- "proc-macro2",
- "quote",
- "unicode-ident",
-]
+checksum = "0c790de23124f9ab44544d7ac05d60440adc586479ce501c1d6d7da3cd8c9cf5"
[[package]]
name = "syn"
@@ -356,12 +473,31 @@ dependencies = [
"winapi-util",
]
+[[package]]
+name = "tokio"
+version = "1.53.1"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "202caea871b69668250d242070849eb495be178ed697a3e98aebce5bc81a0bed"
+dependencies = [
+ "pin-project-lite",
+]
+
[[package]]
name = "unicode-ident"
version = "1.0.12"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "3354b9ac3fae1ff6755cb6db53683adb661634f67557942dea4facebec0fee4b"
+[[package]]
+name = "walkdir"
+version = "2.5.0"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "29790946404f91d9c5d06f9874efddea1dc06c5efe94541a7d6863108e3a5e4b"
+dependencies = [
+ "same-file",
+ "winapi-util",
+]
+
[[package]]
name = "wasm-bindgen"
version = "0.2.129"
@@ -375,6 +511,17 @@ dependencies = [
"wasm-bindgen-shared",
]
+[[package]]
+name = "wasm-bindgen-futures"
+version = "0.4.79"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "3cbab34de2d982e9b48e18d216d04c4a6f641066ff19ffb699980f591ee3610e"
+dependencies = [
+ "js-sys",
+ "tokio",
+ "wasm-bindgen",
+]
+
[[package]]
name = "wasm-bindgen-macro"
version = "0.2.129"
@@ -394,7 +541,7 @@ dependencies = [
"bumpalo",
"proc-macro2",
"quote",
- "syn 3.0.6",
+ "syn",
"wasm-bindgen-shared",
]
@@ -419,6 +566,45 @@ dependencies = [
"unicode-ident",
]
+[[package]]
+name = "wasm-bindgen-test"
+version = "0.3.79"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "ae7499dfd45780a0a91d7ee6bb9ac51970a4479a41a89da443fdda5a39547d42"
+dependencies = [
+ "async-trait",
+ "cast",
+ "js-sys",
+ "libm",
+ "minicov",
+ "nu-ansi-term",
+ "num-traits",
+ "oorandom",
+ "serde",
+ "serde_json",
+ "wasm-bindgen",
+ "wasm-bindgen-futures",
+ "wasm-bindgen-test-macro",
+ "wasm-bindgen-test-shared",
+]
+
+[[package]]
+name = "wasm-bindgen-test-macro"
+version = "0.3.79"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "3b84b5ac638bfb168196a1a461fcc8f46a294a18b1b6be52133b4e0db122cc9f"
+dependencies = [
+ "proc-macro2",
+ "quote",
+ "syn",
+]
+
+[[package]]
+name = "wasm-bindgen-test-shared"
+version = "0.2.129"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "4f692aa943ccd88363733b77063f32cfed5bc6cbea8e6e8b251b302f881606fe"
+
[[package]]
name = "wasm_sync"
version = "0.1.2"
diff --git a/Cargo.toml b/Cargo.toml
index b24d865..cc63a15 100644
--- a/Cargo.toml
+++ b/Cargo.toml
@@ -3,6 +3,7 @@ name = "eigentrust"
version = "0.1.0"
edition = "2021"
authors = ["Jenya "]
+license = "MIT OR Apache-2.0"
description = "Fast EigenTrust reputation algorithm: global trust scores for peer-to-peer networks and social graphs, native and WebAssembly"
homepage = "https://eigentrust.jenyadoesapps.com"
readme = "README.md"
@@ -22,11 +23,15 @@ console_log = { version = "1.0", features = ["color"]}
wasm-bindgen-rayon = { version = "1.3", optional = true }
rayon = { version = "1.8", optional = true }
+[target.'cfg(target_arch = "wasm32")'.dev-dependencies]
+wasm-bindgen-test = "0.3"
+
[target.'cfg(not(target_arch = "wasm32"))'.dependencies]
env_logger = "0.10"
rayon = "1.8"
[dependencies]
+csv = "1.3"
serde = { version = "1.0", features = ["derive"] }
serde_json = "1.0"
log = "0.4"
diff --git a/src/basic/eigentrust.rs b/src/basic/eigentrust.rs
index f294706..9e3254c 100644
--- a/src/basic/eigentrust.rs
+++ b/src/basic/eigentrust.rs
@@ -1,3 +1,4 @@
+use crate::error::{Error, Result};
use crate::sparse::entry::Entry;
#[cfg(test)]
use crate::sparse::matrix::CSMatrix;
@@ -6,64 +7,8 @@ use crate::sparse::util::KBNSummer;
use crate::sparse::vector::Vector;
use std::cmp;
-// Canonicalize scales sparse entries in-place so that their values sum to one.
-// If entries sum to zero, Canonicalize returns an error indicating a zero-sum vector.
-pub fn canonicalize(entries: &mut [Entry]) -> Result<(), String> {
- let sum: f64 = entries.iter().map(|entry| entry.value).sum();
- if sum == 0.0 {
- return Err("Zero sum vector".to_string());
- }
- for entry in entries.iter_mut() {
- entry.value /= sum;
- }
- Ok(())
-}
-
-pub struct ConvergenceChecker {
- iter: usize,
- t: Vector,
- d: f64,
- e: f64,
-}
-
-impl ConvergenceChecker {
- pub fn new(t0: &Vector, e: f64) -> ConvergenceChecker {
- ConvergenceChecker {
- iter: 0,
- t: t0.clone(),
- d: 2.0 * e, // initial sentinel
- e,
- }
- }
-
- pub fn update(&mut self, t: &Vector) -> Result<(), String> {
- let mut td = Vector::new(self.t.dim, vec![]);
- td.sub_vec(t, &self.t)?;
-
- let d = td.norm2();
-
- log::debug!(
- "one iteration={} log10dPace={} log10dRemaining={}",
- self.iter,
- (d / self.d).log10(),
- (d / self.e).log10()
- );
-
- self.t.assign(t);
- self.d = d;
- self.iter += 1;
- Ok(())
- }
-
- pub fn converged(&self) -> bool {
- self.d <= self.e
- }
-
- pub fn delta(&self) -> f64 {
- self.d
- }
-}
-
+// Stops the iteration only once the ranking of the top `num_leaders` peers has stayed
+// the same for `length` checks. compute() uses length 0, which disables it.
pub struct FlatTailChecker {
length: usize,
num_leaders: usize,
@@ -91,7 +36,12 @@ impl FlatTailChecker {
.partial_cmp(&a.value)
.unwrap_or(cmp::Ordering::Equal)
});
- let ranking: Vec = entries.iter().map(|entry| entry.index).collect();
+ // only the top num_leaders positions have to stay stable
+ let ranking: Vec = entries
+ .iter()
+ .take(self.num_leaders)
+ .map(|entry| entry.index)
+ .collect();
if ranking == self.stats.ranking {
self.stats.length += 1;
@@ -117,12 +67,15 @@ pub struct FlatTailStats {
pub ranking: Vec,
}
+// Upper bound on power iterations when the caller does not set one. With alpha > 0
+// the error shrinks by (1 - alpha) per step; alpha = 0 on a periodic graph never converges.
+pub const DEFAULT_MAX_ITERATIONS: usize = 10_000;
+
// Compute function implements the EigenTrust algorithm.
//
// The iteration runs on dense vectors: after a couple of iterations the trust
// vector is (almost) dense anyway, and a dense lookup turns every row dot
// product into O(nnz(row)) instead of a sparse-sparse merge walk.
-// todo Error instead of String
pub fn compute(
c: &CSRMatrix,
p: &Vector,
@@ -130,18 +83,18 @@ pub fn compute(
e: f64,
max_iterations: Option,
min_iterations: Option,
-) -> Result {
- if a.is_nan() {
- return Err("Error: alpha cannot be NaN".to_string());
+) -> Result {
+ if !(0.0..=1.0).contains(&a) {
+ return Err(Error::InvalidAlpha(a));
}
let n = c.cs_matrix.major_dim;
if n == 0 {
- return Err("Empty local trust matrix".to_string());
+ return Err(Error::EmptyLocalTrust);
}
if p.dim != n {
- return Err("Dimension mismatch".to_string());
+ return Err(Error::DimensionMismatch);
}
let ct = c.transpose()?;
@@ -156,7 +109,7 @@ pub fn compute(
let mut flat_tail_checker = FlatTailChecker::new(flat_tail, n);
let mut iter = 0;
- let max_iters = max_iterations.unwrap_or(usize::MAX);
+ let max_iters = max_iterations.unwrap_or(DEFAULT_MAX_ITERATIONS);
let min_iters = min_iterations.unwrap_or(1);
log::info!(
@@ -170,6 +123,9 @@ pub fn compute(
while iter < max_iters {
if iter >= min_iters {
let d = dense_delta_norm2(&t1, &prev);
+ if !d.is_finite() {
+ return Err(Error::NonFiniteScores);
+ }
prev.copy_from_slice(&t1);
log::trace!("iteration={} delta={}", iter, d);
@@ -195,7 +151,10 @@ pub fn compute(
}
if iter >= max_iters {
- return Err("Reached maximum iterations without convergence".to_string());
+ return Err(Error::NotConverged {
+ iterations: max_iters,
+ alpha: a,
+ });
}
log::info!(
@@ -253,7 +212,9 @@ fn mul_dense(m: &CSRMatrix, v: &[f64], out: &mut [f64]) {
// Subtracts from t each peer's trust-weighted distrust row:
// t -= sum_i t[i] * discounts[i], applied in distruster order.
-pub fn discount_trust_vector(t: &mut Vector, discounts: &CSRMatrix) -> Result<(), String> {
+// Subtracts each peer's trust-weighted distrust row. Not used by calculate_from_csv, see
+// extract_distrust.
+pub fn discount_trust_vector(t: &mut Vector, discounts: &CSRMatrix) -> Result<()> {
if discounts.cs_matrix.entries.iter().all(|row| row.is_empty()) {
return Ok(());
}
@@ -268,7 +229,7 @@ pub fn discount_trust_vector(t: &mut Vector, discounts: &CSRMatrix) -> Result<()
};
for entry in distrusts {
if entry.index >= result.len() {
- return Err("Dimension mismatch".to_string());
+ return Err(Error::DimensionMismatch);
}
let scaled = weight * entry.value;
if scaled != 0.0 {
@@ -616,4 +577,24 @@ mod tests {
let result = compute(&c, &p, a, e, None, None).unwrap();
assert_eq!(result, expected);
}
+
+ #[test]
+ fn test_alpha_zero_on_periodic_graph_stops() {
+ // a <-> b with all seed trust on a oscillates forever without teleport
+ let c = CSRMatrix::new(2, 2, vec![(0, 1, 1.0), (1, 0, 1.0)]);
+ let p = Vector::new(2, vec![Entry::new(0, 1.0)]);
+ let err = compute(&c, &p, 0.0, 1e-9, None, None).unwrap_err();
+ assert!(matches!(err, Error::NotConverged { .. }), "{}", err);
+ // with teleport it converges
+ assert!(compute(&c, &p, 0.1, 1e-9, None, None).is_ok());
+ }
+
+ #[test]
+ fn test_compute_rejects_bad_alpha() {
+ let c = CSRMatrix::new(1, 1, vec![(0, 0, 1.0)]);
+ let p = Vector::new(1, vec![Entry::new(0, 1.0)]);
+ for a in [f64::NAN, -0.1, 1.5] {
+ assert!(compute(&c, &p, a, 1e-9, None, None).is_err());
+ }
+ }
}
diff --git a/src/basic/engine.rs b/src/basic/engine.rs
index 64dc512..612fda7 100644
--- a/src/basic/engine.rs
+++ b/src/basic/engine.rs
@@ -1,29 +1,26 @@
-use super::util::strip_headers;
use crate::basic::eigentrust::compute;
-use crate::basic::eigentrust::discount_trust_vector;
-use crate::basic::localtrust::{canonicalize_local_trust, extract_distrust, read_local_trust_from_csv};
+use crate::basic::localtrust::{canonicalize_local_trust, read_local_trust_from_csv};
use crate::basic::trustvector::canonicalize_trust_vector;
use crate::basic::trustvector::read_trust_vector_from_csv;
+use crate::error::{Error, Result};
#[cfg(test)]
use std::fs;
-// todo array inputs
+// Ranks every peer from CSV input: `from,to[,weight]` local trust and `peer[,weight]`
+// pre-trust. Returns (peer, score) sorted by score, highest first; scores sum to 1.
pub fn calculate_from_csv(
localtrust_csv: &str,
pretrust_csv: &str,
alpha: Option,
-) -> Result, String> {
+) -> Result> {
log::info!("Compute starting...");
let a = alpha.unwrap_or(0.5);
if !(0.0..=1.0).contains(&a) {
- return Err(format!("alpha must be in [0, 1], got {}", a));
+ return Err(Error::InvalidAlpha(a));
}
- let localtrust_csv = strip_headers(localtrust_csv, 2);
- let pretrust_csv = strip_headers(pretrust_csv, 1);
-
let (mut local_trust, peers) = read_local_trust_from_csv(localtrust_csv)?;
let mut pre_trust = read_trust_vector_from_csv(pretrust_csv, &peers.map)?;
@@ -41,18 +38,11 @@ pub fn calculate_from_csv(
canonicalize_trust_vector(&mut pre_trust);
- let mut discounts = extract_distrust(&mut local_trust)?;
-
- canonicalize_local_trust(&mut local_trust, Some(pre_trust.clone()))?;
- canonicalize_local_trust(&mut discounts, None)?;
+ // negative (distrust) weights are rejected while parsing, see localtrust.rs
+ canonicalize_local_trust(&mut local_trust, Some(&pre_trust))?;
let trust_scores = compute(&local_trust, &pre_trust, a, e, None, None)?;
- // todo: discounted scores are computed but not returned (same as before),
- // decide whether distrust should affect the output
- let mut discounted = trust_scores.clone();
- discount_trust_vector(&mut discounted, &discounts)?;
-
let mut entries: Vec<(String, f64)> = trust_scores
.entries
.iter()
@@ -95,9 +85,9 @@ mod tests {
#[test]
fn test_calculate_from_csv_file() {
- let localtrust_csv =fs::read_to_string("./example/localtrust2.csv").expect("Failed to read localtrust CSV file");
- let pretrust_csv = fs
- ::read_to_string("./example/pretrust2.csv")
+ let localtrust_csv = fs::read_to_string("./example/localtrust2.csv")
+ .expect("Failed to read localtrust CSV file");
+ let pretrust_csv = fs::read_to_string("./example/pretrust2.csv")
.expect("Failed to read pretrust CSV file");
let entries = calculate_from_csv(&localtrust_csv, &pretrust_csv, None).unwrap();
@@ -109,4 +99,36 @@ mod tests {
assert_eq!(entries[0].1, 0.40356129084997394);
assert_eq!(entries[1].0, "0x9fc3b33884e1d056a8ca979833d686abd267f9f8");
}
+
+ fn total(entries: &[(String, f64)]) -> f64 {
+ entries.iter().map(|(_, s)| s).sum()
+ }
+
+ #[test]
+ fn test_scores_sum_to_one_with_duplicate_pretrust() {
+ let lt = "a,b,1\nb,c,1\nc,a,1\nc,d,1";
+ let pt = "a,1\nd,1\na,5\na,2";
+ let entries = calculate_from_csv(lt, pt, Some(0.3)).unwrap();
+ assert!((total(&entries) - 1.0).abs() < 1e-9, "{}", total(&entries));
+ // last value wins: a=2, d=1 is the same as a clean 2:1 pretrust
+ let clean = calculate_from_csv(lt, "a,2\nd,1", Some(0.3)).unwrap();
+ assert_eq!(entries, clean);
+ }
+
+ #[test]
+ fn test_rejects_bad_input() {
+ for (lt, pt) in [
+ ("a,b,NaN", "a"),
+ ("a,b,inf", "a"),
+ ("a,b,-1", "a"),
+ ("a,b,1", "a,NaN"),
+ ] {
+ assert!(calculate_from_csv(lt, pt, None).is_err(), "{} / {}", lt, pt);
+ }
+ }
+
+ #[test]
+ fn test_alpha_zero_periodic_returns_error() {
+ assert!(calculate_from_csv("a,b\nb,a", "a", Some(0.0)).is_err());
+ }
}
diff --git a/src/basic/input.rs b/src/basic/input.rs
new file mode 100644
index 0000000..2452eac
--- /dev/null
+++ b/src/basic/input.rs
@@ -0,0 +1,212 @@
+// CSV input: a real CSV reader (quotes, embedded commas, escaped quotes), an optional
+// header row, surrounding whitespace, CRLF and a UTF-8 BOM.
+
+use crate::error::{Error, Input, RecordError, Result};
+use csv::{ByteRecord, ReaderBuilder};
+
+const HEADER_NAMES: &[&str] = &[
+ "i", "j", "v", "from", "to", "value", "weight", "trust", "level", "peer", "id", "score",
+ "source", "target", "src", "dst", "truster", "trustee",
+];
+
+// Surrounding whitespace and quotes the CSV reader keeps, e.g. in ` "alice" `.
+pub fn clean_field(field: &str) -> &str {
+ let field = field.trim();
+ field
+ .strip_prefix('"')
+ .and_then(|f| f.strip_suffix('"'))
+ .unwrap_or(field)
+ .trim()
+}
+
+// A first record is a header when its value column is not a number, or, when the value
+// column is omitted (implicit weight 1), when every field is a well-known header name.
+// `value_column` is 2 for local trust and 1 for pre-trust.
+pub fn is_header(fields: &[&str], value_column: usize) -> bool {
+ match fields.get(value_column) {
+ Some(value) => value.parse::().is_err(),
+ None => fields
+ .iter()
+ .all(|f| HEADER_NAMES.contains(&f.to_ascii_lowercase().as_str())),
+ }
+}
+
+// Calls `f(line, fields)` for every non-empty record after the optional header.
+pub fn for_each_record(data: &str, input: Input, value_column: usize, mut f: F) -> Result<()>
+where
+ F: FnMut(u64, &[&str]) -> Result<()>,
+{
+ let data = data.trim_start_matches('\u{feff}');
+ let mut reader = ReaderBuilder::new()
+ .has_headers(false)
+ .flexible(true)
+ .from_reader(data.as_bytes());
+
+ // the csv crate reports a record's position before any blank lines it skipped and
+ // does not count those lines, so line numbers come from byte offsets
+ let bytes = data.as_bytes();
+ let mut counted = (0usize, 1u64);
+ let mut line_at = |byte: u64| {
+ let mut byte = (byte as usize).min(bytes.len());
+ while byte < bytes.len() && (bytes[byte] == b'\n' || bytes[byte] == b'\r') {
+ byte += 1;
+ }
+ if byte >= counted.0 {
+ counted.1 += bytes[counted.0..byte]
+ .iter()
+ .filter(|&&b| b == b'\n')
+ .count() as u64;
+ counted.0 = byte;
+ }
+ counted.1
+ };
+
+ let mut record = ByteRecord::new();
+ let mut first = true;
+ loop {
+ let more = match reader.read_byte_record(&mut record) {
+ Ok(more) => more,
+ Err(e) => {
+ return Err(Error::Record {
+ input,
+ line: e.position().map_or(0, |p| line_at(p.byte())),
+ error: RecordError::Malformed(e.to_string()),
+ })
+ }
+ };
+ if !more {
+ return Ok(());
+ }
+ let line = record.position().map_or(0, |p| line_at(p.byte()));
+
+ // at most 3 columns are used, extra columns are ignored
+ let mut buf = [""; 3];
+ let mut n = 0;
+ for field in record.iter().take(buf.len()) {
+ // the input is a &str, but a quoted field could still split a code point
+ let field = std::str::from_utf8(field).map_err(|e| Error::Record {
+ input,
+ line,
+ error: RecordError::Malformed(e.to_string()),
+ })?;
+ buf[n] = clean_field(field);
+ n += 1;
+ }
+ let fields = &buf[..n];
+ if fields.iter().all(|f| f.is_empty()) {
+ continue;
+ }
+ if first {
+ first = false;
+ if is_header(fields, value_column) {
+ continue;
+ }
+ }
+ f(line, fields)?;
+ }
+}
+
+// A trust weight: finite and non-negative, 1 when omitted.
+pub fn parse_weight(field: Option<&&str>) -> std::result::Result {
+ let Some(&field) = field else {
+ return Ok(1.0);
+ };
+ let weight = field
+ .parse::()
+ .map_err(|_| RecordError::InvalidWeight(field.to_string()))?;
+ if !weight.is_finite() {
+ return Err(RecordError::NonFiniteWeight(field.to_string()));
+ }
+ if weight < 0.0 {
+ return Err(RecordError::NegativeWeight(field.to_string()));
+ }
+ Ok(weight)
+}
+
+#[cfg(test)]
+mod tests {
+ use super::*;
+
+ fn records(data: &str, value_column: usize) -> Result)>> {
+ let mut out = vec![];
+ for_each_record(data, Input::LocalTrust, value_column, |line, f| {
+ out.push((line, f.iter().map(|s| s.to_string()).collect()));
+ Ok(())
+ })?;
+ Ok(out)
+ }
+
+ fn rows(data: &str, value_column: usize) -> Vec> {
+ records(data, value_column)
+ .unwrap()
+ .into_iter()
+ .map(|(_, f)| f)
+ .collect()
+ }
+
+ #[test]
+ fn test_headers() {
+ assert_eq!(rows("i,j,v\na,b,1\n", 2), [["a", "b", "1"]]);
+ assert_eq!(rows("i,j,v\r\na,b,1\r\n", 2), [["a", "b", "1"]]);
+ assert_eq!(rows("\u{feff}from,to,weight\na,b,1", 2), [["a", "b", "1"]]);
+ // headerless two-column local trust keeps its first edge
+ assert_eq!(rows("a,b\nb,c", 2), [["a", "b"], ["b", "c"]]);
+ assert_eq!(rows("i,j\na,b", 2), [["a", "b"]]);
+ assert_eq!(rows("i,v\nalice,1", 1), [["alice", "1"]]);
+ assert_eq!(rows("alice\nbob", 1), [["alice"], ["bob"]]);
+ assert_eq!(rows("peer\nalice", 1), [["alice"]]);
+ assert!(rows("", 1).is_empty());
+ }
+
+ #[test]
+ fn test_real_csv() {
+ // quoted fields with commas and escaped quotes, blank lines, spaces
+ let data = "\"Smith, J\",\"O\"\"Neil\",2\n\n a , b , 3 \n \"c\" , \"d\" ,1\n";
+ assert_eq!(
+ rows(data, 2),
+ [
+ ["Smith, J", "O\"Neil", "2"],
+ ["a", "b", "3"],
+ ["c", "d", "1"]
+ ]
+ );
+ }
+
+ #[test]
+ fn test_line_numbers() {
+ let lines: Vec = records("i,j,v\na,b,1\n\nc,d,2\r\n\r\n\r\ne,f", 2)
+ .unwrap()
+ .into_iter()
+ .map(|(l, _)| l)
+ .collect();
+ assert_eq!(lines, [2, 4, 7]);
+ }
+
+ #[test]
+ fn test_parse_weight() {
+ assert_eq!(parse_weight(None), Ok(1.0));
+ assert_eq!(parse_weight(Some(&"2.5")), Ok(2.5));
+ assert!(matches!(
+ parse_weight(Some(&"x")),
+ Err(RecordError::InvalidWeight(_))
+ ));
+ for bad in ["NaN", "inf", "-inf"] {
+ assert!(matches!(
+ parse_weight(Some(&bad)),
+ Err(RecordError::NonFiniteWeight(_))
+ ));
+ }
+ assert!(matches!(
+ parse_weight(Some(&"-1")),
+ Err(RecordError::NegativeWeight(_))
+ ));
+ }
+
+ #[test]
+ fn test_clean_field() {
+ assert_eq!(clean_field(" alice "), "alice");
+ assert_eq!(clean_field("\"alice\""), "alice");
+ assert_eq!(clean_field(" \" 0.5 \" "), "0.5");
+ assert_eq!(clean_field("\""), "\"");
+ }
+}
diff --git a/src/basic/localtrust.rs b/src/basic/localtrust.rs
index cbee4a5..9008232 100644
--- a/src/basic/localtrust.rs
+++ b/src/basic/localtrust.rs
@@ -1,113 +1,97 @@
-use super::util::{clean_field, PeersMap};
+use super::input::{for_each_record, parse_weight};
+use super::util::PeersMap;
+use crate::error::{Error, Input, RecordError, Result};
use crate::sparse::entry::Entry;
use crate::sparse::matrix::CSRMatrix;
use crate::sparse::vector::Vector;
+// Scales every row to sum to one. Rows without trust (dangling peers) get the pre-trust
+// distribution, so their share flows back to the seeds.
pub fn canonicalize_local_trust(
local_trust: &mut CSRMatrix,
- pre_trust: Option,
-) -> Result<(), String> {
+ pre_trust: Option<&Vector>,
+) -> Result<()> {
let n = local_trust.dims().0;
- if let Some(ref pre_trust_vec) = pre_trust {
- // if pre_trust_vec.entries.len() != n {
- if pre_trust_vec.entries.len() > n {
- return Err("Dimension mismatch".to_string());
+ if let Some(pre_trust) = pre_trust {
+ if pre_trust.entries.len() > n {
+ return Err(Error::DimensionMismatch);
}
}
- for i in 0..n {
- let mut in_row = local_trust.row_vector(i);
- let row_sum: f64 = in_row.entries.iter().map(|entry| entry.value).sum();
-
+ for row in local_trust.cs_matrix.entries.iter_mut() {
+ let row_sum: f64 = row.iter().map(|entry| entry.value).sum();
if row_sum == 0.0 {
- if let Some(ref pre_trust_vec) = pre_trust {
- local_trust.set_row_vector(i, Vector::new(n, pre_trust_vec.entries.clone()));
+ if let Some(pre_trust) = pre_trust {
+ row.clone_from(&pre_trust.entries);
}
} else {
- for entry in &mut in_row.entries {
+ for entry in row.iter_mut() {
entry.value /= row_sum;
}
- local_trust.set_row_vector(i, in_row);
}
}
Ok(())
}
-pub fn extract_distrust(local_trust: &mut CSRMatrix) -> Result {
+// Splits negative entries off into a separate distrust matrix (as positive values).
+// Not used by calculate_from_csv: negative weights are rejected while parsing until the
+// effect of distrust on the scores is defined.
+pub fn extract_distrust(local_trust: &mut CSRMatrix) -> CSRMatrix {
let n = local_trust.dims().0;
let mut distrust = CSRMatrix::new(n, n, vec![]);
- for truster in 0..n {
- let mut trust_row = local_trust.row_vector(truster);
+ for (truster, row) in local_trust.cs_matrix.entries.iter_mut().enumerate() {
let mut distrust_row = Vec::new();
-
- trust_row.entries.retain(|entry| {
+ row.retain(|entry| {
if entry.value >= 0.0 {
true
} else {
- distrust_row.push(Entry {
- index: entry.index,
- value: -entry.value,
- });
+ distrust_row.push(Entry::new(entry.index, -entry.value));
false
}
});
-
- local_trust.set_row_vector(truster, trust_row);
distrust.set_row_vector(truster, Vector::new(n, distrust_row));
}
- Ok(distrust)
+ distrust
}
-fn parse_csv_line(line: &str, peer_indices: &mut PeersMap) -> Result<(usize, usize, f64), String> {
- let mut fields = line.split(',').map(clean_field);
-
- let (from, to) = match (fields.next(), fields.next()) {
- (Some(from), Some(to)) => (from, to),
- _ => return Err("Too few fields".to_string()),
- };
- let from = peer_indices.insert_or_get(from);
- let to = peer_indices.insert_or_get(to);
- let level = match fields.next() {
- Some(level) => level.parse::().map_err(|_| "Invalid trust level")?,
- None => 1.0,
- };
- Ok((from, to, level))
-}
-
-// todo move csv logic out of this scope, cooentry
-pub fn read_local_trust_from_csv(csv_data: &str) -> Result<(CSRMatrix, PeersMap), String> {
+// Reads `from,to[,weight]` records into a square CSR matrix, one row per truster.
+// Peers get indices in order of first appearance. A repeated `from,to` pair keeps the
+// last weight.
+pub fn read_local_trust_from_csv(csv_data: &str) -> Result<(CSRMatrix, PeersMap)> {
// rows are filled directly while parsing; the matrix grows as new peers appear
let mut rows: Vec> = Vec::new();
- let mut peer_indices = PeersMap::new();
-
- for (count, line) in csv_data.lines().enumerate() {
- if line.trim().is_empty() {
- continue;
- }
- let (from, to, level) = parse_csv_line(line, &mut peer_indices).map_err(|e| {
- format!(
- "Cannot parse local trust CSV record #{}: {:?} {:?}",
- count + 1,
- e,
- line
- )
- })?;
+ let mut peers = PeersMap::new();
+
+ for_each_record(csv_data, Input::LocalTrust, 2, |line, fields| {
+ let record_error = |error| Error::Record {
+ input: Input::LocalTrust,
+ line,
+ error,
+ };
+ let (from, to) = match fields {
+ [from, to, ..] => (*from, *to),
+ _ => return Err(record_error(RecordError::TooFewFields)),
+ };
+ let level = parse_weight(fields.get(2)).map_err(record_error)?;
+ let from = peers.insert_or_get(from);
+ let to = peers.insert_or_get(to);
if from >= rows.len() {
rows.resize_with(from + 1, Vec::new);
}
rows[from].push(Entry::new(to, level));
- }
+ Ok(())
+ })?;
- let dim = peer_indices.names.len();
+ let dim = peers.names.len();
if dim == 0 {
- return Err("Local trust is empty".to_string());
+ return Err(Error::EmptyLocalTrust);
}
rows.resize_with(dim, Vec::new);
- Ok((CSRMatrix::from_rows(dim, rows), peer_indices))
+ Ok((CSRMatrix::from_rows(dim, rows), peers))
}
#[cfg(test)]
@@ -140,7 +124,7 @@ mod tests {
for test in test_cases {
let mut local_trust = test.local_trust.clone();
- let distrust = extract_distrust(&mut local_trust).expect("Failed to extract distrust");
+ let distrust = extract_distrust(&mut local_trust);
assert_eq!(
local_trust, test.expected_trust,
@@ -154,4 +138,12 @@ mod tests {
);
}
}
+
+ #[test]
+ fn test_local_trust_rejects_bad_levels() {
+ for bad in ["a,b,NaN", "a,b,inf", "a,b,-inf", "a,b,-1", "a,b,x"] {
+ assert!(read_local_trust_from_csv(bad).is_err(), "{}", bad);
+ }
+ assert!(read_local_trust_from_csv("a,b,0\nb,a,2.5").is_ok());
+ }
}
diff --git a/src/basic/mod.rs b/src/basic/mod.rs
index 6e07e00..c33416d 100644
--- a/src/basic/mod.rs
+++ b/src/basic/mod.rs
@@ -1,5 +1,6 @@
pub mod eigentrust;
pub mod engine;
+pub mod input;
pub mod localtrust;
pub mod trustvector;
pub mod util;
diff --git a/src/basic/trustvector.rs b/src/basic/trustvector.rs
index e4a975b..a6be60e 100644
--- a/src/basic/trustvector.rs
+++ b/src/basic/trustvector.rs
@@ -1,13 +1,14 @@
-use super::util::clean_field;
+use super::input::{for_each_record, parse_weight};
+use crate::error::{Error, Input, RecordError, Result};
use crate::sparse::entry::Entry;
use crate::sparse::vector::Vector;
-use std::collections::{HashMap, HashSet};
+use std::collections::HashMap;
// CanonicalizeTrustVector canonicalizes the trust vector in-place,
// scaling it so that the elements sum to one,
// or making it a uniform vector that sums to one if it's a zero vector.
pub fn canonicalize_trust_vector(v: &mut Vector) {
- if canonicalize(&mut v.entries).is_err() {
+ if !canonicalize(&mut v.entries) {
let dim = v.dim;
let c = 1.0 / dim as f64;
v.entries.clear();
@@ -17,118 +18,96 @@ pub fn canonicalize_trust_vector(v: &mut Vector) {
}
}
-// Helper function to canonicalize a vector in-place.
-// Returns an error if the vector is a zero vector.
-fn canonicalize(entries: &mut Vec) -> Result<(), &'static str> {
+// Scales entries in place to sum to one. Returns false for a zero vector.
+fn canonicalize(entries: &mut [Entry]) -> bool {
let sum: f64 = entries.iter().map(|entry| entry.value).sum();
-
if sum == 0.0 {
- return Err("Zero sum vector");
+ return false;
}
-
for entry in entries.iter_mut() {
entry.value /= sum;
}
-
- Ok(())
+ true
}
-enum DuplicateHandling {
- Allow,
- Remove,
- Fail,
-}
-
-// todo move csv logic out of this scope
+// Reads `peer[,weight]` records. Weights must be finite and non-negative, default 1.
+// A peer listed more than once keeps its last weight, same as local trust.
pub fn read_trust_vector_from_csv(
input: &str,
peer_indices: &HashMap,
-) -> Result {
- let mut count = 0;
+) -> Result {
+ let mut levels: HashMap = HashMap::new();
let mut max_peer = -1;
- let mut entries = Vec::new();
- let mut seen_peers = HashSet::new();
- let duplicate_handling = DuplicateHandling::Allow;
- let mut dublicate_count = 0;
-
- for line in input.lines() {
- count += 1;
- if line.trim().is_empty() {
- continue;
- }
- let fields: Vec<&str> = line.split(',').map(clean_field).collect();
+ let mut duplicate_count = 0;
- let (peer, level) = match fields.len() {
- 0 => return Err(format!("Too few fields in line {}", count)),
- _ => {
- let peer = parse_peer_id(fields[0], peer_indices).map_err(|e| {
- format!("Invalid peer {:?} in line {}: {}", fields[0], count, e)
- })?;
- let level = if fields.len() >= 2 {
- parse_trust_level(fields[1]).map_err(|e| {
- format!(
- "Invalid trust level {:?} in line {}: {}",
- fields[1], count, e
- )
- })?
- } else {
- 1.0
- };
- (peer, level)
- }
+ for_each_record(input, Input::PreTrust, 1, |line, fields| {
+ let record_error = |error| Error::Record {
+ input: Input::PreTrust,
+ line,
+ error,
};
+ let peer = *peer_indices
+ .get(fields[0])
+ .ok_or_else(|| record_error(RecordError::UnknownPeer(fields[0].to_string())))?;
+ let level = parse_weight(fields.get(1)).map_err(record_error)?;
- if seen_peers.contains(&peer) {
- match duplicate_handling {
- DuplicateHandling::Fail => {
- return Err(format!("Duplicate peer {:?} in line {}", fields[0], count));
- }
- DuplicateHandling::Remove => {
- dublicate_count += 1;
- continue;
- }
- DuplicateHandling::Allow => {
- dublicate_count += 1;
- }
- }
- } else {
- seen_peers.insert(peer);
+ if levels.insert(peer, level).is_some() {
+ duplicate_count += 1;
}
-
- if max_peer < peer as isize {
- max_peer = peer as isize;
- }
-
- entries.push(Entry {
- index: peer,
- value: level,
- });
- }
-
- if dublicate_count > 0 {
- log::warn!("Pretrust contains {} duplicate peers", dublicate_count);
+ max_peer = max_peer.max(peer as isize);
+ Ok(())
+ })?;
+
+ if duplicate_count > 0 {
+ log::warn!(
+ "Pretrust contains {} duplicate peers, the last value wins",
+ duplicate_count
+ );
}
+ let entries = levels
+ .into_iter()
+ .map(|(index, value)| Entry { index, value })
+ .collect();
Ok(Vector::new((max_peer + 1) as usize, entries))
}
-fn parse_peer_id(peer_str: &str, peer_indices: &HashMap) -> Result {
- peer_indices
- .get(peer_str)
- .cloned()
- .ok_or_else(|| format!("Invalid peer: {}", peer_str))
-}
-
-fn parse_trust_level(level_str: &str) -> Result {
- level_str
- .parse::()
- .map_err(|_| format!("Invalid trust level: {}", level_str))
-}
-
#[cfg(test)]
mod tests {
use super::*;
+ fn peers(names: &[&str]) -> HashMap {
+ names
+ .iter()
+ .enumerate()
+ .map(|(i, n)| (n.to_string(), i))
+ .collect()
+ }
+
+ #[test]
+ fn test_duplicate_pretrust_last_wins() {
+ let v = read_trust_vector_from_csv("a,1\nb,1\na,3", &peers(&["a", "b"])).unwrap();
+ assert_eq!(v.entries, vec![Entry::new(0, 3.0), Entry::new(1, 1.0)]);
+ }
+
+ #[test]
+ fn test_pretrust_rejects_non_finite_and_negative() {
+ // a bad first line reads as a header, so each bad line follows a valid one
+ for bad in [
+ "a,1\na,NaN",
+ "a,1\na,inf",
+ "a,1\na,-1",
+ "a,1\na,x",
+ "a,1\nb,1",
+ ] {
+ assert!(
+ read_trust_vector_from_csv(bad, &peers(&["a"])).is_err(),
+ "{}",
+ bad
+ );
+ }
+ }
+
#[test]
fn test_canonicalize_zero_vector_is_uniform_over_dim() {
let mut v = Vector::new(4, vec![]);
diff --git a/src/basic/util.rs b/src/basic/util.rs
index 276d859..a898196 100644
--- a/src/basic/util.rs
+++ b/src/basic/util.rs
@@ -23,6 +23,12 @@ pub struct PeersMap {
pub names: Vec,
}
+impl Default for PeersMap {
+ fn default() -> Self {
+ Self::new()
+ }
+}
+
impl PeersMap {
pub fn new() -> Self {
PeersMap {
@@ -46,73 +52,3 @@ impl PeersMap {
self.names.len()
}
}
-
-// Cleans a raw CSV field: surrounding whitespace and quotes.
-pub fn clean_field(field: &str) -> &str {
- let field = field.trim();
- field
- .strip_prefix('"')
- .and_then(|f| f.strip_suffix('"'))
- .unwrap_or(field)
- .trim()
-}
-
-const HEADER_NAMES: &[&str] = &[
- "i", "j", "v", "from", "to", "value", "weight", "trust", "level", "peer", "id", "score",
- "source", "target", "src", "dst", "truster", "trustee",
-];
-
-// Strips a leading UTF-8 BOM and a header line, if present.
-// `value_column` is the index of the numeric column (2 for local trust, 1 for pre-trust).
-// A first line is a header when its value column is not a number, or, when the value column
-// is omitted (implicit weight 1), when every field is a well-known header name.
-pub fn strip_headers(csv_content: &str, value_column: usize) -> &str {
- let csv_content = csv_content.trim_start_matches('\u{feff}');
- let (first_line, rest) = match csv_content.find('\n') {
- Some(i) => (&csv_content[..i], &csv_content[i + 1..]),
- None => (csv_content, ""),
- };
-
- let fields: Vec<&str> = first_line.split(',').map(clean_field).collect();
- let is_header = match fields.get(value_column) {
- Some(value) => value.parse::().is_err(),
- None => fields
- .iter()
- .all(|f| HEADER_NAMES.contains(&f.to_ascii_lowercase().as_str())),
- };
-
- if is_header {
- rest
- } else {
- csv_content
- }
-}
-
-#[cfg(test)]
-mod tests {
- use super::*;
-
- #[test]
- fn test_strip_headers() {
- assert_eq!(strip_headers("i,j,v\na,b,1\n", 2), "a,b,1\n");
- assert_eq!(strip_headers("i,j,v\r\na,b,1\r\n", 2), "a,b,1\r\n");
- assert_eq!(strip_headers("\u{feff}from,to,weight\na,b,1", 2), "a,b,1");
- // headerless two-column local trust keeps its first edge
- assert_eq!(strip_headers("a,b\nb,c", 2), "a,b\nb,c");
- assert_eq!(strip_headers("i,j\na,b", 2), "a,b");
- assert_eq!(strip_headers("a,b,1", 2), "a,b,1");
- assert_eq!(strip_headers("i,v\nalice,1", 1), "alice,1");
- assert_eq!(strip_headers("alice,1\nbob,1", 1), "alice,1\nbob,1");
- assert_eq!(strip_headers("alice\nbob", 1), "alice\nbob");
- assert_eq!(strip_headers("peer\nalice", 1), "alice");
- assert_eq!(strip_headers("", 1), "");
- }
-
- #[test]
- fn test_clean_field() {
- assert_eq!(clean_field(" alice "), "alice");
- assert_eq!(clean_field("\"alice\""), "alice");
- assert_eq!(clean_field(" \" 0.5 \" "), "0.5");
- assert_eq!(clean_field("\""), "\"");
- }
-}
diff --git a/src/error.rs b/src/error.rs
new file mode 100644
index 0000000..64c82a1
--- /dev/null
+++ b/src/error.rs
@@ -0,0 +1,93 @@
+use std::fmt;
+
+/// Which CSV input an error refers to.
+#[derive(Debug, Clone, Copy, PartialEq, Eq)]
+pub enum Input {
+ LocalTrust,
+ PreTrust,
+}
+
+impl fmt::Display for Input {
+ fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
+ f.write_str(match self {
+ Input::LocalTrust => "local trust",
+ Input::PreTrust => "pre-trust",
+ })
+ }
+}
+
+/// What is wrong with a single CSV record.
+#[derive(Debug, Clone, PartialEq)]
+pub enum RecordError {
+ /// The CSV itself is malformed, e.g. an unterminated quote.
+ Malformed(String),
+ TooFewFields,
+ InvalidWeight(String),
+ NonFiniteWeight(String),
+ /// Distrust has no defined effect on the scores yet, so it is rejected.
+ NegativeWeight(String),
+ /// A pre-trust peer that never appears in local trust.
+ UnknownPeer(String),
+}
+
+impl fmt::Display for RecordError {
+ fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
+ match self {
+ RecordError::Malformed(msg) => write!(f, "malformed CSV: {}", msg),
+ RecordError::TooFewFields => f.write_str("too few fields"),
+ RecordError::InvalidWeight(w) => write!(f, "weight {:?} is not a number", w),
+ RecordError::NonFiniteWeight(w) => write!(f, "weight {:?} must be finite", w),
+ RecordError::NegativeWeight(w) => {
+ write!(f, "weight {:?} is negative; distrust is not supported", w)
+ }
+ RecordError::UnknownPeer(p) => {
+ write!(f, "peer {:?} does not appear in local trust", p)
+ }
+ }
+ }
+}
+
+#[derive(Debug, Clone, PartialEq)]
+pub enum Error {
+ /// A CSV record could not be used. `line` is 1-based.
+ Record {
+ input: Input,
+ line: u64,
+ error: RecordError,
+ },
+ EmptyLocalTrust,
+ InvalidAlpha(f64),
+ DimensionMismatch,
+ /// The iteration produced NaN or infinity.
+ NonFiniteScores,
+ NotConverged {
+ iterations: usize,
+ alpha: f64,
+ },
+ InvalidUtf8(Input),
+}
+
+impl fmt::Display for Error {
+ fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
+ match self {
+ Error::Record { input, line, error } => {
+ write!(f, "{} CSV, line {}: {}", input, line, error)
+ }
+ Error::EmptyLocalTrust => f.write_str("local trust is empty"),
+ Error::InvalidAlpha(a) => write!(f, "alpha must be in [0, 1], got {}", a),
+ Error::DimensionMismatch => f.write_str("dimension mismatch"),
+ Error::NonFiniteScores => f.write_str("trust scores are not finite"),
+ Error::NotConverged { iterations, alpha } => write!(
+ f,
+ "did not converge in {} iterations with alpha {}; \
+ a small alpha on a periodic trust graph can oscillate, try a larger alpha",
+ iterations, alpha
+ ),
+ Error::InvalidUtf8(input) => write!(f, "{} CSV is not valid UTF-8", input),
+ }
+ }
+}
+
+impl std::error::Error for Error {}
+
+pub type Result = std::result::Result;
diff --git a/src/lib.rs b/src/lib.rs
index 818a009..fefed8f 100644
--- a/src/lib.rs
+++ b/src/lib.rs
@@ -1,8 +1,10 @@
#![cfg(target_arch = "wasm32")]
use crate::basic::engine::calculate_from_csv;
+use crate::error::{Error, Input};
use wasm_bindgen::prelude::*;
pub mod basic;
+pub mod error;
pub mod sparse;
use crate::basic::util::init_logger;
use std::panic;
@@ -12,7 +14,7 @@ use std::str;
pub use wasm_bindgen_rayon::init_thread_pool;
#[wasm_bindgen(start)]
-fn main() {
+fn start() {
panic::set_hook(Box::new(console_error_panic_hook::hook));
init_logger();
log::debug!("WASM Eigentrust connected");
@@ -21,10 +23,59 @@ fn main() {
#[wasm_bindgen]
// Returns JSON: {"Ok": [[peer, score], ...]} sorted by score, or {"Err": "message"}
pub fn run(localtrust_csv: &[u8], pretrust_csv: &[u8], alpha: f64) -> String {
- let result = match (str::from_utf8(localtrust_csv), str::from_utf8(pretrust_csv)) {
- (Ok(lt), Ok(pt)) => calculate_from_csv(lt, pt, Some(alpha)),
- _ => Err("CSV input is not valid UTF-8".to_string()),
- };
+ let result = str::from_utf8(localtrust_csv)
+ .map_err(|_| Error::InvalidUtf8(Input::LocalTrust))
+ .and_then(|lt| {
+ let pt =
+ str::from_utf8(pretrust_csv).map_err(|_| Error::InvalidUtf8(Input::PreTrust))?;
+ calculate_from_csv(lt, pt, Some(alpha))
+ })
+ // the JS side gets the message: {"Err": "local trust CSV, line 3: ..."}
+ .map_err(|e| e.to_string());
- serde_json::to_string(&result).unwrap_or_else(|e| format!("{{\"Err\":\"{}\"}}", e))
+ serde_json::to_string(&result)
+ .unwrap_or_else(|e| serde_json::json!({ "Err": e.to_string() }).to_string())
+}
+
+#[cfg(test)]
+mod tests {
+ use super::run;
+ use wasm_bindgen_test::wasm_bindgen_test;
+
+ fn scores(json: &str) -> Vec<(String, f64)> {
+ let v: serde_json::Value = serde_json::from_str(json).unwrap();
+ v["Ok"]
+ .as_array()
+ .expect("Ok result")
+ .iter()
+ .map(|e| (e[0].as_str().unwrap().to_string(), e[1].as_f64().unwrap()))
+ .collect()
+ }
+
+ #[wasm_bindgen_test]
+ fn run_returns_sorted_scores_summing_to_one() {
+ let out = run(b"alice,bob,2\nbob,carol,1\n", b"alice\n", 0.5);
+ let s = scores(&out);
+ assert_eq!(
+ s.iter().map(|(p, _)| p.as_str()).collect::>(),
+ ["alice", "bob", "carol"]
+ );
+ let total: f64 = s.iter().map(|(_, v)| v).sum();
+ assert!((total - 1.0).abs() < 1e-9);
+ }
+
+ #[wasm_bindgen_test]
+ fn run_reports_errors_as_err() {
+ for (lt, pt) in [(&b"a,b,NaN"[..], &b"a"[..]), (b"a,b,-1", b"a"), (b"", b"")] {
+ let out = run(lt, pt, 0.5);
+ assert!(out.starts_with("{\"Err\":"), "{}", out);
+ }
+ assert!(run(&[0xff, 0xfe], b"a", 0.5).starts_with("{\"Err\":"));
+ }
+
+ #[wasm_bindgen_test]
+ fn run_rejects_bad_alpha() {
+ assert!(run(b"a,b", b"a", f64::NAN).starts_with("{\"Err\":"));
+ assert!(run(b"a,b", b"a", 2.0).starts_with("{\"Err\":"));
+ }
}
diff --git a/src/main.rs b/src/main.rs
index 13b9dca..fcce6d2 100644
--- a/src/main.rs
+++ b/src/main.rs
@@ -1,44 +1,62 @@
use std::env;
use std::fs;
-use std::io::{self, BufWriter, Write};
-use std::process;
+use std::io;
+use std::process::ExitCode;
use crate::basic::engine::calculate_from_csv;
use crate::basic::util::init_logger;
pub mod basic;
+pub mod error;
pub mod sparse;
-fn main() {
- let args: Vec = env::args().collect();
- init_logger();
+const USAGE: &str = "usage: eigentrust [alpha]";
- if args.len() < 3 {
- log::error!(
- "Usage: {} [alpha]",
- args[0]
- );
- process::exit(1);
+fn main() -> ExitCode {
+ init_logger();
+ match run(env::args().skip(1).collect()) {
+ Ok(()) => ExitCode::SUCCESS,
+ Err(message) => {
+ eprintln!("error: {}", message);
+ ExitCode::FAILURE
+ }
}
+}
+
+fn run(args: Vec) -> Result<(), String> {
+ let (localtrust_path, pretrust_path, alpha) = match args.as_slice() {
+ [lt, pt] => (lt, pt, None),
+ [lt, pt, alpha] => {
+ let alpha = alpha
+ .parse::()
+ .map_err(|_| format!("alpha {:?} is not a number\n{}", alpha, USAGE))?;
+ (lt, pt, Some(alpha))
+ }
+ _ => return Err(USAGE.to_string()),
+ };
- let localtrust_csv_path = &args[1];
- let pretrust_csv_path = &args[2];
- let alpha = args.get(3).map(|a| a.parse::().expect("alpha must be a number"));
+ let read = |path: &String| {
+ fs::read_to_string(path).map_err(|e| format!("cannot read {}: {}", path, e))
+ };
+ let localtrust_csv = read(localtrust_path)?;
+ let pretrust_csv = read(pretrust_path)?;
- let localtrust_csv =
- fs::read_to_string(localtrust_csv_path).expect("Failed to read localtrust CSV file");
- let pretrust_csv =
- fs::read_to_string(pretrust_csv_path).expect("Failed to read pretrust CSV file");
+ let result =
+ calculate_from_csv(&localtrust_csv, &pretrust_csv, alpha).map_err(|e| e.to_string())?;
- let result = match calculate_from_csv(&localtrust_csv, &pretrust_csv, alpha) {
- Ok(result) => result,
- Err(e) => {
- log::error!("{}", e);
- process::exit(1);
+ // proper CSV so peer ids with commas or quotes round-trip
+ let write = || -> csv::Result<()> {
+ let mut out = csv::Writer::from_writer(io::stdout().lock());
+ for (name, score) in &result {
+ out.write_record([name.as_str(), &score.to_string()])?;
}
+ out.flush()?;
+ Ok(())
};
-
- let mut out = BufWriter::new(io::stdout().lock());
- for (name, score) in &result {
- writeln!(out, "{},{}", name, score).unwrap();
+ match write() {
+ // stdout closed early, e.g. `| head`
+ Err(e) if matches!(e.kind(), csv::ErrorKind::Io(io) if io.kind() == io::ErrorKind::BrokenPipe) => {
+ Ok(())
+ }
+ other => other.map_err(|e| format!("cannot write output: {}", e)),
}
}
diff --git a/src/sparse/entry.rs b/src/sparse/entry.rs
index e16a88e..45d0bae 100644
--- a/src/sparse/entry.rs
+++ b/src/sparse/entry.rs
@@ -88,7 +88,7 @@ mod tests {
("Empty", vec![], 0),
];
- for (name, mut entries, expected_len) in tests {
+ for (name, entries, expected_len) in tests {
let len = entries.len();
assert_eq!(
len, expected_len,
@@ -179,7 +179,7 @@ mod tests {
];
for (name, x, y, expected) in tests {
- let entries = vec![x.clone(), y.clone()];
+ let entries = [x.clone(), y.clone()];
let result = entries[0].row < entries[1].row
|| (entries[0].row == entries[1].row && entries[0].column < entries[1].column);
assert_eq!(
@@ -206,7 +206,7 @@ mod tests {
("Empty", vec![], 0),
];
- for (name, mut entries, expected_len) in tests {
+ for (name, entries, expected_len) in tests {
let len = entries.len();
assert_eq!(
len, expected_len,
@@ -297,7 +297,7 @@ mod tests {
];
for (name, x, y, expected) in tests {
- let entries = vec![x.clone(), y.clone()];
+ let entries = [x.clone(), y.clone()];
let result = entries[0].column < entries[1].column
|| (entries[0].column == entries[1].column && entries[0].row < entries[1].row);
assert_eq!(
diff --git a/src/sparse/matrix.rs b/src/sparse/matrix.rs
index 5ac000c..b77253b 100644
--- a/src/sparse/matrix.rs
+++ b/src/sparse/matrix.rs
@@ -1,6 +1,6 @@
use super::entry::Entry;
use super::vector::Vector;
-
+use crate::error::{Error, Result};
#[derive(Clone, PartialEq, Debug)]
pub struct CSMatrix {
@@ -9,6 +9,12 @@ pub struct CSMatrix {
pub entries: Vec>,
}
+impl Default for CSMatrix {
+ fn default() -> Self {
+ Self::new()
+ }
+}
+
impl CSMatrix {
pub fn new() -> Self {
Self {
@@ -24,9 +30,9 @@ impl CSMatrix {
self.entries.clear();
}
- pub fn dim(&self) -> Result {
+ pub fn dim(&self) -> Result {
if self.major_dim != self.minor_dim {
- return Err("Dimension mismatch");
+ return Err(Error::DimensionMismatch);
}
Ok(self.major_dim)
}
@@ -34,7 +40,7 @@ impl CSMatrix {
pub fn set_major_dim(&mut self, dim: usize) {
if self.entries.capacity() < dim {
let mut new_entries = Vec::with_capacity(dim);
- new_entries.extend(self.entries.drain(..));
+ new_entries.append(&mut self.entries);
self.entries = new_entries;
}
self.entries.resize_with(dim, Vec::new);
@@ -52,7 +58,7 @@ impl CSMatrix {
self.entries.iter().map(|row| row.len()).sum()
}
- pub fn transpose(&self) -> Result {
+ pub fn transpose(&self) -> Result {
let mut nnzs = vec![0; self.minor_dim];
for row_entries in &self.entries {
for entry in row_entries {
@@ -92,7 +98,6 @@ impl CSMatrix {
other.reset();
}
}
-//--
fn merge_span(s1: &[Entry], s2: &[Entry]) -> Vec {
let mut s = Vec::with_capacity(s1.len() + s2.len());
let mut i1 = 0;
@@ -193,13 +198,12 @@ impl CSRMatrix {
self.cs_matrix.entries[index] = vector.entries;
}
- pub fn transpose(&self) -> Result {
+ pub fn transpose(&self) -> Result {
let transposed = self.cs_matrix.transpose()?;
Ok(CSRMatrix {
cs_matrix: transposed,
})
}
- //--
pub fn transpose_to_csc(&self) -> CSCMatrix {
CSCMatrix {
cs_matrix: CSMatrix {
@@ -226,7 +230,6 @@ impl CSCMatrix {
self.cs_matrix.set_minor_dim(rows);
}
- // -
pub fn column_vector(&self, index: usize) -> Vector {
Vector {
dim: self.cs_matrix.minor_dim,
@@ -234,15 +237,13 @@ impl CSCMatrix {
}
}
- pub fn transpose(&self) -> Result {
+ pub fn transpose(&self) -> Result {
let transposed = self.cs_matrix.transpose()?;
Ok(CSCMatrix {
cs_matrix: transposed,
})
}
-
- // -
pub fn transpose_to_csr(&self) -> CSRMatrix {
CSRMatrix {
cs_matrix: CSMatrix {
@@ -254,20 +255,6 @@ impl CSCMatrix {
}
}
-// todo cooentry
-//--
-pub fn create_csr_matrix(rows: usize, cols: usize, entries: Vec<(usize, usize, f64)>) -> CSRMatrix {
- CSRMatrix::new(rows, cols, entries)
-}
-//--
-pub fn transpose_csr_matrix(matrix: &CSRMatrix) -> Result {
- matrix.transpose()
-}
-//-
-pub fn transpose_to_csc(matrix: &CSRMatrix) -> CSCMatrix {
- matrix.transpose_to_csc()
-}
-
#[cfg(test)]
mod tests {
use super::*;
@@ -512,7 +499,11 @@ mod tests {
#[test]
fn test_new_csr_matrix_duplicates_last_wins() {
- let m = CSRMatrix::new(2, 2, vec![(0, 1, 1.0), (0, 0, 3.0), (0, 1, 5.0), (1, 1, 2.0)]);
+ let m = CSRMatrix::new(
+ 2,
+ 2,
+ vec![(0, 1, 1.0), (0, 0, 3.0), (0, 1, 5.0), (1, 1, 2.0)],
+ );
assert_eq!(
m.cs_matrix.entries,
vec![
diff --git a/src/sparse/util.rs b/src/sparse/util.rs
index f24b06a..26c2f63 100644
--- a/src/sparse/util.rs
+++ b/src/sparse/util.rs
@@ -1,4 +1,3 @@
-
pub fn nil_if_empty(slice: Vec) -> Option> {
if slice.is_empty() {
None
@@ -20,6 +19,12 @@ pub struct KBNSummer {
compensation: f64,
}
+impl Default for KBNSummer {
+ fn default() -> Self {
+ Self::new()
+ }
+}
+
impl KBNSummer {
pub fn new() -> Self {
Self {
diff --git a/src/sparse/vector.rs b/src/sparse/vector.rs
index 76bcb5d..ce4ea5d 100644
--- a/src/sparse/vector.rs
+++ b/src/sparse/vector.rs
@@ -6,6 +6,7 @@ use std::cmp::Ordering;
use super::entry::Entry;
use super::matrix::CSRMatrix;
use super::util::KBNSummer;
+use crate::error::{Error, Result};
#[derive(Clone, PartialEq, Debug, Serialize)]
pub struct Vector {
@@ -22,8 +23,9 @@ impl Vector {
pub fn to_dense(&self) -> Vec {
let mut dense = vec![0.0; self.dim];
+ // add, not assign: a sparse vector with repeated indices means their sum
for e in &self.entries {
- dense[e.index] = e.value;
+ dense[e.index] += e.value;
}
dense
}
@@ -60,17 +62,17 @@ impl Vector {
self.entries.iter().map(|e| e.value).sum()
}
- pub fn add_vec(&mut self, v1: &Self, v2: &Self) -> Result<(), String> {
+ pub fn add_vec(&mut self, v1: &Self, v2: &Self) -> Result<()> {
self.binary_operation(v1, v2, |x, y| x + y)
}
- pub fn sub_vec(&mut self, v1: &Self, v2: &Self) -> Result<(), String> {
+ pub fn sub_vec(&mut self, v1: &Self, v2: &Self) -> Result<()> {
self.binary_operation(v1, v2, |x, y| x - y)
}
- pub fn scale_vec(&mut self, a: f64, v1: &Self) -> Result<(), String> {
+ pub fn scale_vec(&mut self, a: f64, v1: &Self) -> Result<()> {
if a.is_nan() {
- return Err("alpha cannot be NaN".to_string());
+ return Err(Error::InvalidAlpha(a));
}
if a == 0.0 {
self.dim = v1.dim;
@@ -91,10 +93,10 @@ impl Vector {
}
#[cfg(not(target_arch = "wasm32"))]
- pub fn mul_vec(&mut self, m: &CSRMatrix, v1: &Self) -> Result<(), String> {
+ pub fn mul_vec(&mut self, m: &CSRMatrix, v1: &Self) -> Result<()> {
let dim = m.cs_matrix.dim()?;
if dim != v1.dim {
- return Err("Dimension mismatch".to_string());
+ return Err(Error::DimensionMismatch);
}
let dense = v1.to_dense();
@@ -120,10 +122,10 @@ impl Vector {
}
#[cfg(target_arch = "wasm32")]
- pub fn mul_vec(&mut self, m: &CSRMatrix, v1: &Self) -> Result<(), String> {
+ pub fn mul_vec(&mut self, m: &CSRMatrix, v1: &Self) -> Result<()> {
let dim = m.cs_matrix.dim()?;
if dim != v1.dim {
- return Err("Dimension mismatch".to_string());
+ return Err(Error::DimensionMismatch);
}
let dense = v1.to_dense();
@@ -148,12 +150,12 @@ impl Vector {
self.entries.sort_by_key(|e| e.index);
}
- fn binary_operation(&mut self, v1: &Self, v2: &Self, op: F) -> Result<(), String>
+ fn binary_operation(&mut self, v1: &Self, v2: &Self, op: F) -> Result<()>
where
F: Fn(f64, f64) -> f64,
{
if v1.dim != v2.dim {
- return Err("Dimension mismatch".to_string());
+ return Err(Error::DimensionMismatch);
}
let mut entries = Vec::with_capacity(v1.entries.len() + v2.entries.len());
From e4d9bd0685e742d319abb98e2a693c7d72238862 Mon Sep 17 00:00:00 2001
From: jenya
Date: Wed, 30 Sep 2026 17:00:59 +0300
Subject: [PATCH 2/4] CI: fmt, clippy (native and wasm32), wasm tests and
builds, cargo audit
---
.github/workflows/ci.yml | 60 ++++++++++++++++++++++++++++++++++++++++
1 file changed, 60 insertions(+)
diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml
index 3e970b5..3cf5940 100644
--- a/.github/workflows/ci.yml
+++ b/.github/workflows/ci.yml
@@ -4,14 +4,74 @@ on:
push:
pull_request:
+env:
+ CARGO_TERM_COLOR: always
+
jobs:
+ fmt:
+ runs-on: ubuntu-latest
+ steps:
+ - uses: actions/checkout@v4
+ - uses: dtolnay/rust-toolchain@stable
+ with:
+ components: rustfmt
+ - run: cargo fmt --check
+
+ clippy:
+ runs-on: ubuntu-latest
+ steps:
+ - uses: actions/checkout@v4
+ - uses: dtolnay/rust-toolchain@stable
+ with:
+ components: clippy
+ targets: wasm32-unknown-unknown
+ - uses: Swatinem/rust-cache@v2
+ - name: Native
+ run: cargo clippy --release --all-targets -- -D warnings
+ - name: WebAssembly
+ run: cargo clippy --release --target wasm32-unknown-unknown --all-targets -- -D warnings
+
test:
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
- uses: dtolnay/rust-toolchain@stable
+ - uses: Swatinem/rust-cache@v2
- run: cargo test --release
+ wasm:
+ runs-on: ubuntu-latest
+ steps:
+ - uses: actions/checkout@v4
+ - uses: dtolnay/rust-toolchain@stable
+ with:
+ targets: wasm32-unknown-unknown
+ - uses: dtolnay/rust-toolchain@nightly
+ with:
+ components: rust-src
+ targets: wasm32-unknown-unknown
+ - run: rustup default stable
+ - uses: Swatinem/rust-cache@v2
+ - uses: taiki-e/install-action@v2
+ with:
+ tool: wasm-pack
+ - uses: actions/setup-node@v4
+ with:
+ node-version: 22
+ - name: Tests in Node
+ run: wasm-pack test --node --release
+ - name: Single-threaded and multithreaded builds
+ run: ./build.sh
+
+ audit:
+ runs-on: ubuntu-latest
+ steps:
+ - uses: actions/checkout@v4
+ - uses: taiki-e/install-action@v2
+ with:
+ tool: cargo-audit
+ - run: cargo audit
+
attribution:
runs-on: ubuntu-latest
steps:
From fa0418bde8a92a8c7ec6ef28559117c1ba1ff135 Mon Sep 17 00:00:00 2001
From: jenya
Date: Wed, 30 Sep 2026 17:00:59 +0300
Subject: [PATCH 3/4] Playground: quote-aware CSV, show engine errors
- CSV tab parses and writes quoted fields the same way as the engine
- engine errors (e.g. alpha 0 on a periodic network) are shown under
the ranking; before they were overwritten right away
---
demo/app.js | 53 ++++++++++++++++++++++++++++++++++++-------------
demo/index.html | 1 +
demo/style.css | 1 +
3 files changed, 41 insertions(+), 14 deletions(-)
diff --git a/demo/app.js b/demo/app.js
index 27bcb8f..6676f97 100644
--- a/demo/app.js
+++ b/demo/app.js
@@ -21,6 +21,7 @@ const state = {
hover: null,
scores: [], // [[peer, score]] from the last run, sorted
large: null, // { lt, pt, peers } when the network is too big to draw
+ error: null, // engine error for the current network
lastRun: null, // { ms, mode }
}
@@ -71,6 +72,7 @@ function clearNetwork() {
state.selected = null
state.large = null
state.scores = []
+ state.error = null
}
// ---------- presets ----------
@@ -157,11 +159,11 @@ function networkCsv() {
const lt = []
const linked = new Set()
for (const e of state.edges.values()) {
- lt.push(`${e.from},${e.to},${e.w}`)
+ lt.push(`${csvField(e.from)},${csvField(e.to)},${e.w}`)
linked.add(e.from)
linked.add(e.to)
}
- const pt = [...state.seeds].filter((s) => linked.has(s)).map((s) => `${s},1`)
+ const pt = [...state.seeds].filter((s) => linked.has(s)).map((s) => `${csvField(s)},1`)
return { lt: lt.join('\n'), pt: pt.join('\n') }
}
@@ -186,10 +188,10 @@ async function compute() {
const { lt, pt } = state.large || networkCsv()
const res = await runEngine(lt, pt, state.alpha)
running = false
+ state.error = res.error || null
if (res.error) {
state.scores = []
state.lastRun = null
- $('stats').textContent = res.error
} else {
state.scores = res.scores
state.lastRun = { ms: res.ms, threads: res.threads }
@@ -602,6 +604,8 @@ function renderRanking() {
}
stats.textContent = parts.join(lang() === 'zh' || lang() === 'ja' ? ',' : lang() === 'ar' ? '، ' : ', ')
}
+ $('engineError').textContent = state.error || ''
+ $('engineError').hidden = !state.error
$('engine').textContent = engineText()
}
@@ -737,23 +741,44 @@ function countLines(s) {
const HEADER_NAMES = new Set(['i', 'j', 'v', 'from', 'to', 'value', 'weight', 'trust', 'level', 'peer', 'id', 'score',
'source', 'target', 'src', 'dst', 'truster', 'trustee'])
-const clean = (f) => f.trim().replace(/^"(.*)"$/, '$1').trim()
+// one CSV line: quoted fields may contain commas and doubled quotes, same rules as the engine
+function splitCsvLine(line) {
+ const out = []
+ let cur = ''
+ let quoted = false
+ for (let i = 0; i < line.length; i++) {
+ const c = line[i]
+ if (quoted) {
+ if (c !== '"') cur += c
+ else if (line[i + 1] === '"') { cur += '"'; i++ }
+ else quoted = false
+ } else if (c === '"' && !cur.trim()) { cur = ''; quoted = true }
+ else if (c === ',') { out.push(cur.trim()); cur = '' }
+ else cur += c
+ }
+ out.push(cur.trim())
+ return out
+}
function parseRows(text, valueCol) {
- const lines = text.replace(/^/, '').split('\n')
const rows = []
- lines.forEach((line, i) => {
- if (!line.trim()) return
- const f = line.split(',').map(clean)
- if (i === 0) {
- const header = f.length > valueCol ? isNaN(parseFloat(f[valueCol])) : f.every((x) => HEADER_NAMES.has(x.toLowerCase()))
- if (header) return
+ let first = true
+ for (const line of text.replace(/^\uFEFF/, '').split(/\r?\n/)) {
+ if (!line.trim()) continue
+ const f = splitCsvLine(line)
+ if (first) {
+ first = false
+ const header = f.length > valueCol ? isNaN(Number(f[valueCol])) : f.every((x) => HEADER_NAMES.has(x.toLowerCase()))
+ if (header) continue
}
rows.push(f)
- })
+ }
return rows
}
+// quote a field when it needs it
+const csvField = (v) => /[",\r\n]|^\s|\s$/.test(String(v)) ? '"' + String(v).replace(/"/g, '""') + '"' : String(v)
+
$('csvLoad').addEventListener('click', async () => {
const err = $('csvError')
err.hidden = true
@@ -775,7 +800,7 @@ $('csvLoad').addEventListener('click', async () => {
if (ltRows) for (const r of ltRows) { peers.add(r[0]); peers.add(r[1]) }
if (!ltRows || peers.size > MAX_GRAPH_PEERS) {
clearNetwork()
- const count = ltRows ? peers.size : new Set(lt.split('\n').flatMap((l) => l.split(',').slice(0, 2).map(clean)).filter(Boolean)).size
+ const count = ltRows ? peers.size : new Set(lt.split('\n').flatMap((l) => splitCsvLine(l).slice(0, 2)).filter(Boolean)).size
state.large = { lt, pt, peers: count }
state.scores = res.scores
state.lastRun = { ms: res.ms, threads: res.threads }
@@ -790,7 +815,7 @@ $('csvLoad').addEventListener('click', async () => {
})
$('csvDownload').addEventListener('click', () => {
- const csv = 'peer,score\n' + state.scores.map(([p, s]) => `${p},${s}`).join('\n') + '\n'
+ const csv = 'peer,score\n' + state.scores.map(([p, s]) => `${csvField(p)},${s}`).join('\n') + '\n'
const a = el('a', { href: URL.createObjectURL(new Blob([csv], { type: 'text/csv' })), download: 'eigentrust-scores.csv' })
a.click()
setTimeout(() => URL.revokeObjectURL(a.href), 1000)
diff --git a/demo/index.html b/demo/index.html
index 8c194a9..0441441 100644
--- a/demo/index.html
+++ b/demo/index.html
@@ -94,6 +94,7 @@ Trust flows from the peers you already trust.
Ranking
+
diff --git a/demo/style.css b/demo/style.css
index 28e37c1..fd3f8fe 100644
--- a/demo/style.css
+++ b/demo/style.css
@@ -300,6 +300,7 @@ textarea {
.actions { display: flex; gap: 8px; flex-wrap: wrap; }
.error { color: var(--error); font-size: 14px; }
+#engineError { margin: 4px 0 8px; font-size: 13.5px; }
/* selected peer */
From 5ba266fbce85acf2a8705eedcdfd72b32b230db5 Mon Sep 17 00:00:00 2001
From: jenya
Date: Wed, 30 Sep 2026 17:00:59 +0300
Subject: [PATCH 4/4] License under MIT OR Apache-2.0, document input rules
---
CLAUDE.md | 18 +++--
LICENSE-APACHE | 202 +++++++++++++++++++++++++++++++++++++++++++++++++
LICENSE-MIT | 21 +++++
README.md | 12 ++-
4 files changed, 246 insertions(+), 7 deletions(-)
create mode 100644 LICENSE-APACHE
create mode 100644 LICENSE-MIT
diff --git a/CLAUDE.md b/CLAUDE.md
index c9b8c16..81c76fa 100644
--- a/CLAUDE.md
+++ b/CLAUDE.md
@@ -5,26 +5,31 @@
- NEVER add `Co-Authored-By: Claude ...`, `noreply@anthropic.com`, "Generated with Claude Code"
or any other Claude / Anthropic attribution to commit messages, PR titles or PR descriptions.
This project was written by its author before Claude was involved. Plain commit messages only.
-- This is enforced by `.githooks/commit-msg`. Enable it once per clone:
+- This is enforced by `.githooks/commit-msg` and by CI. Enable the hook once per clone:
`git config core.hooksPath .githooks`
- Never bypass it with `--no-verify`.
## Project
-Rust port of EigenTrust (go-eigentrust), runs natively and as WASM in the browser.
+EigenTrust in Rust, runs natively and as WASM in the browser.
- `src/basic/engine.rs` - `calculate_from_csv`: CSV in, sorted `(peer, score)` out
+- `src/basic/input.rs` - CSV reading (csv crate), header detection, weight validation
- `src/basic/eigentrust.rs` - power iteration (`compute`), runs on dense vectors
+- `src/error.rs` - `Error` enum used everywhere; `Display` gives user-facing messages
- `src/sparse/` - CSR matrix / sparse vector
- `src/lib.rs` - wasm-bindgen entry `run(localtrust, pretrust, alpha)` (wasm32 only)
-- `src/main.rs` - native CLI
+- `src/main.rs` - native CLI, reports errors as `error: ...` with exit code 1
- `demo/` - static interactive web demo deployed to Vercel
## Commands
- Test: `cargo test --release`
+- WASM tests: `wasm-pack test --node --release`
+- Lint: `cargo fmt --check`, `cargo clippy --release --all-targets -- -D warnings`
+ (also with `--target wasm32-unknown-unknown`)
- CLI: `cargo run --release -- ./example/localtrust.csv ./example/pretrust.csv [alpha]`
-- WASM: `./build.sh` (builds `pkg/` and copies it into `demo/pkg/`)
+- WASM: `./build.sh` (builds `pkg/` and `pkg-parallel/` and copies them into `demo/`)
- Demo locally: `python3 -m http.server -d demo` then open http://localhost:8000
- Deploy: `vercel deploy --prod` from `demo/`
@@ -32,4 +37,7 @@ Rust port of EigenTrust (go-eigentrust), runs natively and as WASM in the browse
- Tests compare floats with exact equality. The iteration uses Kahan-Babuska-Neumaier summation
in row order; keep the summation order if you touch `compute` / `mul_dense`.
-- Duplicate `(i, j)` records in local trust: the last one wins.
+- Duplicate `(i, j)` local trust records and duplicate pre-trust peers: the last one wins.
+- Weights must be finite and non-negative. Distrust (negative weights) is rejected until its
+ effect on the scores is defined; `extract_distrust` / `discount_trust_vector` are kept for that.
+- `compute` stops after `DEFAULT_MAX_ITERATIONS` (10,000) with `Error::NotConverged`.
diff --git a/LICENSE-APACHE b/LICENSE-APACHE
new file mode 100644
index 0000000..d645695
--- /dev/null
+++ b/LICENSE-APACHE
@@ -0,0 +1,202 @@
+
+ Apache License
+ Version 2.0, January 2004
+ http://www.apache.org/licenses/
+
+ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
+
+ 1. Definitions.
+
+ "License" shall mean the terms and conditions for use, reproduction,
+ and distribution as defined by Sections 1 through 9 of this document.
+
+ "Licensor" shall mean the copyright owner or entity authorized by
+ the copyright owner that is granting the License.
+
+ "Legal Entity" shall mean the union of the acting entity and all
+ other entities that control, are controlled by, or are under common
+ control with that entity. For the purposes of this definition,
+ "control" means (i) the power, direct or indirect, to cause the
+ direction or management of such entity, whether by contract or
+ otherwise, or (ii) ownership of fifty percent (50%) or more of the
+ outstanding shares, or (iii) beneficial ownership of such entity.
+
+ "You" (or "Your") shall mean an individual or Legal Entity
+ exercising permissions granted by this License.
+
+ "Source" form shall mean the preferred form for making modifications,
+ including but not limited to software source code, documentation
+ source, and configuration files.
+
+ "Object" form shall mean any form resulting from mechanical
+ transformation or translation of a Source form, including but
+ not limited to compiled object code, generated documentation,
+ and conversions to other media types.
+
+ "Work" shall mean the work of authorship, whether in Source or
+ Object form, made available under the License, as indicated by a
+ copyright notice that is included in or attached to the work
+ (an example is provided in the Appendix below).
+
+ "Derivative Works" shall mean any work, whether in Source or Object
+ form, that is based on (or derived from) the Work and for which the
+ editorial revisions, annotations, elaborations, or other modifications
+ represent, as a whole, an original work of authorship. For the purposes
+ of this License, Derivative Works shall not include works that remain
+ separable from, or merely link (or bind by name) to the interfaces of,
+ the Work and Derivative Works thereof.
+
+ "Contribution" shall mean any work of authorship, including
+ the original version of the Work and any modifications or additions
+ to that Work or Derivative Works thereof, that is intentionally
+ submitted to Licensor for inclusion in the Work by the copyright owner
+ or by an individual or Legal Entity authorized to submit on behalf of
+ the copyright owner. For the purposes of this definition, "submitted"
+ means any form of electronic, verbal, or written communication sent
+ to the Licensor or its representatives, including but not limited to
+ communication on electronic mailing lists, source code control systems,
+ and issue tracking systems that are managed by, or on behalf of, the
+ Licensor for the purpose of discussing and improving the Work, but
+ excluding communication that is conspicuously marked or otherwise
+ designated in writing by the copyright owner as "Not a Contribution."
+
+ "Contributor" shall mean Licensor and any individual or Legal Entity
+ on behalf of whom a Contribution has been received by Licensor and
+ subsequently incorporated within the Work.
+
+ 2. Grant of Copyright License. Subject to the terms and conditions of
+ this License, each Contributor hereby grants to You a perpetual,
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
+ copyright license to reproduce, prepare Derivative Works of,
+ publicly display, publicly perform, sublicense, and distribute the
+ Work and such Derivative Works in Source or Object form.
+
+ 3. Grant of Patent License. Subject to the terms and conditions of
+ this License, each Contributor hereby grants to You a perpetual,
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
+ (except as stated in this section) patent license to make, have made,
+ use, offer to sell, sell, import, and otherwise transfer the Work,
+ where such license applies only to those patent claims licensable
+ by such Contributor that are necessarily infringed by their
+ Contribution(s) alone or by combination of their Contribution(s)
+ with the Work to which such Contribution(s) was submitted. If You
+ institute patent litigation against any entity (including a
+ cross-claim or counterclaim in a lawsuit) alleging that the Work
+ or a Contribution incorporated within the Work constitutes direct
+ or contributory patent infringement, then any patent licenses
+ granted to You under this License for that Work shall terminate
+ as of the date such litigation is filed.
+
+ 4. Redistribution. You may reproduce and distribute copies of the
+ Work or Derivative Works thereof in any medium, with or without
+ modifications, and in Source or Object form, provided that You
+ meet the following conditions:
+
+ (a) You must give any other recipients of the Work or
+ Derivative Works a copy of this License; and
+
+ (b) You must cause any modified files to carry prominent notices
+ stating that You changed the files; and
+
+ (c) You must retain, in the Source form of any Derivative Works
+ that You distribute, all copyright, patent, trademark, and
+ attribution notices from the Source form of the Work,
+ excluding those notices that do not pertain to any part of
+ the Derivative Works; and
+
+ (d) If the Work includes a "NOTICE" text file as part of its
+ distribution, then any Derivative Works that You distribute must
+ include a readable copy of the attribution notices contained
+ within such NOTICE file, excluding those notices that do not
+ pertain to any part of the Derivative Works, in at least one
+ of the following places: within a NOTICE text file distributed
+ as part of the Derivative Works; within the Source form or
+ documentation, if provided along with the Derivative Works; or,
+ within a display generated by the Derivative Works, if and
+ wherever such third-party notices normally appear. The contents
+ of the NOTICE file are for informational purposes only and
+ do not modify the License. You may add Your own attribution
+ notices within Derivative Works that You distribute, alongside
+ or as an addendum to the NOTICE text from the Work, provided
+ that such additional attribution notices cannot be construed
+ as modifying the License.
+
+ You may add Your own copyright statement to Your modifications and
+ may provide additional or different license terms and conditions
+ for use, reproduction, or distribution of Your modifications, or
+ for any such Derivative Works as a whole, provided Your use,
+ reproduction, and distribution of the Work otherwise complies with
+ the conditions stated in this License.
+
+ 5. Submission of Contributions. Unless You explicitly state otherwise,
+ any Contribution intentionally submitted for inclusion in the Work
+ by You to the Licensor shall be under the terms and conditions of
+ this License, without any additional terms or conditions.
+ Notwithstanding the above, nothing herein shall supersede or modify
+ the terms of any separate license agreement you may have executed
+ with Licensor regarding such Contributions.
+
+ 6. Trademarks. This License does not grant permission to use the trade
+ names, trademarks, service marks, or product names of the Licensor,
+ except as required for reasonable and customary use in describing the
+ origin of the Work and reproducing the content of the NOTICE file.
+
+ 7. Disclaimer of Warranty. Unless required by applicable law or
+ agreed to in writing, Licensor provides the Work (and each
+ Contributor provides its Contributions) on an "AS IS" BASIS,
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
+ implied, including, without limitation, any warranties or conditions
+ of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
+ PARTICULAR PURPOSE. You are solely responsible for determining the
+ appropriateness of using or redistributing the Work and assume any
+ risks associated with Your exercise of permissions under this License.
+
+ 8. Limitation of Liability. In no event and under no legal theory,
+ whether in tort (including negligence), contract, or otherwise,
+ unless required by applicable law (such as deliberate and grossly
+ negligent acts) or agreed to in writing, shall any Contributor be
+ liable to You for damages, including any direct, indirect, special,
+ incidental, or consequential damages of any character arising as a
+ result of this License or out of the use or inability to use the
+ Work (including but not limited to damages for loss of goodwill,
+ work stoppage, computer failure or malfunction, or any and all
+ other commercial damages or losses), even if such Contributor
+ has been advised of the possibility of such damages.
+
+ 9. Accepting Warranty or Additional Liability. While redistributing
+ the Work or Derivative Works thereof, You may choose to offer,
+ and charge a fee for, acceptance of support, warranty, indemnity,
+ or other liability obligations and/or rights consistent with this
+ License. However, in accepting such obligations, You may act only
+ on Your own behalf and on Your sole responsibility, not on behalf
+ of any other Contributor, and only if You agree to indemnify,
+ defend, and hold each Contributor harmless for any liability
+ incurred by, or claims asserted against, such Contributor by reason
+ of your accepting any such warranty or additional liability.
+
+ END OF TERMS AND CONDITIONS
+
+ APPENDIX: How to apply the Apache License to your work.
+
+ To apply the Apache License to your work, attach the following
+ boilerplate notice, with the fields enclosed by brackets "[]"
+ replaced with your own identifying information. (Don't include
+ the brackets!) The text should be enclosed in the appropriate
+ comment syntax for the file format. We also recommend that a
+ file or class name and description of purpose be included on the
+ same "printed page" as the copyright notice for easier
+ identification within third-party archives.
+
+ Copyright [yyyy] [name of copyright owner]
+
+ Licensed under the Apache License, Version 2.0 (the "License");
+ you may not use this file except in compliance with the License.
+ You may obtain a copy of the License at
+
+ http://www.apache.org/licenses/LICENSE-2.0
+
+ Unless required by applicable law or agreed to in writing, software
+ distributed under the License is distributed on an "AS IS" BASIS,
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ See the License for the specific language governing permissions and
+ limitations under the License.
diff --git a/LICENSE-MIT b/LICENSE-MIT
new file mode 100644
index 0000000..c137d6b
--- /dev/null
+++ b/LICENSE-MIT
@@ -0,0 +1,21 @@
+MIT License
+
+Copyright (c) 2024 Jenya
+
+Permission is hereby granted, free of charge, to any person obtaining a copy
+of this software and associated documentation files (the "Software"), to deal
+in the Software without restriction, including without limitation the rights
+to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
+copies of the Software, and to permit persons to whom the Software is
+furnished to do so, subject to the following conditions:
+
+The above copyright notice and this permission notice shall be included in all
+copies or substantial portions of the Software.
+
+THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
+IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
+FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
+AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
+LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
+OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
+SOFTWARE.
diff --git a/README.md b/README.md
index 1bf4e58..6d1671b 100644
--- a/README.md
+++ b/README.md
@@ -10,6 +10,7 @@
+
@@ -70,8 +71,11 @@ Build with `./build.sh`. For a ready-made Web Worker, see [`demo/worker.js`](dem
| Local trust | `from,to[,weight]` | `alice,bob,2` |
| Seeds (pre-trust) | `peer[,weight]` | `alice,1` |
-- A header row is optional. Spaces, quoted fields, CRLF and a UTF-8 BOM are fine.
-- Weights default to 1. Repeated `from,to` pairs: the last line wins.
+- Standard CSV: quoted fields (`"Smith, J"`), a header row, spaces, CRLF and a UTF-8 BOM are fine.
+- Weights default to 1 and must be finite and non-negative. Negative trust (distrust) is not supported yet.
+- A repeated `from,to` pair or seed peer: the last line wins.
+- α must be in [0, 1]. At α = 0 some networks oscillate instead of converging; the engine stops after 10,000 iterations with an error.
+- Errors name the file and line, e.g. `local trust CSV, line 5: weight "NaN" must be finite`.
## Performance
@@ -93,3 +97,7 @@ cargo test --release # tests
python3 -m http.server -d demo # run the playground locally
git config core.hooksPath .githooks # once per clone
```
+
+## License
+
+Licensed under either of [Apache License 2.0](LICENSE-APACHE) or [MIT](LICENSE-MIT), at your option.