From 82e87c5f6bf48df79ce54e2235affc11a50ddaf9 Mon Sep 17 00:00:00 2001 From: jenya Date: Wed, 30 Sep 2026 17:00:59 +0300 Subject: [PATCH 1/4] Engine: reject invalid input, typed errors, real CSV parsing Correctness - duplicate pre-trust peers: the last value wins (same as local trust); before, duplicates were normalized separately and then collapsed by to_dense, so pre-trust and the scores could stop summing to 1 - Vector::to_dense adds repeated indices instead of overwriting them - weights must be finite and non-negative; NaN/inf are rejected with the line number instead of spinning forever - negative weights (distrust) are rejected until their effect on the scores is defined; the unused discount step is off the engine path - compute() caps iterations at 10,000 (alpha = 0 on a periodic graph used to loop forever) and guards against non-finite deltas - flat tail check compares only the top num_leaders peers Robustness and cleanup - Error enum with file, line and reason instead of String errors - CSV read with the csv crate: quotes, embedded commas, escaped quotes; line numbers stay correct across blank lines; same speed as before - CLI: usage message, error: ... with exit code 1, no expect/unwrap, CSV-quoted output, quiet on a closed pipe - wasm-bindgen tests for run(); start function renamed from main - removed ConvergenceChecker, duplicate canonicalize, stale wrappers and resolved todos; cargo fmt and clippy clean on native and wasm32 - crossbeam-epoch 0.9.21 (RUSTSEC-2026-0204) - Cargo.toml license field --- Cargo.lock | 224 +++++++++++++++++++++++++++++++++++---- Cargo.toml | 5 + src/basic/eigentrust.rs | 119 +++++++++------------ src/basic/engine.rs | 64 +++++++---- src/basic/input.rs | 212 ++++++++++++++++++++++++++++++++++++ src/basic/localtrust.rs | 120 ++++++++++----------- src/basic/mod.rs | 1 + src/basic/trustvector.rs | 159 ++++++++++++--------------- src/basic/util.rs | 76 ++----------- src/error.rs | 93 ++++++++++++++++ src/lib.rs | 63 +++++++++-- src/main.rs | 72 ++++++++----- src/sparse/entry.rs | 8 +- src/sparse/matrix.rs | 45 ++++---- src/sparse/util.rs | 7 +- src/sparse/vector.rs | 24 +++-- 16 files changed, 883 insertions(+), 409 deletions(-) create mode 100644 src/basic/input.rs create mode 100644 src/error.rs diff --git a/Cargo.lock b/Cargo.lock index dd6e4fd..b99bdc3 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -11,12 +11,45 @@ dependencies = [ "memchr", ] +[[package]] +name = "async-trait" +version = "0.1.92" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "82f6aeea286b8eb4dd3431a1be1b59d290ace00f5bfd8e2a159bc2a05e2c1667" +dependencies = [ + "proc-macro2", + "quote", + "syn", +] + +[[package]] +name = "autocfg" +version = "1.5.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f2032f911046de80f0a198e0901378627c33f59ea0ac00e363d481118bd70a53" + [[package]] name = "bumpalo" version = "3.16.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "79296716171880943b8470b5f8d03aa55eb2e645a4874bdbb28adb49162e012c" +[[package]] +name = "cast" +version = "0.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "37b2a672a2cb129a2e41c10b1224bb368f9f37a2b16b612598138befd7b37eb5" + +[[package]] +name = "cc" +version = "1.5.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f360145194ee8e21db5ee7f3fcd4fe52210864c75c985dae33218202c8bbe040" +dependencies = [ + "find-msvc-tools", + "shlex", +] + [[package]] name = "cfg-if" version = "1.0.0" @@ -65,9 +98,9 @@ dependencies = [ [[package]] name = "crossbeam-epoch" -version = "0.9.18" +version = "0.9.21" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5b82ac4a3c2ca9c3460964f020e1402edd5753411d7737aa39c3714ad1b5420e" +checksum = "dc74980687109a3b14c72fd458107bf0baa1da1a1a805e178d15501ba9b86d9d" dependencies = [ "crossbeam-utils", ] @@ -78,12 +111,34 @@ version = "0.8.20" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "22ec99545bb0ed0ea7bb9b8e1e9122ea386ff8a48c0922e43f36d45ab09e0e80" +[[package]] +name = "csv" +version = "1.4.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "52cd9d68cf7efc6ddfaaee42e7288d3a99d613d4b50f76ce9827ae0c6e14f938" +dependencies = [ + "csv-core", + "itoa", + "ryu", + "serde_core", +] + +[[package]] +name = "csv-core" +version = "0.1.13" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "704a3c26996a80471189265814dbc2c257598b96b8a7feae2d31ace646bb9782" +dependencies = [ + "memchr", +] + [[package]] name = "eigentrust" version = "0.1.0" dependencies = [ "console_error_panic_hook", "console_log", + "csv", "env_logger", "log", "rayon", @@ -91,6 +146,7 @@ dependencies = [ "serde_json", "wasm-bindgen", "wasm-bindgen-rayon", + "wasm-bindgen-test", ] [[package]] @@ -112,6 +168,12 @@ dependencies = [ "termcolor", ] +[[package]] +name = "find-msvc-tools" +version = "0.1.14" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "aedcfb3409746eddb02b9e19ebda1c3394f759a152e48ee875a0844d1b955484" + [[package]] name = "futures-core" version = "0.3.34" @@ -182,6 +244,12 @@ version = "0.2.158" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "d8adc4bb1803a324070e64a98ae98f38934d91957a99cfb3a43dcbc01bc56439" +[[package]] +name = "libm" +version = "0.2.16" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b6d2cec3eae94f9f509c767b45932f1ada8350c4bdb85af2fcab4a3c14807981" + [[package]] name = "log" version = "0.4.22" @@ -194,12 +262,47 @@ version = "2.7.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "78ca9ab1a0babb1e7d5695e3530886289c18cf2f87ec19a575a0abdce112e3a3" +[[package]] +name = "minicov" +version = "0.3.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4869b6a491569605d66d3952bcdf03df789e5b536e5f0cf7758a7f08a55ae24d" +dependencies = [ + "cc", + "walkdir", +] + +[[package]] +name = "nu-ansi-term" +version = "0.50.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7957b9740744892f114936ab4a57b3f487491bbeafaf8083688b16841a4240e5" +dependencies = [ + "windows-sys 0.59.0", +] + +[[package]] +name = "num-traits" +version = "0.2.19" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "071dfc062690e90b734c0b2273ce72ad0ffa95f0c74596bc250dcfd960262841" +dependencies = [ + "autocfg", + "libm", +] + [[package]] name = "once_cell" version = "1.19.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "3fdb12b2476b595f9358c5161aa467c2438859caa136dec86c26fdd2efe17b92" +[[package]] +name = "oorandom" +version = "11.1.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d6790f58c7ff633d8771f42965289203411a5e5c68388703c06e14f24770b41e" + [[package]] name = "pin-project-lite" version = "0.2.17" @@ -287,24 +390,43 @@ version = "1.0.18" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "f3cb5ba0dc43242ce17de99c180e96db90b235b8a9fdc9543c96d2209116bd9f" +[[package]] +name = "same-file" +version = "1.0.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "93fc1dc3aaa9bfed95e02e6eadabb4baf7e3078b0bd1b4d7b6b0b68378900502" +dependencies = [ + "winapi-util", +] + [[package]] name = "serde" -version = "1.0.209" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba" +dependencies = [ + "serde_core", + "serde_derive", +] + +[[package]] +name = "serde_core" +version = "1.0.229" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "99fce0ffe7310761ca6bf9faf5115afbc19688edd00171d81b1bb1b116c63e09" +checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48" dependencies = [ "serde_derive", ] [[package]] name = "serde_derive" -version = "1.0.209" +version = "1.0.229" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a5831b979fd7b5439637af1752d535ff49f4860c0f341d1baeb6faf0f4242170" +checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348" dependencies = [ "proc-macro2", "quote", - "syn 2.0.76", + "syn", ] [[package]] @@ -320,21 +442,16 @@ dependencies = [ ] [[package]] -name = "slab" -version = "0.4.12" +name = "shlex" +version = "2.0.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0c790de23124f9ab44544d7ac05d60440adc586479ce501c1d6d7da3cd8c9cf5" +checksum = "f8fadd59c855ef2080decdef8ff161eb6661b86933c9d82e5ba29dc602a55aba" [[package]] -name = "syn" -version = "2.0.76" +name = "slab" +version = "0.4.12" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "578e081a14e0cefc3279b0472138c513f37b41a08d5a3cca9b6e4e8ceb6cd525" -dependencies = [ - "proc-macro2", - "quote", - "unicode-ident", -] +checksum = "0c790de23124f9ab44544d7ac05d60440adc586479ce501c1d6d7da3cd8c9cf5" [[package]] name = "syn" @@ -356,12 +473,31 @@ dependencies = [ "winapi-util", ] +[[package]] +name = "tokio" +version = "1.53.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "202caea871b69668250d242070849eb495be178ed697a3e98aebce5bc81a0bed" +dependencies = [ + "pin-project-lite", +] + [[package]] name = "unicode-ident" version = "1.0.12" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "3354b9ac3fae1ff6755cb6db53683adb661634f67557942dea4facebec0fee4b" +[[package]] +name = "walkdir" +version = "2.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "29790946404f91d9c5d06f9874efddea1dc06c5efe94541a7d6863108e3a5e4b" +dependencies = [ + "same-file", + "winapi-util", +] + [[package]] name = "wasm-bindgen" version = "0.2.129" @@ -375,6 +511,17 @@ dependencies = [ "wasm-bindgen-shared", ] +[[package]] +name = "wasm-bindgen-futures" +version = "0.4.79" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3cbab34de2d982e9b48e18d216d04c4a6f641066ff19ffb699980f591ee3610e" +dependencies = [ + "js-sys", + "tokio", + "wasm-bindgen", +] + [[package]] name = "wasm-bindgen-macro" version = "0.2.129" @@ -394,7 +541,7 @@ dependencies = [ "bumpalo", "proc-macro2", "quote", - "syn 3.0.6", + "syn", "wasm-bindgen-shared", ] @@ -419,6 +566,45 @@ dependencies = [ "unicode-ident", ] +[[package]] +name = "wasm-bindgen-test" +version = "0.3.79" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ae7499dfd45780a0a91d7ee6bb9ac51970a4479a41a89da443fdda5a39547d42" +dependencies = [ + "async-trait", + "cast", + "js-sys", + "libm", + "minicov", + "nu-ansi-term", + "num-traits", + "oorandom", + "serde", + "serde_json", + "wasm-bindgen", + "wasm-bindgen-futures", + "wasm-bindgen-test-macro", + "wasm-bindgen-test-shared", +] + +[[package]] +name = "wasm-bindgen-test-macro" +version = "0.3.79" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3b84b5ac638bfb168196a1a461fcc8f46a294a18b1b6be52133b4e0db122cc9f" +dependencies = [ + "proc-macro2", + "quote", + "syn", +] + +[[package]] +name = "wasm-bindgen-test-shared" +version = "0.2.129" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4f692aa943ccd88363733b77063f32cfed5bc6cbea8e6e8b251b302f881606fe" + [[package]] name = "wasm_sync" version = "0.1.2" diff --git a/Cargo.toml b/Cargo.toml index b24d865..cc63a15 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -3,6 +3,7 @@ name = "eigentrust" version = "0.1.0" edition = "2021" authors = ["Jenya "] +license = "MIT OR Apache-2.0" description = "Fast EigenTrust reputation algorithm: global trust scores for peer-to-peer networks and social graphs, native and WebAssembly" homepage = "https://eigentrust.jenyadoesapps.com" readme = "README.md" @@ -22,11 +23,15 @@ console_log = { version = "1.0", features = ["color"]} wasm-bindgen-rayon = { version = "1.3", optional = true } rayon = { version = "1.8", optional = true } +[target.'cfg(target_arch = "wasm32")'.dev-dependencies] +wasm-bindgen-test = "0.3" + [target.'cfg(not(target_arch = "wasm32"))'.dependencies] env_logger = "0.10" rayon = "1.8" [dependencies] +csv = "1.3" serde = { version = "1.0", features = ["derive"] } serde_json = "1.0" log = "0.4" diff --git a/src/basic/eigentrust.rs b/src/basic/eigentrust.rs index f294706..9e3254c 100644 --- a/src/basic/eigentrust.rs +++ b/src/basic/eigentrust.rs @@ -1,3 +1,4 @@ +use crate::error::{Error, Result}; use crate::sparse::entry::Entry; #[cfg(test)] use crate::sparse::matrix::CSMatrix; @@ -6,64 +7,8 @@ use crate::sparse::util::KBNSummer; use crate::sparse::vector::Vector; use std::cmp; -// Canonicalize scales sparse entries in-place so that their values sum to one. -// If entries sum to zero, Canonicalize returns an error indicating a zero-sum vector. -pub fn canonicalize(entries: &mut [Entry]) -> Result<(), String> { - let sum: f64 = entries.iter().map(|entry| entry.value).sum(); - if sum == 0.0 { - return Err("Zero sum vector".to_string()); - } - for entry in entries.iter_mut() { - entry.value /= sum; - } - Ok(()) -} - -pub struct ConvergenceChecker { - iter: usize, - t: Vector, - d: f64, - e: f64, -} - -impl ConvergenceChecker { - pub fn new(t0: &Vector, e: f64) -> ConvergenceChecker { - ConvergenceChecker { - iter: 0, - t: t0.clone(), - d: 2.0 * e, // initial sentinel - e, - } - } - - pub fn update(&mut self, t: &Vector) -> Result<(), String> { - let mut td = Vector::new(self.t.dim, vec![]); - td.sub_vec(t, &self.t)?; - - let d = td.norm2(); - - log::debug!( - "one iteration={} log10dPace={} log10dRemaining={}", - self.iter, - (d / self.d).log10(), - (d / self.e).log10() - ); - - self.t.assign(t); - self.d = d; - self.iter += 1; - Ok(()) - } - - pub fn converged(&self) -> bool { - self.d <= self.e - } - - pub fn delta(&self) -> f64 { - self.d - } -} - +// Stops the iteration only once the ranking of the top `num_leaders` peers has stayed +// the same for `length` checks. compute() uses length 0, which disables it. pub struct FlatTailChecker { length: usize, num_leaders: usize, @@ -91,7 +36,12 @@ impl FlatTailChecker { .partial_cmp(&a.value) .unwrap_or(cmp::Ordering::Equal) }); - let ranking: Vec = entries.iter().map(|entry| entry.index).collect(); + // only the top num_leaders positions have to stay stable + let ranking: Vec = entries + .iter() + .take(self.num_leaders) + .map(|entry| entry.index) + .collect(); if ranking == self.stats.ranking { self.stats.length += 1; @@ -117,12 +67,15 @@ pub struct FlatTailStats { pub ranking: Vec, } +// Upper bound on power iterations when the caller does not set one. With alpha > 0 +// the error shrinks by (1 - alpha) per step; alpha = 0 on a periodic graph never converges. +pub const DEFAULT_MAX_ITERATIONS: usize = 10_000; + // Compute function implements the EigenTrust algorithm. // // The iteration runs on dense vectors: after a couple of iterations the trust // vector is (almost) dense anyway, and a dense lookup turns every row dot // product into O(nnz(row)) instead of a sparse-sparse merge walk. -// todo Error instead of String pub fn compute( c: &CSRMatrix, p: &Vector, @@ -130,18 +83,18 @@ pub fn compute( e: f64, max_iterations: Option, min_iterations: Option, -) -> Result { - if a.is_nan() { - return Err("Error: alpha cannot be NaN".to_string()); +) -> Result { + if !(0.0..=1.0).contains(&a) { + return Err(Error::InvalidAlpha(a)); } let n = c.cs_matrix.major_dim; if n == 0 { - return Err("Empty local trust matrix".to_string()); + return Err(Error::EmptyLocalTrust); } if p.dim != n { - return Err("Dimension mismatch".to_string()); + return Err(Error::DimensionMismatch); } let ct = c.transpose()?; @@ -156,7 +109,7 @@ pub fn compute( let mut flat_tail_checker = FlatTailChecker::new(flat_tail, n); let mut iter = 0; - let max_iters = max_iterations.unwrap_or(usize::MAX); + let max_iters = max_iterations.unwrap_or(DEFAULT_MAX_ITERATIONS); let min_iters = min_iterations.unwrap_or(1); log::info!( @@ -170,6 +123,9 @@ pub fn compute( while iter < max_iters { if iter >= min_iters { let d = dense_delta_norm2(&t1, &prev); + if !d.is_finite() { + return Err(Error::NonFiniteScores); + } prev.copy_from_slice(&t1); log::trace!("iteration={} delta={}", iter, d); @@ -195,7 +151,10 @@ pub fn compute( } if iter >= max_iters { - return Err("Reached maximum iterations without convergence".to_string()); + return Err(Error::NotConverged { + iterations: max_iters, + alpha: a, + }); } log::info!( @@ -253,7 +212,9 @@ fn mul_dense(m: &CSRMatrix, v: &[f64], out: &mut [f64]) { // Subtracts from t each peer's trust-weighted distrust row: // t -= sum_i t[i] * discounts[i], applied in distruster order. -pub fn discount_trust_vector(t: &mut Vector, discounts: &CSRMatrix) -> Result<(), String> { +// Subtracts each peer's trust-weighted distrust row. Not used by calculate_from_csv, see +// extract_distrust. +pub fn discount_trust_vector(t: &mut Vector, discounts: &CSRMatrix) -> Result<()> { if discounts.cs_matrix.entries.iter().all(|row| row.is_empty()) { return Ok(()); } @@ -268,7 +229,7 @@ pub fn discount_trust_vector(t: &mut Vector, discounts: &CSRMatrix) -> Result<() }; for entry in distrusts { if entry.index >= result.len() { - return Err("Dimension mismatch".to_string()); + return Err(Error::DimensionMismatch); } let scaled = weight * entry.value; if scaled != 0.0 { @@ -616,4 +577,24 @@ mod tests { let result = compute(&c, &p, a, e, None, None).unwrap(); assert_eq!(result, expected); } + + #[test] + fn test_alpha_zero_on_periodic_graph_stops() { + // a <-> b with all seed trust on a oscillates forever without teleport + let c = CSRMatrix::new(2, 2, vec![(0, 1, 1.0), (1, 0, 1.0)]); + let p = Vector::new(2, vec![Entry::new(0, 1.0)]); + let err = compute(&c, &p, 0.0, 1e-9, None, None).unwrap_err(); + assert!(matches!(err, Error::NotConverged { .. }), "{}", err); + // with teleport it converges + assert!(compute(&c, &p, 0.1, 1e-9, None, None).is_ok()); + } + + #[test] + fn test_compute_rejects_bad_alpha() { + let c = CSRMatrix::new(1, 1, vec![(0, 0, 1.0)]); + let p = Vector::new(1, vec![Entry::new(0, 1.0)]); + for a in [f64::NAN, -0.1, 1.5] { + assert!(compute(&c, &p, a, 1e-9, None, None).is_err()); + } + } } diff --git a/src/basic/engine.rs b/src/basic/engine.rs index 64dc512..612fda7 100644 --- a/src/basic/engine.rs +++ b/src/basic/engine.rs @@ -1,29 +1,26 @@ -use super::util::strip_headers; use crate::basic::eigentrust::compute; -use crate::basic::eigentrust::discount_trust_vector; -use crate::basic::localtrust::{canonicalize_local_trust, extract_distrust, read_local_trust_from_csv}; +use crate::basic::localtrust::{canonicalize_local_trust, read_local_trust_from_csv}; use crate::basic::trustvector::canonicalize_trust_vector; use crate::basic::trustvector::read_trust_vector_from_csv; +use crate::error::{Error, Result}; #[cfg(test)] use std::fs; -// todo array inputs +// Ranks every peer from CSV input: `from,to[,weight]` local trust and `peer[,weight]` +// pre-trust. Returns (peer, score) sorted by score, highest first; scores sum to 1. pub fn calculate_from_csv( localtrust_csv: &str, pretrust_csv: &str, alpha: Option, -) -> Result, String> { +) -> Result> { log::info!("Compute starting..."); let a = alpha.unwrap_or(0.5); if !(0.0..=1.0).contains(&a) { - return Err(format!("alpha must be in [0, 1], got {}", a)); + return Err(Error::InvalidAlpha(a)); } - let localtrust_csv = strip_headers(localtrust_csv, 2); - let pretrust_csv = strip_headers(pretrust_csv, 1); - let (mut local_trust, peers) = read_local_trust_from_csv(localtrust_csv)?; let mut pre_trust = read_trust_vector_from_csv(pretrust_csv, &peers.map)?; @@ -41,18 +38,11 @@ pub fn calculate_from_csv( canonicalize_trust_vector(&mut pre_trust); - let mut discounts = extract_distrust(&mut local_trust)?; - - canonicalize_local_trust(&mut local_trust, Some(pre_trust.clone()))?; - canonicalize_local_trust(&mut discounts, None)?; + // negative (distrust) weights are rejected while parsing, see localtrust.rs + canonicalize_local_trust(&mut local_trust, Some(&pre_trust))?; let trust_scores = compute(&local_trust, &pre_trust, a, e, None, None)?; - // todo: discounted scores are computed but not returned (same as before), - // decide whether distrust should affect the output - let mut discounted = trust_scores.clone(); - discount_trust_vector(&mut discounted, &discounts)?; - let mut entries: Vec<(String, f64)> = trust_scores .entries .iter() @@ -95,9 +85,9 @@ mod tests { #[test] fn test_calculate_from_csv_file() { - let localtrust_csv =fs::read_to_string("./example/localtrust2.csv").expect("Failed to read localtrust CSV file"); - let pretrust_csv = fs - ::read_to_string("./example/pretrust2.csv") + let localtrust_csv = fs::read_to_string("./example/localtrust2.csv") + .expect("Failed to read localtrust CSV file"); + let pretrust_csv = fs::read_to_string("./example/pretrust2.csv") .expect("Failed to read pretrust CSV file"); let entries = calculate_from_csv(&localtrust_csv, &pretrust_csv, None).unwrap(); @@ -109,4 +99,36 @@ mod tests { assert_eq!(entries[0].1, 0.40356129084997394); assert_eq!(entries[1].0, "0x9fc3b33884e1d056a8ca979833d686abd267f9f8"); } + + fn total(entries: &[(String, f64)]) -> f64 { + entries.iter().map(|(_, s)| s).sum() + } + + #[test] + fn test_scores_sum_to_one_with_duplicate_pretrust() { + let lt = "a,b,1\nb,c,1\nc,a,1\nc,d,1"; + let pt = "a,1\nd,1\na,5\na,2"; + let entries = calculate_from_csv(lt, pt, Some(0.3)).unwrap(); + assert!((total(&entries) - 1.0).abs() < 1e-9, "{}", total(&entries)); + // last value wins: a=2, d=1 is the same as a clean 2:1 pretrust + let clean = calculate_from_csv(lt, "a,2\nd,1", Some(0.3)).unwrap(); + assert_eq!(entries, clean); + } + + #[test] + fn test_rejects_bad_input() { + for (lt, pt) in [ + ("a,b,NaN", "a"), + ("a,b,inf", "a"), + ("a,b,-1", "a"), + ("a,b,1", "a,NaN"), + ] { + assert!(calculate_from_csv(lt, pt, None).is_err(), "{} / {}", lt, pt); + } + } + + #[test] + fn test_alpha_zero_periodic_returns_error() { + assert!(calculate_from_csv("a,b\nb,a", "a", Some(0.0)).is_err()); + } } diff --git a/src/basic/input.rs b/src/basic/input.rs new file mode 100644 index 0000000..2452eac --- /dev/null +++ b/src/basic/input.rs @@ -0,0 +1,212 @@ +// CSV input: a real CSV reader (quotes, embedded commas, escaped quotes), an optional +// header row, surrounding whitespace, CRLF and a UTF-8 BOM. + +use crate::error::{Error, Input, RecordError, Result}; +use csv::{ByteRecord, ReaderBuilder}; + +const HEADER_NAMES: &[&str] = &[ + "i", "j", "v", "from", "to", "value", "weight", "trust", "level", "peer", "id", "score", + "source", "target", "src", "dst", "truster", "trustee", +]; + +// Surrounding whitespace and quotes the CSV reader keeps, e.g. in ` "alice" `. +pub fn clean_field(field: &str) -> &str { + let field = field.trim(); + field + .strip_prefix('"') + .and_then(|f| f.strip_suffix('"')) + .unwrap_or(field) + .trim() +} + +// A first record is a header when its value column is not a number, or, when the value +// column is omitted (implicit weight 1), when every field is a well-known header name. +// `value_column` is 2 for local trust and 1 for pre-trust. +pub fn is_header(fields: &[&str], value_column: usize) -> bool { + match fields.get(value_column) { + Some(value) => value.parse::().is_err(), + None => fields + .iter() + .all(|f| HEADER_NAMES.contains(&f.to_ascii_lowercase().as_str())), + } +} + +// Calls `f(line, fields)` for every non-empty record after the optional header. +pub fn for_each_record(data: &str, input: Input, value_column: usize, mut f: F) -> Result<()> +where + F: FnMut(u64, &[&str]) -> Result<()>, +{ + let data = data.trim_start_matches('\u{feff}'); + let mut reader = ReaderBuilder::new() + .has_headers(false) + .flexible(true) + .from_reader(data.as_bytes()); + + // the csv crate reports a record's position before any blank lines it skipped and + // does not count those lines, so line numbers come from byte offsets + let bytes = data.as_bytes(); + let mut counted = (0usize, 1u64); + let mut line_at = |byte: u64| { + let mut byte = (byte as usize).min(bytes.len()); + while byte < bytes.len() && (bytes[byte] == b'\n' || bytes[byte] == b'\r') { + byte += 1; + } + if byte >= counted.0 { + counted.1 += bytes[counted.0..byte] + .iter() + .filter(|&&b| b == b'\n') + .count() as u64; + counted.0 = byte; + } + counted.1 + }; + + let mut record = ByteRecord::new(); + let mut first = true; + loop { + let more = match reader.read_byte_record(&mut record) { + Ok(more) => more, + Err(e) => { + return Err(Error::Record { + input, + line: e.position().map_or(0, |p| line_at(p.byte())), + error: RecordError::Malformed(e.to_string()), + }) + } + }; + if !more { + return Ok(()); + } + let line = record.position().map_or(0, |p| line_at(p.byte())); + + // at most 3 columns are used, extra columns are ignored + let mut buf = [""; 3]; + let mut n = 0; + for field in record.iter().take(buf.len()) { + // the input is a &str, but a quoted field could still split a code point + let field = std::str::from_utf8(field).map_err(|e| Error::Record { + input, + line, + error: RecordError::Malformed(e.to_string()), + })?; + buf[n] = clean_field(field); + n += 1; + } + let fields = &buf[..n]; + if fields.iter().all(|f| f.is_empty()) { + continue; + } + if first { + first = false; + if is_header(fields, value_column) { + continue; + } + } + f(line, fields)?; + } +} + +// A trust weight: finite and non-negative, 1 when omitted. +pub fn parse_weight(field: Option<&&str>) -> std::result::Result { + let Some(&field) = field else { + return Ok(1.0); + }; + let weight = field + .parse::() + .map_err(|_| RecordError::InvalidWeight(field.to_string()))?; + if !weight.is_finite() { + return Err(RecordError::NonFiniteWeight(field.to_string())); + } + if weight < 0.0 { + return Err(RecordError::NegativeWeight(field.to_string())); + } + Ok(weight) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn records(data: &str, value_column: usize) -> Result)>> { + let mut out = vec![]; + for_each_record(data, Input::LocalTrust, value_column, |line, f| { + out.push((line, f.iter().map(|s| s.to_string()).collect())); + Ok(()) + })?; + Ok(out) + } + + fn rows(data: &str, value_column: usize) -> Vec> { + records(data, value_column) + .unwrap() + .into_iter() + .map(|(_, f)| f) + .collect() + } + + #[test] + fn test_headers() { + assert_eq!(rows("i,j,v\na,b,1\n", 2), [["a", "b", "1"]]); + assert_eq!(rows("i,j,v\r\na,b,1\r\n", 2), [["a", "b", "1"]]); + assert_eq!(rows("\u{feff}from,to,weight\na,b,1", 2), [["a", "b", "1"]]); + // headerless two-column local trust keeps its first edge + assert_eq!(rows("a,b\nb,c", 2), [["a", "b"], ["b", "c"]]); + assert_eq!(rows("i,j\na,b", 2), [["a", "b"]]); + assert_eq!(rows("i,v\nalice,1", 1), [["alice", "1"]]); + assert_eq!(rows("alice\nbob", 1), [["alice"], ["bob"]]); + assert_eq!(rows("peer\nalice", 1), [["alice"]]); + assert!(rows("", 1).is_empty()); + } + + #[test] + fn test_real_csv() { + // quoted fields with commas and escaped quotes, blank lines, spaces + let data = "\"Smith, J\",\"O\"\"Neil\",2\n\n a , b , 3 \n \"c\" , \"d\" ,1\n"; + assert_eq!( + rows(data, 2), + [ + ["Smith, J", "O\"Neil", "2"], + ["a", "b", "3"], + ["c", "d", "1"] + ] + ); + } + + #[test] + fn test_line_numbers() { + let lines: Vec = records("i,j,v\na,b,1\n\nc,d,2\r\n\r\n\r\ne,f", 2) + .unwrap() + .into_iter() + .map(|(l, _)| l) + .collect(); + assert_eq!(lines, [2, 4, 7]); + } + + #[test] + fn test_parse_weight() { + assert_eq!(parse_weight(None), Ok(1.0)); + assert_eq!(parse_weight(Some(&"2.5")), Ok(2.5)); + assert!(matches!( + parse_weight(Some(&"x")), + Err(RecordError::InvalidWeight(_)) + )); + for bad in ["NaN", "inf", "-inf"] { + assert!(matches!( + parse_weight(Some(&bad)), + Err(RecordError::NonFiniteWeight(_)) + )); + } + assert!(matches!( + parse_weight(Some(&"-1")), + Err(RecordError::NegativeWeight(_)) + )); + } + + #[test] + fn test_clean_field() { + assert_eq!(clean_field(" alice "), "alice"); + assert_eq!(clean_field("\"alice\""), "alice"); + assert_eq!(clean_field(" \" 0.5 \" "), "0.5"); + assert_eq!(clean_field("\""), "\""); + } +} diff --git a/src/basic/localtrust.rs b/src/basic/localtrust.rs index cbee4a5..9008232 100644 --- a/src/basic/localtrust.rs +++ b/src/basic/localtrust.rs @@ -1,113 +1,97 @@ -use super::util::{clean_field, PeersMap}; +use super::input::{for_each_record, parse_weight}; +use super::util::PeersMap; +use crate::error::{Error, Input, RecordError, Result}; use crate::sparse::entry::Entry; use crate::sparse::matrix::CSRMatrix; use crate::sparse::vector::Vector; +// Scales every row to sum to one. Rows without trust (dangling peers) get the pre-trust +// distribution, so their share flows back to the seeds. pub fn canonicalize_local_trust( local_trust: &mut CSRMatrix, - pre_trust: Option, -) -> Result<(), String> { + pre_trust: Option<&Vector>, +) -> Result<()> { let n = local_trust.dims().0; - if let Some(ref pre_trust_vec) = pre_trust { - // if pre_trust_vec.entries.len() != n { - if pre_trust_vec.entries.len() > n { - return Err("Dimension mismatch".to_string()); + if let Some(pre_trust) = pre_trust { + if pre_trust.entries.len() > n { + return Err(Error::DimensionMismatch); } } - for i in 0..n { - let mut in_row = local_trust.row_vector(i); - let row_sum: f64 = in_row.entries.iter().map(|entry| entry.value).sum(); - + for row in local_trust.cs_matrix.entries.iter_mut() { + let row_sum: f64 = row.iter().map(|entry| entry.value).sum(); if row_sum == 0.0 { - if let Some(ref pre_trust_vec) = pre_trust { - local_trust.set_row_vector(i, Vector::new(n, pre_trust_vec.entries.clone())); + if let Some(pre_trust) = pre_trust { + row.clone_from(&pre_trust.entries); } } else { - for entry in &mut in_row.entries { + for entry in row.iter_mut() { entry.value /= row_sum; } - local_trust.set_row_vector(i, in_row); } } Ok(()) } -pub fn extract_distrust(local_trust: &mut CSRMatrix) -> Result { +// Splits negative entries off into a separate distrust matrix (as positive values). +// Not used by calculate_from_csv: negative weights are rejected while parsing until the +// effect of distrust on the scores is defined. +pub fn extract_distrust(local_trust: &mut CSRMatrix) -> CSRMatrix { let n = local_trust.dims().0; let mut distrust = CSRMatrix::new(n, n, vec![]); - for truster in 0..n { - let mut trust_row = local_trust.row_vector(truster); + for (truster, row) in local_trust.cs_matrix.entries.iter_mut().enumerate() { let mut distrust_row = Vec::new(); - - trust_row.entries.retain(|entry| { + row.retain(|entry| { if entry.value >= 0.0 { true } else { - distrust_row.push(Entry { - index: entry.index, - value: -entry.value, - }); + distrust_row.push(Entry::new(entry.index, -entry.value)); false } }); - - local_trust.set_row_vector(truster, trust_row); distrust.set_row_vector(truster, Vector::new(n, distrust_row)); } - Ok(distrust) + distrust } -fn parse_csv_line(line: &str, peer_indices: &mut PeersMap) -> Result<(usize, usize, f64), String> { - let mut fields = line.split(',').map(clean_field); - - let (from, to) = match (fields.next(), fields.next()) { - (Some(from), Some(to)) => (from, to), - _ => return Err("Too few fields".to_string()), - }; - let from = peer_indices.insert_or_get(from); - let to = peer_indices.insert_or_get(to); - let level = match fields.next() { - Some(level) => level.parse::().map_err(|_| "Invalid trust level")?, - None => 1.0, - }; - Ok((from, to, level)) -} - -// todo move csv logic out of this scope, cooentry -pub fn read_local_trust_from_csv(csv_data: &str) -> Result<(CSRMatrix, PeersMap), String> { +// Reads `from,to[,weight]` records into a square CSR matrix, one row per truster. +// Peers get indices in order of first appearance. A repeated `from,to` pair keeps the +// last weight. +pub fn read_local_trust_from_csv(csv_data: &str) -> Result<(CSRMatrix, PeersMap)> { // rows are filled directly while parsing; the matrix grows as new peers appear let mut rows: Vec> = Vec::new(); - let mut peer_indices = PeersMap::new(); - - for (count, line) in csv_data.lines().enumerate() { - if line.trim().is_empty() { - continue; - } - let (from, to, level) = parse_csv_line(line, &mut peer_indices).map_err(|e| { - format!( - "Cannot parse local trust CSV record #{}: {:?} {:?}", - count + 1, - e, - line - ) - })?; + let mut peers = PeersMap::new(); + + for_each_record(csv_data, Input::LocalTrust, 2, |line, fields| { + let record_error = |error| Error::Record { + input: Input::LocalTrust, + line, + error, + }; + let (from, to) = match fields { + [from, to, ..] => (*from, *to), + _ => return Err(record_error(RecordError::TooFewFields)), + }; + let level = parse_weight(fields.get(2)).map_err(record_error)?; + let from = peers.insert_or_get(from); + let to = peers.insert_or_get(to); if from >= rows.len() { rows.resize_with(from + 1, Vec::new); } rows[from].push(Entry::new(to, level)); - } + Ok(()) + })?; - let dim = peer_indices.names.len(); + let dim = peers.names.len(); if dim == 0 { - return Err("Local trust is empty".to_string()); + return Err(Error::EmptyLocalTrust); } rows.resize_with(dim, Vec::new); - Ok((CSRMatrix::from_rows(dim, rows), peer_indices)) + Ok((CSRMatrix::from_rows(dim, rows), peers)) } #[cfg(test)] @@ -140,7 +124,7 @@ mod tests { for test in test_cases { let mut local_trust = test.local_trust.clone(); - let distrust = extract_distrust(&mut local_trust).expect("Failed to extract distrust"); + let distrust = extract_distrust(&mut local_trust); assert_eq!( local_trust, test.expected_trust, @@ -154,4 +138,12 @@ mod tests { ); } } + + #[test] + fn test_local_trust_rejects_bad_levels() { + for bad in ["a,b,NaN", "a,b,inf", "a,b,-inf", "a,b,-1", "a,b,x"] { + assert!(read_local_trust_from_csv(bad).is_err(), "{}", bad); + } + assert!(read_local_trust_from_csv("a,b,0\nb,a,2.5").is_ok()); + } } diff --git a/src/basic/mod.rs b/src/basic/mod.rs index 6e07e00..c33416d 100644 --- a/src/basic/mod.rs +++ b/src/basic/mod.rs @@ -1,5 +1,6 @@ pub mod eigentrust; pub mod engine; +pub mod input; pub mod localtrust; pub mod trustvector; pub mod util; diff --git a/src/basic/trustvector.rs b/src/basic/trustvector.rs index e4a975b..a6be60e 100644 --- a/src/basic/trustvector.rs +++ b/src/basic/trustvector.rs @@ -1,13 +1,14 @@ -use super::util::clean_field; +use super::input::{for_each_record, parse_weight}; +use crate::error::{Error, Input, RecordError, Result}; use crate::sparse::entry::Entry; use crate::sparse::vector::Vector; -use std::collections::{HashMap, HashSet}; +use std::collections::HashMap; // CanonicalizeTrustVector canonicalizes the trust vector in-place, // scaling it so that the elements sum to one, // or making it a uniform vector that sums to one if it's a zero vector. pub fn canonicalize_trust_vector(v: &mut Vector) { - if canonicalize(&mut v.entries).is_err() { + if !canonicalize(&mut v.entries) { let dim = v.dim; let c = 1.0 / dim as f64; v.entries.clear(); @@ -17,118 +18,96 @@ pub fn canonicalize_trust_vector(v: &mut Vector) { } } -// Helper function to canonicalize a vector in-place. -// Returns an error if the vector is a zero vector. -fn canonicalize(entries: &mut Vec) -> Result<(), &'static str> { +// Scales entries in place to sum to one. Returns false for a zero vector. +fn canonicalize(entries: &mut [Entry]) -> bool { let sum: f64 = entries.iter().map(|entry| entry.value).sum(); - if sum == 0.0 { - return Err("Zero sum vector"); + return false; } - for entry in entries.iter_mut() { entry.value /= sum; } - - Ok(()) + true } -enum DuplicateHandling { - Allow, - Remove, - Fail, -} - -// todo move csv logic out of this scope +// Reads `peer[,weight]` records. Weights must be finite and non-negative, default 1. +// A peer listed more than once keeps its last weight, same as local trust. pub fn read_trust_vector_from_csv( input: &str, peer_indices: &HashMap, -) -> Result { - let mut count = 0; +) -> Result { + let mut levels: HashMap = HashMap::new(); let mut max_peer = -1; - let mut entries = Vec::new(); - let mut seen_peers = HashSet::new(); - let duplicate_handling = DuplicateHandling::Allow; - let mut dublicate_count = 0; - - for line in input.lines() { - count += 1; - if line.trim().is_empty() { - continue; - } - let fields: Vec<&str> = line.split(',').map(clean_field).collect(); + let mut duplicate_count = 0; - let (peer, level) = match fields.len() { - 0 => return Err(format!("Too few fields in line {}", count)), - _ => { - let peer = parse_peer_id(fields[0], peer_indices).map_err(|e| { - format!("Invalid peer {:?} in line {}: {}", fields[0], count, e) - })?; - let level = if fields.len() >= 2 { - parse_trust_level(fields[1]).map_err(|e| { - format!( - "Invalid trust level {:?} in line {}: {}", - fields[1], count, e - ) - })? - } else { - 1.0 - }; - (peer, level) - } + for_each_record(input, Input::PreTrust, 1, |line, fields| { + let record_error = |error| Error::Record { + input: Input::PreTrust, + line, + error, }; + let peer = *peer_indices + .get(fields[0]) + .ok_or_else(|| record_error(RecordError::UnknownPeer(fields[0].to_string())))?; + let level = parse_weight(fields.get(1)).map_err(record_error)?; - if seen_peers.contains(&peer) { - match duplicate_handling { - DuplicateHandling::Fail => { - return Err(format!("Duplicate peer {:?} in line {}", fields[0], count)); - } - DuplicateHandling::Remove => { - dublicate_count += 1; - continue; - } - DuplicateHandling::Allow => { - dublicate_count += 1; - } - } - } else { - seen_peers.insert(peer); + if levels.insert(peer, level).is_some() { + duplicate_count += 1; } - - if max_peer < peer as isize { - max_peer = peer as isize; - } - - entries.push(Entry { - index: peer, - value: level, - }); - } - - if dublicate_count > 0 { - log::warn!("Pretrust contains {} duplicate peers", dublicate_count); + max_peer = max_peer.max(peer as isize); + Ok(()) + })?; + + if duplicate_count > 0 { + log::warn!( + "Pretrust contains {} duplicate peers, the last value wins", + duplicate_count + ); } + let entries = levels + .into_iter() + .map(|(index, value)| Entry { index, value }) + .collect(); Ok(Vector::new((max_peer + 1) as usize, entries)) } -fn parse_peer_id(peer_str: &str, peer_indices: &HashMap) -> Result { - peer_indices - .get(peer_str) - .cloned() - .ok_or_else(|| format!("Invalid peer: {}", peer_str)) -} - -fn parse_trust_level(level_str: &str) -> Result { - level_str - .parse::() - .map_err(|_| format!("Invalid trust level: {}", level_str)) -} - #[cfg(test)] mod tests { use super::*; + fn peers(names: &[&str]) -> HashMap { + names + .iter() + .enumerate() + .map(|(i, n)| (n.to_string(), i)) + .collect() + } + + #[test] + fn test_duplicate_pretrust_last_wins() { + let v = read_trust_vector_from_csv("a,1\nb,1\na,3", &peers(&["a", "b"])).unwrap(); + assert_eq!(v.entries, vec![Entry::new(0, 3.0), Entry::new(1, 1.0)]); + } + + #[test] + fn test_pretrust_rejects_non_finite_and_negative() { + // a bad first line reads as a header, so each bad line follows a valid one + for bad in [ + "a,1\na,NaN", + "a,1\na,inf", + "a,1\na,-1", + "a,1\na,x", + "a,1\nb,1", + ] { + assert!( + read_trust_vector_from_csv(bad, &peers(&["a"])).is_err(), + "{}", + bad + ); + } + } + #[test] fn test_canonicalize_zero_vector_is_uniform_over_dim() { let mut v = Vector::new(4, vec![]); diff --git a/src/basic/util.rs b/src/basic/util.rs index 276d859..a898196 100644 --- a/src/basic/util.rs +++ b/src/basic/util.rs @@ -23,6 +23,12 @@ pub struct PeersMap { pub names: Vec, } +impl Default for PeersMap { + fn default() -> Self { + Self::new() + } +} + impl PeersMap { pub fn new() -> Self { PeersMap { @@ -46,73 +52,3 @@ impl PeersMap { self.names.len() } } - -// Cleans a raw CSV field: surrounding whitespace and quotes. -pub fn clean_field(field: &str) -> &str { - let field = field.trim(); - field - .strip_prefix('"') - .and_then(|f| f.strip_suffix('"')) - .unwrap_or(field) - .trim() -} - -const HEADER_NAMES: &[&str] = &[ - "i", "j", "v", "from", "to", "value", "weight", "trust", "level", "peer", "id", "score", - "source", "target", "src", "dst", "truster", "trustee", -]; - -// Strips a leading UTF-8 BOM and a header line, if present. -// `value_column` is the index of the numeric column (2 for local trust, 1 for pre-trust). -// A first line is a header when its value column is not a number, or, when the value column -// is omitted (implicit weight 1), when every field is a well-known header name. -pub fn strip_headers(csv_content: &str, value_column: usize) -> &str { - let csv_content = csv_content.trim_start_matches('\u{feff}'); - let (first_line, rest) = match csv_content.find('\n') { - Some(i) => (&csv_content[..i], &csv_content[i + 1..]), - None => (csv_content, ""), - }; - - let fields: Vec<&str> = first_line.split(',').map(clean_field).collect(); - let is_header = match fields.get(value_column) { - Some(value) => value.parse::().is_err(), - None => fields - .iter() - .all(|f| HEADER_NAMES.contains(&f.to_ascii_lowercase().as_str())), - }; - - if is_header { - rest - } else { - csv_content - } -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn test_strip_headers() { - assert_eq!(strip_headers("i,j,v\na,b,1\n", 2), "a,b,1\n"); - assert_eq!(strip_headers("i,j,v\r\na,b,1\r\n", 2), "a,b,1\r\n"); - assert_eq!(strip_headers("\u{feff}from,to,weight\na,b,1", 2), "a,b,1"); - // headerless two-column local trust keeps its first edge - assert_eq!(strip_headers("a,b\nb,c", 2), "a,b\nb,c"); - assert_eq!(strip_headers("i,j\na,b", 2), "a,b"); - assert_eq!(strip_headers("a,b,1", 2), "a,b,1"); - assert_eq!(strip_headers("i,v\nalice,1", 1), "alice,1"); - assert_eq!(strip_headers("alice,1\nbob,1", 1), "alice,1\nbob,1"); - assert_eq!(strip_headers("alice\nbob", 1), "alice\nbob"); - assert_eq!(strip_headers("peer\nalice", 1), "alice"); - assert_eq!(strip_headers("", 1), ""); - } - - #[test] - fn test_clean_field() { - assert_eq!(clean_field(" alice "), "alice"); - assert_eq!(clean_field("\"alice\""), "alice"); - assert_eq!(clean_field(" \" 0.5 \" "), "0.5"); - assert_eq!(clean_field("\""), "\""); - } -} diff --git a/src/error.rs b/src/error.rs new file mode 100644 index 0000000..64c82a1 --- /dev/null +++ b/src/error.rs @@ -0,0 +1,93 @@ +use std::fmt; + +/// Which CSV input an error refers to. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum Input { + LocalTrust, + PreTrust, +} + +impl fmt::Display for Input { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.write_str(match self { + Input::LocalTrust => "local trust", + Input::PreTrust => "pre-trust", + }) + } +} + +/// What is wrong with a single CSV record. +#[derive(Debug, Clone, PartialEq)] +pub enum RecordError { + /// The CSV itself is malformed, e.g. an unterminated quote. + Malformed(String), + TooFewFields, + InvalidWeight(String), + NonFiniteWeight(String), + /// Distrust has no defined effect on the scores yet, so it is rejected. + NegativeWeight(String), + /// A pre-trust peer that never appears in local trust. + UnknownPeer(String), +} + +impl fmt::Display for RecordError { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + match self { + RecordError::Malformed(msg) => write!(f, "malformed CSV: {}", msg), + RecordError::TooFewFields => f.write_str("too few fields"), + RecordError::InvalidWeight(w) => write!(f, "weight {:?} is not a number", w), + RecordError::NonFiniteWeight(w) => write!(f, "weight {:?} must be finite", w), + RecordError::NegativeWeight(w) => { + write!(f, "weight {:?} is negative; distrust is not supported", w) + } + RecordError::UnknownPeer(p) => { + write!(f, "peer {:?} does not appear in local trust", p) + } + } + } +} + +#[derive(Debug, Clone, PartialEq)] +pub enum Error { + /// A CSV record could not be used. `line` is 1-based. + Record { + input: Input, + line: u64, + error: RecordError, + }, + EmptyLocalTrust, + InvalidAlpha(f64), + DimensionMismatch, + /// The iteration produced NaN or infinity. + NonFiniteScores, + NotConverged { + iterations: usize, + alpha: f64, + }, + InvalidUtf8(Input), +} + +impl fmt::Display for Error { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + match self { + Error::Record { input, line, error } => { + write!(f, "{} CSV, line {}: {}", input, line, error) + } + Error::EmptyLocalTrust => f.write_str("local trust is empty"), + Error::InvalidAlpha(a) => write!(f, "alpha must be in [0, 1], got {}", a), + Error::DimensionMismatch => f.write_str("dimension mismatch"), + Error::NonFiniteScores => f.write_str("trust scores are not finite"), + Error::NotConverged { iterations, alpha } => write!( + f, + "did not converge in {} iterations with alpha {}; \ + a small alpha on a periodic trust graph can oscillate, try a larger alpha", + iterations, alpha + ), + Error::InvalidUtf8(input) => write!(f, "{} CSV is not valid UTF-8", input), + } + } +} + +impl std::error::Error for Error {} + +pub type Result = std::result::Result; diff --git a/src/lib.rs b/src/lib.rs index 818a009..fefed8f 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -1,8 +1,10 @@ #![cfg(target_arch = "wasm32")] use crate::basic::engine::calculate_from_csv; +use crate::error::{Error, Input}; use wasm_bindgen::prelude::*; pub mod basic; +pub mod error; pub mod sparse; use crate::basic::util::init_logger; use std::panic; @@ -12,7 +14,7 @@ use std::str; pub use wasm_bindgen_rayon::init_thread_pool; #[wasm_bindgen(start)] -fn main() { +fn start() { panic::set_hook(Box::new(console_error_panic_hook::hook)); init_logger(); log::debug!("WASM Eigentrust connected"); @@ -21,10 +23,59 @@ fn main() { #[wasm_bindgen] // Returns JSON: {"Ok": [[peer, score], ...]} sorted by score, or {"Err": "message"} pub fn run(localtrust_csv: &[u8], pretrust_csv: &[u8], alpha: f64) -> String { - let result = match (str::from_utf8(localtrust_csv), str::from_utf8(pretrust_csv)) { - (Ok(lt), Ok(pt)) => calculate_from_csv(lt, pt, Some(alpha)), - _ => Err("CSV input is not valid UTF-8".to_string()), - }; + let result = str::from_utf8(localtrust_csv) + .map_err(|_| Error::InvalidUtf8(Input::LocalTrust)) + .and_then(|lt| { + let pt = + str::from_utf8(pretrust_csv).map_err(|_| Error::InvalidUtf8(Input::PreTrust))?; + calculate_from_csv(lt, pt, Some(alpha)) + }) + // the JS side gets the message: {"Err": "local trust CSV, line 3: ..."} + .map_err(|e| e.to_string()); - serde_json::to_string(&result).unwrap_or_else(|e| format!("{{\"Err\":\"{}\"}}", e)) + serde_json::to_string(&result) + .unwrap_or_else(|e| serde_json::json!({ "Err": e.to_string() }).to_string()) +} + +#[cfg(test)] +mod tests { + use super::run; + use wasm_bindgen_test::wasm_bindgen_test; + + fn scores(json: &str) -> Vec<(String, f64)> { + let v: serde_json::Value = serde_json::from_str(json).unwrap(); + v["Ok"] + .as_array() + .expect("Ok result") + .iter() + .map(|e| (e[0].as_str().unwrap().to_string(), e[1].as_f64().unwrap())) + .collect() + } + + #[wasm_bindgen_test] + fn run_returns_sorted_scores_summing_to_one() { + let out = run(b"alice,bob,2\nbob,carol,1\n", b"alice\n", 0.5); + let s = scores(&out); + assert_eq!( + s.iter().map(|(p, _)| p.as_str()).collect::>(), + ["alice", "bob", "carol"] + ); + let total: f64 = s.iter().map(|(_, v)| v).sum(); + assert!((total - 1.0).abs() < 1e-9); + } + + #[wasm_bindgen_test] + fn run_reports_errors_as_err() { + for (lt, pt) in [(&b"a,b,NaN"[..], &b"a"[..]), (b"a,b,-1", b"a"), (b"", b"")] { + let out = run(lt, pt, 0.5); + assert!(out.starts_with("{\"Err\":"), "{}", out); + } + assert!(run(&[0xff, 0xfe], b"a", 0.5).starts_with("{\"Err\":")); + } + + #[wasm_bindgen_test] + fn run_rejects_bad_alpha() { + assert!(run(b"a,b", b"a", f64::NAN).starts_with("{\"Err\":")); + assert!(run(b"a,b", b"a", 2.0).starts_with("{\"Err\":")); + } } diff --git a/src/main.rs b/src/main.rs index 13b9dca..fcce6d2 100644 --- a/src/main.rs +++ b/src/main.rs @@ -1,44 +1,62 @@ use std::env; use std::fs; -use std::io::{self, BufWriter, Write}; -use std::process; +use std::io; +use std::process::ExitCode; use crate::basic::engine::calculate_from_csv; use crate::basic::util::init_logger; pub mod basic; +pub mod error; pub mod sparse; -fn main() { - let args: Vec = env::args().collect(); - init_logger(); +const USAGE: &str = "usage: eigentrust [alpha]"; - if args.len() < 3 { - log::error!( - "Usage: {} [alpha]", - args[0] - ); - process::exit(1); +fn main() -> ExitCode { + init_logger(); + match run(env::args().skip(1).collect()) { + Ok(()) => ExitCode::SUCCESS, + Err(message) => { + eprintln!("error: {}", message); + ExitCode::FAILURE + } } +} + +fn run(args: Vec) -> Result<(), String> { + let (localtrust_path, pretrust_path, alpha) = match args.as_slice() { + [lt, pt] => (lt, pt, None), + [lt, pt, alpha] => { + let alpha = alpha + .parse::() + .map_err(|_| format!("alpha {:?} is not a number\n{}", alpha, USAGE))?; + (lt, pt, Some(alpha)) + } + _ => return Err(USAGE.to_string()), + }; - let localtrust_csv_path = &args[1]; - let pretrust_csv_path = &args[2]; - let alpha = args.get(3).map(|a| a.parse::().expect("alpha must be a number")); + let read = |path: &String| { + fs::read_to_string(path).map_err(|e| format!("cannot read {}: {}", path, e)) + }; + let localtrust_csv = read(localtrust_path)?; + let pretrust_csv = read(pretrust_path)?; - let localtrust_csv = - fs::read_to_string(localtrust_csv_path).expect("Failed to read localtrust CSV file"); - let pretrust_csv = - fs::read_to_string(pretrust_csv_path).expect("Failed to read pretrust CSV file"); + let result = + calculate_from_csv(&localtrust_csv, &pretrust_csv, alpha).map_err(|e| e.to_string())?; - let result = match calculate_from_csv(&localtrust_csv, &pretrust_csv, alpha) { - Ok(result) => result, - Err(e) => { - log::error!("{}", e); - process::exit(1); + // proper CSV so peer ids with commas or quotes round-trip + let write = || -> csv::Result<()> { + let mut out = csv::Writer::from_writer(io::stdout().lock()); + for (name, score) in &result { + out.write_record([name.as_str(), &score.to_string()])?; } + out.flush()?; + Ok(()) }; - - let mut out = BufWriter::new(io::stdout().lock()); - for (name, score) in &result { - writeln!(out, "{},{}", name, score).unwrap(); + match write() { + // stdout closed early, e.g. `| head` + Err(e) if matches!(e.kind(), csv::ErrorKind::Io(io) if io.kind() == io::ErrorKind::BrokenPipe) => { + Ok(()) + } + other => other.map_err(|e| format!("cannot write output: {}", e)), } } diff --git a/src/sparse/entry.rs b/src/sparse/entry.rs index e16a88e..45d0bae 100644 --- a/src/sparse/entry.rs +++ b/src/sparse/entry.rs @@ -88,7 +88,7 @@ mod tests { ("Empty", vec![], 0), ]; - for (name, mut entries, expected_len) in tests { + for (name, entries, expected_len) in tests { let len = entries.len(); assert_eq!( len, expected_len, @@ -179,7 +179,7 @@ mod tests { ]; for (name, x, y, expected) in tests { - let entries = vec![x.clone(), y.clone()]; + let entries = [x.clone(), y.clone()]; let result = entries[0].row < entries[1].row || (entries[0].row == entries[1].row && entries[0].column < entries[1].column); assert_eq!( @@ -206,7 +206,7 @@ mod tests { ("Empty", vec![], 0), ]; - for (name, mut entries, expected_len) in tests { + for (name, entries, expected_len) in tests { let len = entries.len(); assert_eq!( len, expected_len, @@ -297,7 +297,7 @@ mod tests { ]; for (name, x, y, expected) in tests { - let entries = vec![x.clone(), y.clone()]; + let entries = [x.clone(), y.clone()]; let result = entries[0].column < entries[1].column || (entries[0].column == entries[1].column && entries[0].row < entries[1].row); assert_eq!( diff --git a/src/sparse/matrix.rs b/src/sparse/matrix.rs index 5ac000c..b77253b 100644 --- a/src/sparse/matrix.rs +++ b/src/sparse/matrix.rs @@ -1,6 +1,6 @@ use super::entry::Entry; use super::vector::Vector; - +use crate::error::{Error, Result}; #[derive(Clone, PartialEq, Debug)] pub struct CSMatrix { @@ -9,6 +9,12 @@ pub struct CSMatrix { pub entries: Vec>, } +impl Default for CSMatrix { + fn default() -> Self { + Self::new() + } +} + impl CSMatrix { pub fn new() -> Self { Self { @@ -24,9 +30,9 @@ impl CSMatrix { self.entries.clear(); } - pub fn dim(&self) -> Result { + pub fn dim(&self) -> Result { if self.major_dim != self.minor_dim { - return Err("Dimension mismatch"); + return Err(Error::DimensionMismatch); } Ok(self.major_dim) } @@ -34,7 +40,7 @@ impl CSMatrix { pub fn set_major_dim(&mut self, dim: usize) { if self.entries.capacity() < dim { let mut new_entries = Vec::with_capacity(dim); - new_entries.extend(self.entries.drain(..)); + new_entries.append(&mut self.entries); self.entries = new_entries; } self.entries.resize_with(dim, Vec::new); @@ -52,7 +58,7 @@ impl CSMatrix { self.entries.iter().map(|row| row.len()).sum() } - pub fn transpose(&self) -> Result { + pub fn transpose(&self) -> Result { let mut nnzs = vec![0; self.minor_dim]; for row_entries in &self.entries { for entry in row_entries { @@ -92,7 +98,6 @@ impl CSMatrix { other.reset(); } } -//-- fn merge_span(s1: &[Entry], s2: &[Entry]) -> Vec { let mut s = Vec::with_capacity(s1.len() + s2.len()); let mut i1 = 0; @@ -193,13 +198,12 @@ impl CSRMatrix { self.cs_matrix.entries[index] = vector.entries; } - pub fn transpose(&self) -> Result { + pub fn transpose(&self) -> Result { let transposed = self.cs_matrix.transpose()?; Ok(CSRMatrix { cs_matrix: transposed, }) } - //-- pub fn transpose_to_csc(&self) -> CSCMatrix { CSCMatrix { cs_matrix: CSMatrix { @@ -226,7 +230,6 @@ impl CSCMatrix { self.cs_matrix.set_minor_dim(rows); } - // - pub fn column_vector(&self, index: usize) -> Vector { Vector { dim: self.cs_matrix.minor_dim, @@ -234,15 +237,13 @@ impl CSCMatrix { } } - pub fn transpose(&self) -> Result { + pub fn transpose(&self) -> Result { let transposed = self.cs_matrix.transpose()?; Ok(CSCMatrix { cs_matrix: transposed, }) } - - // - pub fn transpose_to_csr(&self) -> CSRMatrix { CSRMatrix { cs_matrix: CSMatrix { @@ -254,20 +255,6 @@ impl CSCMatrix { } } -// todo cooentry -//-- -pub fn create_csr_matrix(rows: usize, cols: usize, entries: Vec<(usize, usize, f64)>) -> CSRMatrix { - CSRMatrix::new(rows, cols, entries) -} -//-- -pub fn transpose_csr_matrix(matrix: &CSRMatrix) -> Result { - matrix.transpose() -} -//- -pub fn transpose_to_csc(matrix: &CSRMatrix) -> CSCMatrix { - matrix.transpose_to_csc() -} - #[cfg(test)] mod tests { use super::*; @@ -512,7 +499,11 @@ mod tests { #[test] fn test_new_csr_matrix_duplicates_last_wins() { - let m = CSRMatrix::new(2, 2, vec![(0, 1, 1.0), (0, 0, 3.0), (0, 1, 5.0), (1, 1, 2.0)]); + let m = CSRMatrix::new( + 2, + 2, + vec![(0, 1, 1.0), (0, 0, 3.0), (0, 1, 5.0), (1, 1, 2.0)], + ); assert_eq!( m.cs_matrix.entries, vec![ diff --git a/src/sparse/util.rs b/src/sparse/util.rs index f24b06a..26c2f63 100644 --- a/src/sparse/util.rs +++ b/src/sparse/util.rs @@ -1,4 +1,3 @@ - pub fn nil_if_empty(slice: Vec) -> Option> { if slice.is_empty() { None @@ -20,6 +19,12 @@ pub struct KBNSummer { compensation: f64, } +impl Default for KBNSummer { + fn default() -> Self { + Self::new() + } +} + impl KBNSummer { pub fn new() -> Self { Self { diff --git a/src/sparse/vector.rs b/src/sparse/vector.rs index 76bcb5d..ce4ea5d 100644 --- a/src/sparse/vector.rs +++ b/src/sparse/vector.rs @@ -6,6 +6,7 @@ use std::cmp::Ordering; use super::entry::Entry; use super::matrix::CSRMatrix; use super::util::KBNSummer; +use crate::error::{Error, Result}; #[derive(Clone, PartialEq, Debug, Serialize)] pub struct Vector { @@ -22,8 +23,9 @@ impl Vector { pub fn to_dense(&self) -> Vec { let mut dense = vec![0.0; self.dim]; + // add, not assign: a sparse vector with repeated indices means their sum for e in &self.entries { - dense[e.index] = e.value; + dense[e.index] += e.value; } dense } @@ -60,17 +62,17 @@ impl Vector { self.entries.iter().map(|e| e.value).sum() } - pub fn add_vec(&mut self, v1: &Self, v2: &Self) -> Result<(), String> { + pub fn add_vec(&mut self, v1: &Self, v2: &Self) -> Result<()> { self.binary_operation(v1, v2, |x, y| x + y) } - pub fn sub_vec(&mut self, v1: &Self, v2: &Self) -> Result<(), String> { + pub fn sub_vec(&mut self, v1: &Self, v2: &Self) -> Result<()> { self.binary_operation(v1, v2, |x, y| x - y) } - pub fn scale_vec(&mut self, a: f64, v1: &Self) -> Result<(), String> { + pub fn scale_vec(&mut self, a: f64, v1: &Self) -> Result<()> { if a.is_nan() { - return Err("alpha cannot be NaN".to_string()); + return Err(Error::InvalidAlpha(a)); } if a == 0.0 { self.dim = v1.dim; @@ -91,10 +93,10 @@ impl Vector { } #[cfg(not(target_arch = "wasm32"))] - pub fn mul_vec(&mut self, m: &CSRMatrix, v1: &Self) -> Result<(), String> { + pub fn mul_vec(&mut self, m: &CSRMatrix, v1: &Self) -> Result<()> { let dim = m.cs_matrix.dim()?; if dim != v1.dim { - return Err("Dimension mismatch".to_string()); + return Err(Error::DimensionMismatch); } let dense = v1.to_dense(); @@ -120,10 +122,10 @@ impl Vector { } #[cfg(target_arch = "wasm32")] - pub fn mul_vec(&mut self, m: &CSRMatrix, v1: &Self) -> Result<(), String> { + pub fn mul_vec(&mut self, m: &CSRMatrix, v1: &Self) -> Result<()> { let dim = m.cs_matrix.dim()?; if dim != v1.dim { - return Err("Dimension mismatch".to_string()); + return Err(Error::DimensionMismatch); } let dense = v1.to_dense(); @@ -148,12 +150,12 @@ impl Vector { self.entries.sort_by_key(|e| e.index); } - fn binary_operation(&mut self, v1: &Self, v2: &Self, op: F) -> Result<(), String> + fn binary_operation(&mut self, v1: &Self, v2: &Self, op: F) -> Result<()> where F: Fn(f64, f64) -> f64, { if v1.dim != v2.dim { - return Err("Dimension mismatch".to_string()); + return Err(Error::DimensionMismatch); } let mut entries = Vec::with_capacity(v1.entries.len() + v2.entries.len()); From e4d9bd0685e742d319abb98e2a693c7d72238862 Mon Sep 17 00:00:00 2001 From: jenya Date: Wed, 30 Sep 2026 17:00:59 +0300 Subject: [PATCH 2/4] CI: fmt, clippy (native and wasm32), wasm tests and builds, cargo audit --- .github/workflows/ci.yml | 60 ++++++++++++++++++++++++++++++++++++++++ 1 file changed, 60 insertions(+) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 3e970b5..3cf5940 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -4,14 +4,74 @@ on: push: pull_request: +env: + CARGO_TERM_COLOR: always + jobs: + fmt: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + - uses: dtolnay/rust-toolchain@stable + with: + components: rustfmt + - run: cargo fmt --check + + clippy: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + - uses: dtolnay/rust-toolchain@stable + with: + components: clippy + targets: wasm32-unknown-unknown + - uses: Swatinem/rust-cache@v2 + - name: Native + run: cargo clippy --release --all-targets -- -D warnings + - name: WebAssembly + run: cargo clippy --release --target wasm32-unknown-unknown --all-targets -- -D warnings + test: runs-on: ubuntu-latest steps: - uses: actions/checkout@v4 - uses: dtolnay/rust-toolchain@stable + - uses: Swatinem/rust-cache@v2 - run: cargo test --release + wasm: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + - uses: dtolnay/rust-toolchain@stable + with: + targets: wasm32-unknown-unknown + - uses: dtolnay/rust-toolchain@nightly + with: + components: rust-src + targets: wasm32-unknown-unknown + - run: rustup default stable + - uses: Swatinem/rust-cache@v2 + - uses: taiki-e/install-action@v2 + with: + tool: wasm-pack + - uses: actions/setup-node@v4 + with: + node-version: 22 + - name: Tests in Node + run: wasm-pack test --node --release + - name: Single-threaded and multithreaded builds + run: ./build.sh + + audit: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + - uses: taiki-e/install-action@v2 + with: + tool: cargo-audit + - run: cargo audit + attribution: runs-on: ubuntu-latest steps: From fa0418bde8a92a8c7ec6ef28559117c1ba1ff135 Mon Sep 17 00:00:00 2001 From: jenya Date: Wed, 30 Sep 2026 17:00:59 +0300 Subject: [PATCH 3/4] Playground: quote-aware CSV, show engine errors - CSV tab parses and writes quoted fields the same way as the engine - engine errors (e.g. alpha 0 on a periodic network) are shown under the ranking; before they were overwritten right away --- demo/app.js | 53 ++++++++++++++++++++++++++++++++++++------------- demo/index.html | 1 + demo/style.css | 1 + 3 files changed, 41 insertions(+), 14 deletions(-) diff --git a/demo/app.js b/demo/app.js index 27bcb8f..6676f97 100644 --- a/demo/app.js +++ b/demo/app.js @@ -21,6 +21,7 @@ const state = { hover: null, scores: [], // [[peer, score]] from the last run, sorted large: null, // { lt, pt, peers } when the network is too big to draw + error: null, // engine error for the current network lastRun: null, // { ms, mode } } @@ -71,6 +72,7 @@ function clearNetwork() { state.selected = null state.large = null state.scores = [] + state.error = null } // ---------- presets ---------- @@ -157,11 +159,11 @@ function networkCsv() { const lt = [] const linked = new Set() for (const e of state.edges.values()) { - lt.push(`${e.from},${e.to},${e.w}`) + lt.push(`${csvField(e.from)},${csvField(e.to)},${e.w}`) linked.add(e.from) linked.add(e.to) } - const pt = [...state.seeds].filter((s) => linked.has(s)).map((s) => `${s},1`) + const pt = [...state.seeds].filter((s) => linked.has(s)).map((s) => `${csvField(s)},1`) return { lt: lt.join('\n'), pt: pt.join('\n') } } @@ -186,10 +188,10 @@ async function compute() { const { lt, pt } = state.large || networkCsv() const res = await runEngine(lt, pt, state.alpha) running = false + state.error = res.error || null if (res.error) { state.scores = [] state.lastRun = null - $('stats').textContent = res.error } else { state.scores = res.scores state.lastRun = { ms: res.ms, threads: res.threads } @@ -602,6 +604,8 @@ function renderRanking() { } stats.textContent = parts.join(lang() === 'zh' || lang() === 'ja' ? ',' : lang() === 'ar' ? '، ' : ', ') } + $('engineError').textContent = state.error || '' + $('engineError').hidden = !state.error $('engine').textContent = engineText() } @@ -737,23 +741,44 @@ function countLines(s) { const HEADER_NAMES = new Set(['i', 'j', 'v', 'from', 'to', 'value', 'weight', 'trust', 'level', 'peer', 'id', 'score', 'source', 'target', 'src', 'dst', 'truster', 'trustee']) -const clean = (f) => f.trim().replace(/^"(.*)"$/, '$1').trim() +// one CSV line: quoted fields may contain commas and doubled quotes, same rules as the engine +function splitCsvLine(line) { + const out = [] + let cur = '' + let quoted = false + for (let i = 0; i < line.length; i++) { + const c = line[i] + if (quoted) { + if (c !== '"') cur += c + else if (line[i + 1] === '"') { cur += '"'; i++ } + else quoted = false + } else if (c === '"' && !cur.trim()) { cur = ''; quoted = true } + else if (c === ',') { out.push(cur.trim()); cur = '' } + else cur += c + } + out.push(cur.trim()) + return out +} function parseRows(text, valueCol) { - const lines = text.replace(/^/, '').split('\n') const rows = [] - lines.forEach((line, i) => { - if (!line.trim()) return - const f = line.split(',').map(clean) - if (i === 0) { - const header = f.length > valueCol ? isNaN(parseFloat(f[valueCol])) : f.every((x) => HEADER_NAMES.has(x.toLowerCase())) - if (header) return + let first = true + for (const line of text.replace(/^\uFEFF/, '').split(/\r?\n/)) { + if (!line.trim()) continue + const f = splitCsvLine(line) + if (first) { + first = false + const header = f.length > valueCol ? isNaN(Number(f[valueCol])) : f.every((x) => HEADER_NAMES.has(x.toLowerCase())) + if (header) continue } rows.push(f) - }) + } return rows } +// quote a field when it needs it +const csvField = (v) => /[",\r\n]|^\s|\s$/.test(String(v)) ? '"' + String(v).replace(/"/g, '""') + '"' : String(v) + $('csvLoad').addEventListener('click', async () => { const err = $('csvError') err.hidden = true @@ -775,7 +800,7 @@ $('csvLoad').addEventListener('click', async () => { if (ltRows) for (const r of ltRows) { peers.add(r[0]); peers.add(r[1]) } if (!ltRows || peers.size > MAX_GRAPH_PEERS) { clearNetwork() - const count = ltRows ? peers.size : new Set(lt.split('\n').flatMap((l) => l.split(',').slice(0, 2).map(clean)).filter(Boolean)).size + const count = ltRows ? peers.size : new Set(lt.split('\n').flatMap((l) => splitCsvLine(l).slice(0, 2)).filter(Boolean)).size state.large = { lt, pt, peers: count } state.scores = res.scores state.lastRun = { ms: res.ms, threads: res.threads } @@ -790,7 +815,7 @@ $('csvLoad').addEventListener('click', async () => { }) $('csvDownload').addEventListener('click', () => { - const csv = 'peer,score\n' + state.scores.map(([p, s]) => `${p},${s}`).join('\n') + '\n' + const csv = 'peer,score\n' + state.scores.map(([p, s]) => `${csvField(p)},${s}`).join('\n') + '\n' const a = el('a', { href: URL.createObjectURL(new Blob([csv], { type: 'text/csv' })), download: 'eigentrust-scores.csv' }) a.click() setTimeout(() => URL.revokeObjectURL(a.href), 1000) diff --git a/demo/index.html b/demo/index.html index 8c194a9..0441441 100644 --- a/demo/index.html +++ b/demo/index.html @@ -94,6 +94,7 @@

Trust flows from the peers you already trust.

Ranking

+
    diff --git a/demo/style.css b/demo/style.css index 28e37c1..fd3f8fe 100644 --- a/demo/style.css +++ b/demo/style.css @@ -300,6 +300,7 @@ textarea { .actions { display: flex; gap: 8px; flex-wrap: wrap; } .error { color: var(--error); font-size: 14px; } +#engineError { margin: 4px 0 8px; font-size: 13.5px; } /* selected peer */ From 5ba266fbce85acf2a8705eedcdfd72b32b230db5 Mon Sep 17 00:00:00 2001 From: jenya Date: Wed, 30 Sep 2026 17:00:59 +0300 Subject: [PATCH 4/4] License under MIT OR Apache-2.0, document input rules --- CLAUDE.md | 18 +++-- LICENSE-APACHE | 202 +++++++++++++++++++++++++++++++++++++++++++++++++ LICENSE-MIT | 21 +++++ README.md | 12 ++- 4 files changed, 246 insertions(+), 7 deletions(-) create mode 100644 LICENSE-APACHE create mode 100644 LICENSE-MIT diff --git a/CLAUDE.md b/CLAUDE.md index c9b8c16..81c76fa 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -5,26 +5,31 @@ - NEVER add `Co-Authored-By: Claude ...`, `noreply@anthropic.com`, "Generated with Claude Code" or any other Claude / Anthropic attribution to commit messages, PR titles or PR descriptions. This project was written by its author before Claude was involved. Plain commit messages only. -- This is enforced by `.githooks/commit-msg`. Enable it once per clone: +- This is enforced by `.githooks/commit-msg` and by CI. Enable the hook once per clone: `git config core.hooksPath .githooks` - Never bypass it with `--no-verify`. ## Project -Rust port of EigenTrust (go-eigentrust), runs natively and as WASM in the browser. +EigenTrust in Rust, runs natively and as WASM in the browser. - `src/basic/engine.rs` - `calculate_from_csv`: CSV in, sorted `(peer, score)` out +- `src/basic/input.rs` - CSV reading (csv crate), header detection, weight validation - `src/basic/eigentrust.rs` - power iteration (`compute`), runs on dense vectors +- `src/error.rs` - `Error` enum used everywhere; `Display` gives user-facing messages - `src/sparse/` - CSR matrix / sparse vector - `src/lib.rs` - wasm-bindgen entry `run(localtrust, pretrust, alpha)` (wasm32 only) -- `src/main.rs` - native CLI +- `src/main.rs` - native CLI, reports errors as `error: ...` with exit code 1 - `demo/` - static interactive web demo deployed to Vercel ## Commands - Test: `cargo test --release` +- WASM tests: `wasm-pack test --node --release` +- Lint: `cargo fmt --check`, `cargo clippy --release --all-targets -- -D warnings` + (also with `--target wasm32-unknown-unknown`) - CLI: `cargo run --release -- ./example/localtrust.csv ./example/pretrust.csv [alpha]` -- WASM: `./build.sh` (builds `pkg/` and copies it into `demo/pkg/`) +- WASM: `./build.sh` (builds `pkg/` and `pkg-parallel/` and copies them into `demo/`) - Demo locally: `python3 -m http.server -d demo` then open http://localhost:8000 - Deploy: `vercel deploy --prod` from `demo/` @@ -32,4 +37,7 @@ Rust port of EigenTrust (go-eigentrust), runs natively and as WASM in the browse - Tests compare floats with exact equality. The iteration uses Kahan-Babuska-Neumaier summation in row order; keep the summation order if you touch `compute` / `mul_dense`. -- Duplicate `(i, j)` records in local trust: the last one wins. +- Duplicate `(i, j)` local trust records and duplicate pre-trust peers: the last one wins. +- Weights must be finite and non-negative. Distrust (negative weights) is rejected until its + effect on the scores is defined; `extract_distrust` / `discount_trust_vector` are kept for that. +- `compute` stops after `DEFAULT_MAX_ITERATIONS` (10,000) with `Error::NotConverged`. diff --git a/LICENSE-APACHE b/LICENSE-APACHE new file mode 100644 index 0000000..d645695 --- /dev/null +++ b/LICENSE-APACHE @@ -0,0 +1,202 @@ + + Apache License + Version 2.0, January 2004 + http://www.apache.org/licenses/ + + TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION + + 1. Definitions. + + "License" shall mean the terms and conditions for use, reproduction, + and distribution as defined by Sections 1 through 9 of this document. + + "Licensor" shall mean the copyright owner or entity authorized by + the copyright owner that is granting the License. + + "Legal Entity" shall mean the union of the acting entity and all + other entities that control, are controlled by, or are under common + control with that entity. For the purposes of this definition, + "control" means (i) the power, direct or indirect, to cause the + direction or management of such entity, whether by contract or + otherwise, or (ii) ownership of fifty percent (50%) or more of the + outstanding shares, or (iii) beneficial ownership of such entity. + + "You" (or "Your") shall mean an individual or Legal Entity + exercising permissions granted by this License. + + "Source" form shall mean the preferred form for making modifications, + including but not limited to software source code, documentation + source, and configuration files. + + "Object" form shall mean any form resulting from mechanical + transformation or translation of a Source form, including but + not limited to compiled object code, generated documentation, + and conversions to other media types. + + "Work" shall mean the work of authorship, whether in Source or + Object form, made available under the License, as indicated by a + copyright notice that is included in or attached to the work + (an example is provided in the Appendix below). + + "Derivative Works" shall mean any work, whether in Source or Object + form, that is based on (or derived from) the Work and for which the + editorial revisions, annotations, elaborations, or other modifications + represent, as a whole, an original work of authorship. For the purposes + of this License, Derivative Works shall not include works that remain + separable from, or merely link (or bind by name) to the interfaces of, + the Work and Derivative Works thereof. + + "Contribution" shall mean any work of authorship, including + the original version of the Work and any modifications or additions + to that Work or Derivative Works thereof, that is intentionally + submitted to Licensor for inclusion in the Work by the copyright owner + or by an individual or Legal Entity authorized to submit on behalf of + the copyright owner. For the purposes of this definition, "submitted" + means any form of electronic, verbal, or written communication sent + to the Licensor or its representatives, including but not limited to + communication on electronic mailing lists, source code control systems, + and issue tracking systems that are managed by, or on behalf of, the + Licensor for the purpose of discussing and improving the Work, but + excluding communication that is conspicuously marked or otherwise + designated in writing by the copyright owner as "Not a Contribution." + + "Contributor" shall mean Licensor and any individual or Legal Entity + on behalf of whom a Contribution has been received by Licensor and + subsequently incorporated within the Work. + + 2. Grant of Copyright License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + copyright license to reproduce, prepare Derivative Works of, + publicly display, publicly perform, sublicense, and distribute the + Work and such Derivative Works in Source or Object form. + + 3. Grant of Patent License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + (except as stated in this section) patent license to make, have made, + use, offer to sell, sell, import, and otherwise transfer the Work, + where such license applies only to those patent claims licensable + by such Contributor that are necessarily infringed by their + Contribution(s) alone or by combination of their Contribution(s) + with the Work to which such Contribution(s) was submitted. If You + institute patent litigation against any entity (including a + cross-claim or counterclaim in a lawsuit) alleging that the Work + or a Contribution incorporated within the Work constitutes direct + or contributory patent infringement, then any patent licenses + granted to You under this License for that Work shall terminate + as of the date such litigation is filed. + + 4. Redistribution. You may reproduce and distribute copies of the + Work or Derivative Works thereof in any medium, with or without + modifications, and in Source or Object form, provided that You + meet the following conditions: + + (a) You must give any other recipients of the Work or + Derivative Works a copy of this License; and + + (b) You must cause any modified files to carry prominent notices + stating that You changed the files; and + + (c) You must retain, in the Source form of any Derivative Works + that You distribute, all copyright, patent, trademark, and + attribution notices from the Source form of the Work, + excluding those notices that do not pertain to any part of + the Derivative Works; and + + (d) If the Work includes a "NOTICE" text file as part of its + distribution, then any Derivative Works that You distribute must + include a readable copy of the attribution notices contained + within such NOTICE file, excluding those notices that do not + pertain to any part of the Derivative Works, in at least one + of the following places: within a NOTICE text file distributed + as part of the Derivative Works; within the Source form or + documentation, if provided along with the Derivative Works; or, + within a display generated by the Derivative Works, if and + wherever such third-party notices normally appear. The contents + of the NOTICE file are for informational purposes only and + do not modify the License. You may add Your own attribution + notices within Derivative Works that You distribute, alongside + or as an addendum to the NOTICE text from the Work, provided + that such additional attribution notices cannot be construed + as modifying the License. + + You may add Your own copyright statement to Your modifications and + may provide additional or different license terms and conditions + for use, reproduction, or distribution of Your modifications, or + for any such Derivative Works as a whole, provided Your use, + reproduction, and distribution of the Work otherwise complies with + the conditions stated in this License. + + 5. Submission of Contributions. Unless You explicitly state otherwise, + any Contribution intentionally submitted for inclusion in the Work + by You to the Licensor shall be under the terms and conditions of + this License, without any additional terms or conditions. + Notwithstanding the above, nothing herein shall supersede or modify + the terms of any separate license agreement you may have executed + with Licensor regarding such Contributions. + + 6. Trademarks. This License does not grant permission to use the trade + names, trademarks, service marks, or product names of the Licensor, + except as required for reasonable and customary use in describing the + origin of the Work and reproducing the content of the NOTICE file. + + 7. Disclaimer of Warranty. Unless required by applicable law or + agreed to in writing, Licensor provides the Work (and each + Contributor provides its Contributions) on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or + implied, including, without limitation, any warranties or conditions + of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A + PARTICULAR PURPOSE. You are solely responsible for determining the + appropriateness of using or redistributing the Work and assume any + risks associated with Your exercise of permissions under this License. + + 8. Limitation of Liability. In no event and under no legal theory, + whether in tort (including negligence), contract, or otherwise, + unless required by applicable law (such as deliberate and grossly + negligent acts) or agreed to in writing, shall any Contributor be + liable to You for damages, including any direct, indirect, special, + incidental, or consequential damages of any character arising as a + result of this License or out of the use or inability to use the + Work (including but not limited to damages for loss of goodwill, + work stoppage, computer failure or malfunction, or any and all + other commercial damages or losses), even if such Contributor + has been advised of the possibility of such damages. + + 9. Accepting Warranty or Additional Liability. While redistributing + the Work or Derivative Works thereof, You may choose to offer, + and charge a fee for, acceptance of support, warranty, indemnity, + or other liability obligations and/or rights consistent with this + License. However, in accepting such obligations, You may act only + on Your own behalf and on Your sole responsibility, not on behalf + of any other Contributor, and only if You agree to indemnify, + defend, and hold each Contributor harmless for any liability + incurred by, or claims asserted against, such Contributor by reason + of your accepting any such warranty or additional liability. + + END OF TERMS AND CONDITIONS + + APPENDIX: How to apply the Apache License to your work. + + To apply the Apache License to your work, attach the following + boilerplate notice, with the fields enclosed by brackets "[]" + replaced with your own identifying information. (Don't include + the brackets!) The text should be enclosed in the appropriate + comment syntax for the file format. We also recommend that a + file or class name and description of purpose be included on the + same "printed page" as the copyright notice for easier + identification within third-party archives. + + Copyright [yyyy] [name of copyright owner] + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. diff --git a/LICENSE-MIT b/LICENSE-MIT new file mode 100644 index 0000000..c137d6b --- /dev/null +++ b/LICENSE-MIT @@ -0,0 +1,21 @@ +MIT License + +Copyright (c) 2024 Jenya + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. diff --git a/README.md b/README.md index 1bf4e58..6d1671b 100644 --- a/README.md +++ b/README.md @@ -10,6 +10,7 @@ Live demo Rust WebAssembly + License: MIT OR Apache-2.0

    @@ -70,8 +71,11 @@ Build with `./build.sh`. For a ready-made Web Worker, see [`demo/worker.js`](dem | Local trust | `from,to[,weight]` | `alice,bob,2` | | Seeds (pre-trust) | `peer[,weight]` | `alice,1` | -- A header row is optional. Spaces, quoted fields, CRLF and a UTF-8 BOM are fine. -- Weights default to 1. Repeated `from,to` pairs: the last line wins. +- Standard CSV: quoted fields (`"Smith, J"`), a header row, spaces, CRLF and a UTF-8 BOM are fine. +- Weights default to 1 and must be finite and non-negative. Negative trust (distrust) is not supported yet. +- A repeated `from,to` pair or seed peer: the last line wins. +- α must be in [0, 1]. At α = 0 some networks oscillate instead of converging; the engine stops after 10,000 iterations with an error. +- Errors name the file and line, e.g. `local trust CSV, line 5: weight "NaN" must be finite`. ## Performance @@ -93,3 +97,7 @@ cargo test --release # tests python3 -m http.server -d demo # run the playground locally git config core.hooksPath .githooks # once per clone ``` + +## License + +Licensed under either of [Apache License 2.0](LICENSE-APACHE) or [MIT](LICENSE-MIT), at your option.