diff --git a/Cargo.lock b/Cargo.lock index 5be8781..e4583c6 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -2,17 +2,6 @@ # It is not intended for manual editing. version = 3 -[[package]] -name = "atty" -version = "0.2.14" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d9b39be18770d11421cdb1b9947a45dd3f37e93092cbf377614828a319d5fee8" -dependencies = [ - "hermit-abi", - "libc", - "winapi", -] - [[package]] name = "autocfg" version = "1.1.0" @@ -37,15 +26,6 @@ version = "0.4.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "95505c38b4572b2d910cecb0281560f54b440a19336cbbcb27bf6ce6adc6f5a8" -[[package]] -name = "hermit-abi" -version = "0.1.19" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "62b467343b94ba476dcb2500d242dadbb39557df889310ac77c5d99100aaac33" -dependencies = [ - "libc", -] - [[package]] name = "indoc" version = "2.0.4" @@ -123,9 +103,8 @@ dependencies = [ [[package]] name = "pyagushka" -version = "1.0.0" +version = "1.1.0" dependencies = [ - "atty", "pyo3", "serde", "serde_json", @@ -288,28 +267,6 @@ version = "0.2.3" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "c7de7d73e1754487cb58364ee906a499937a0dfabd86bcb980fa99ec8c8fa2ce" -[[package]] -name = "winapi" -version = "0.3.9" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5c839a674fcd7a98952e593242ea400abe93992746761e38641405d28b00f419" -dependencies = [ - "winapi-i686-pc-windows-gnu", - "winapi-x86_64-pc-windows-gnu", -] - -[[package]] -name = "winapi-i686-pc-windows-gnu" -version = "0.4.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ac3b87c63620426dd9b991e5ce0329eff545bccbbb34f3be09ff6fb6ab51b7b6" - -[[package]] -name = "winapi-x86_64-pc-windows-gnu" -version = "0.4.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "712e227841d057c1ee1cd2fb22fa7e5a5461ae8e48fa2ca79ec42cfc1931183f" - [[package]] name = "windows-targets" version = "0.48.5" diff --git a/Cargo.toml b/Cargo.toml index ed35651..0827562 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "pyagushka" -version = "1.0.0" +version = "1.1.0" edition = "2021" [lib] @@ -8,7 +8,6 @@ name = "pyagushka" crate-type = ["cdylib"] [dependencies] -atty = "0.2.14" pyo3 = "0.20.2" serde = { version = "1.0.196", features = ["derive"] } serde_json = "1.0.113" diff --git a/readme.md b/readme.md index 63b7893..830f24c 100644 --- a/readme.md +++ b/readme.md @@ -13,7 +13,7 @@ With a Rust/Cargo and Python3/Pip environment set up, run: ```sh $ pip install maturin $ maturin build --release -$ pip install target/wheels/pyagushka-1.0.0-*.whl +$ pip install target/wheels/pyagushka-1.1.0-*.whl ``` ## Usage @@ -58,14 +58,15 @@ The tool outputs a JSON string that includes details about the identified attrac To analyze a dataset from a file, provide the filename as an argument, followed by the factor and minimum cluster size parameters ```Python -from pyagushka import lyagushka +from pyagushka import Lyagushka dataset = [] with open('random_values.txt', 'r') as file: for line in file: random_data.append(int(line.strip())) -analysis_results = json.loads(lyagushka(dataset, 4.0, 7)) +zhaba = Lyagushka(dataset) +analysis_results = json.loads(lyagushka.search(4.0, 7)) print(analysis_result) ``` diff --git a/src/lib.rs b/src/lib.rs index ff6caea..7a799c9 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -1,8 +1,5 @@ use pyo3::prelude::*; -use pyo3::types::PyList; -use pyo3::wrap_pyfunction; use serde::Serialize; -use serde_json::to_string_pretty; #[derive(Debug, Clone, Serialize)] struct Anomaly { @@ -15,181 +12,149 @@ struct Anomaly { z_score: Option, } -fn anomaly_info(cluster: &[i32]) -> Anomaly { - let num_elements: usize = cluster.len(); - let start: i32 = *cluster.first().expect("Cluster has no start"); - let end: i32 = *cluster.last().expect("Cluster has no end"); - let span_length: i32 = end - start; - let centroid: f32 = start as f32 + span_length as f32 / 2.0; +impl Anomaly { - Anomaly { - elements: cluster.to_vec(), - start, - end, - span_length, - num_elements, - centroid, - z_score: None, // Placeholder for actual Z-score calculation + pub fn new(cluster: &[i32]) -> Self { + let num_elements: usize = cluster.len(); + let start: i32 = *cluster.first().expect("Cluster has no start"); + let end: i32 = *cluster.last().expect("Cluster has no end"); + let span_length: i32 = end - start; + let centroid: f32 = start as f32 + span_length as f32 / 2.0; + + Anomaly { + elements: cluster.to_vec(), + start, + end, + span_length, + num_elements, + centroid, + z_score: None, + } } } +#[pyclass] +struct Lyagushka { + dataset: Vec, + anomalies: Vec, +} -/// Calculates the densities (clusters) and significant gaps between points in a dataset. -/// -/// This function iterates over a dataset of points, identifying clusters based on a distance threshold -/// (calculated from the mean distance between points and adjusted by a given factor) and identifying significant gaps -/// that exceed a certain threshold. Each cluster or significant gap identified is summarized in a `Anomaly` object. -/// -/// # Arguments -/// * `dataset`: A slice of `i32` objects representing the dataset to be analyzed. -/// * `factor`: A multiplier used to define the thresholds for clustering and gap identification. -/// A lower factor tightens the cluster threshold and widens the gap threshold, and vice versa. -/// * `min_cluster_size`: The minimum number of points required for a group of points to be considered a cluster. -/// -/// # Returns -/// A vector of `Anomaly` objects, each representing either a cluster of points or a significant gap between points. -/// -fn scan_anomalies(dataset: &[i32], factor: f32, min_cluster_size: usize) -> Vec { +#[pymethods] +impl Lyagushka { - // Return early if the dataset is too small to form any clusters or gaps. - if dataset.len() < 2 { return Vec::new(); } - - // Calculate the mean distance between consecutive points in the dataset. - let mean_distance: f32 = dataset.windows(2) - .map(|w| (w[1] - w[0]) as f32) - .sum::() / (dataset.len() - 1) as f32; - - // Define thresholds for clustering and gap identification based on the mean distance and factor. - let cluster_threshold: f32 = mean_distance / factor; - let gap_threshold: f32 = factor * mean_distance; - - let mut results: Vec = Vec::new(); // Stores the resulting clusters and gaps. - let mut current_cluster: Vec = Vec::new(); // Temporary storage for points in the current cluster. - - // Iterate through pairs of consecutive points to find clusters and significant gaps. - for window in dataset.windows(2) { - let gap_size: f32 = (window[1] - window[0]) as f32; - - if gap_size <= cluster_threshold { - // Add points to the current cluster - if current_cluster.is_empty() { - current_cluster.push(window[0]); // Start a new cluster with the first point - } - current_cluster.push(window[1]); // Add the second point to the cluster - } else { - // End the current cluster and start a new gap - if !current_cluster.is_empty() && current_cluster.len() >= min_cluster_size { - results.push(anomaly_info(¤t_cluster)); - current_cluster.clear(); - } - - // Record the gap - if gap_size > gap_threshold { - results.push(Anomaly { - elements: Vec::new(), // No elements in a gap - start: window[0], - end: window[1], - span_length: gap_size as i32, - num_elements: 0, - centroid: (window[0] as f32 + window[1] as f32) / 2.0, - z_score: None, - }); - } + #[new] + pub fn new(dataset: Vec) -> Self { + Lyagushka { + dataset, + anomalies: vec![] } } - // Finalize the last cluster if applicable - if !current_cluster.is_empty() && current_cluster.len() >= min_cluster_size { - results.push(anomaly_info(¤t_cluster)); - } - - results -} - -/// Analyzes a dataset of integers to identify clusters and gaps, then calculates Z-scores -/// for each based on their deviation from mean metrics. The analysis aims to highlight -/// significant clusters of closely grouped points and notable gaps between them, providing -/// a statistical measure of their significance through Z-scores. The results, including -/// clusters, gaps, and their Z-scores, are serialized into a JSON string. -/// -/// # Arguments -/// * `_py` - The Python interpreter instance, used for Python-Rust interoperability. -/// This argument is necessary for functions exposed to Python via PyO3 but is not -/// directly used within the function. -/// * `int_list` - A Python list of integers representing the dataset to be analyzed. -/// This list is converted into a Vec for internal processing. -/// * `factor` - A floating-point value used as a threshold factor to adjust the sensitivity -/// of cluster and gap detection. This factor influences the identification of clusters -/// by defining the minimum density or separation required. -/// * `min_cluster_size` - An integer specifying the minimum number of contiguous points -/// required for a group of points to be considered a cluster. This parameter helps -/// filter out noise by defining a threshold for the minimum cluster size. -/// -/// # Returns -/// Returns a `PyResult` containing a JSON-formatted string of the analysis results. -/// The JSON string includes detailed information about each identified cluster and gap, -/// such as their span length, number of elements (if applicable), centroid, and calculated -/// Z-score. In case of an error during processing or serialization, a Python exception is -/// returned. -/// -#[pyfunction] -fn lyagushka(_py: Python, int_list: &PyList, factor: f32, min_cluster_size: usize) -> PyResult { - // Extract integers from a Python list and create a vector. - let mut dataset: Vec = int_list.extract::>()?; + fn scan_anomalies(&mut self, factor: f32, min_cluster_size: usize) { - // Sort the vector - dataset.sort_unstable(); - - // Calculate clusters and gaps from the dataset using predefined criteria. - let mut anomalies: Vec = scan_anomalies(&dataset, factor, min_cluster_size); - - // Calculate the mean density of clusters in the dataset for comparison. - let mean_density: f32 = anomalies.iter() - .filter(|info: &&Anomaly| info.num_elements > 0) - .map(|info: &Anomaly| info.num_elements as f32 / info.span_length as f32) - .sum::() / anomalies.iter().filter(|info: &&Anomaly| info.num_elements > 0).count() as f32; - - // Calculate the standard deviation of cluster densities to evaluate variation. - let variance_density: f32 = anomalies.iter() - .filter(|info: &&Anomaly| info.num_elements > 0) - .map(|info: &Anomaly| info.num_elements as f32 / info.span_length as f32) - .map(|density| (density - mean_density).powi(2)) - .sum::() / anomalies.iter().filter(|info: &&Anomaly| info.num_elements > 0).count() as f32; - let std_dev_density = variance_density.sqrt(); - - // Calculate mean span length - let mean_span_length: f32 = anomalies.iter() - .map(|info: &Anomaly| info.span_length as f32) - .sum::() / anomalies.len() as f32; - - // Calculate variance - let variance: f32 = anomalies.iter() - .map(|info: &Anomaly| (info.span_length as f32 - mean_span_length).powi(2)) - .sum::() / anomalies.len() as f32; - - // Standard deviation is the square root of variance - let std_dev_span_length: f32 = variance.sqrt(); - - // Update Z-scores for both clusters and gaps based on their deviation from mean metrics. - for info in anomalies.iter_mut() { - if info.num_elements > 0 { - // Calculate and update Z-score for clusters based on density deviation. - let cluster_density: f32 = info.num_elements as f32 / info.span_length as f32; - info.z_score = Some((cluster_density - mean_density) / std_dev_density); - } else { - // Calculate and update Z-score for gaps based on span length deviation. - info.z_score = Some((info.span_length as f32 / std_dev_span_length) * -1.0); + // Calculate the mean distance between consecutive points in the dataset. + let mean_distance: f32 = self.dataset.windows(2) + .map(|w| (w[1] - w[0]) as f32) + .sum::() / (self.dataset.len() - 1) as f32; + + // Define thresholds for clustering and gap identification based on the mean distance and factor. + let cluster_threshold: f32 = mean_distance / factor; + let gap_threshold: f32 = factor * mean_distance; + + let mut current_cluster: Vec = Vec::new(); // Temporary storage for points in the current cluster. + + // Iterate through pairs of consecutive points to find clusters and significant gaps. + for window in self.dataset.windows(2) { + let gap_size: f32 = (window[1] - window[0]) as f32; + + if gap_size <= cluster_threshold { + // Add points to the current cluster + if current_cluster.is_empty() { + current_cluster.push(window[0]); // Start a new cluster with the first point + } + current_cluster.push(window[1]); // Add the second point to the cluster + } else { + // End the current cluster and start a new gap + if !current_cluster.is_empty() && current_cluster.len() >= min_cluster_size { + self.anomalies.push(Anomaly::new(¤t_cluster)); + current_cluster.clear(); + } + + // Record the gap + if gap_size > gap_threshold { + self.anomalies.push(Anomaly { + elements: Vec::new(), // No elements in a gap + start: window[0], + end: window[1], + span_length: gap_size as i32, + num_elements: 0, + centroid: (window[0] as f32 + window[1] as f32) / 2.0, + z_score: None, + }); + } + } } + + // Finalize the last cluster if applicable + if !current_cluster.is_empty() && current_cluster.len() >= min_cluster_size { + self.anomalies.push(Anomaly::new(¤t_cluster)); + } + } - // Serialize the updated cluster and gap information, including Z-scores, to a JSON string. - to_string_pretty(&anomalies) - .map_err(|e| PyErr::new::(format!("JSON Serialization Error: {}", e))) -} + pub fn search(&mut self, factor: f32, min_cluster_size: usize) -> String { + // Sort the vector + self.dataset.sort_unstable(); + + // Calculate clusters and gaps from the dataset using predefined criteria. + self.scan_anomalies(factor, min_cluster_size); + + // Calculate the mean density of clusters in the dataset for comparison. + let mean_density: f32 = self.anomalies.iter() + .filter(|info: &&Anomaly| info.num_elements > 0) + .map(|info: &Anomaly| info.num_elements as f32 / info.span_length as f32) + .sum::() / self.anomalies.iter().filter(|info: &&Anomaly| info.num_elements > 0).count() as f32; + + // Calculate the standard deviation of cluster densities to evaluate variation. + let variance_density: f32 = self.anomalies.iter() + .filter(|info: &&Anomaly| info.num_elements > 0) + .map(|info: &Anomaly| info.num_elements as f32 / info.span_length as f32) + .map(|density: f32| (density - mean_density).powi(2)) + .sum::() / self.anomalies.iter().filter(|info: &&Anomaly| info.num_elements > 0).count() as f32; + let std_dev_density: f32 = variance_density.sqrt(); + + // Calculate mean span length + let mean_span_length: f32 = self.anomalies.iter() + .map(|info: &Anomaly| info.span_length as f32) + .sum::() / self.anomalies.len() as f32; + + // Calculate variance + let variance: f32 = self.anomalies.iter() + .map(|info: &Anomaly| (info.span_length as f32 - mean_span_length).powi(2)) + .sum::() / self.anomalies.len() as f32; + + // Standard deviation is the square root of variance + let std_dev_span_length: f32 = variance.sqrt(); + + // Update Z-scores for both clusters and gaps based on their deviation from mean metrics. + for info in self.anomalies.iter_mut() { + if info.num_elements > 0 { + // Calculate and update Z-score for clusters based on density deviation. + let cluster_density: f32 = info.num_elements as f32 / info.span_length as f32; + info.z_score = Some((cluster_density - mean_density) / std_dev_density); + } else { + // Calculate and update Z-score for gaps based on span length deviation. + info.z_score = Some((info.span_length as f32 / std_dev_span_length) * -1.0); + } + } + + serde_json::to_string_pretty(&self.anomalies).unwrap_or_else(|_| "Failed to serialize data".to_string()) + } +} #[pymodule] fn pyagushka(_py: Python, m: &PyModule) -> PyResult<()> { - m.add_function(wrap_pyfunction!(lyagushka, m)?)?; + m.add_class::()?; Ok(()) } diff --git a/test.py b/test.py index 129515a..111ded7 100644 --- a/test.py +++ b/test.py @@ -1,4 +1,4 @@ -from pyagushka import lyagushka +from pyagushka import Lyagushka from randonautentropy import rndo import json import matplotlib.pyplot as plt @@ -41,7 +41,8 @@ with open('dataset.json', 'w') as r: r.write(json.dumps(dataset, indent=4)) # calculate the anomalies in the data -analysis_results = json.loads(lyagushka(dataset, 4.0, 7)) +zhaba = Lyagushka(dataset) +analysis_results = json.loads(zhaba.search(4.0, 7)) analysis_results = filter_by_z_score(analysis_results, 1.0) with open('result.json', 'w') as r: