From b9ec578d1f9fb1222c3764ee808b32a5bb57726d Mon Sep 17 00:00:00 2001 From: Patrick Lam Date: Wed, 28 May 2025 12:29:31 +1200 Subject: [PATCH 01/16] add unsafe-finder tool --- tools/unsafe-finder/Cargo.toml | 8 ++ tools/unsafe-finder/LICENSE-APACHE | 176 ++++++++++++++++++++++++++ tools/unsafe-finder/LICENSE-MIT | 23 ++++ tools/unsafe-finder/README.md | 49 +++++++ tools/unsafe-finder/src/main.rs | 197 +++++++++++++++++++++++++++++ 5 files changed, 453 insertions(+) create mode 100644 tools/unsafe-finder/Cargo.toml create mode 100644 tools/unsafe-finder/LICENSE-APACHE create mode 100644 tools/unsafe-finder/LICENSE-MIT create mode 100644 tools/unsafe-finder/README.md create mode 100644 tools/unsafe-finder/src/main.rs diff --git a/tools/unsafe-finder/Cargo.toml b/tools/unsafe-finder/Cargo.toml new file mode 100644 index 0000000000000..7c3284b8f73b0 --- /dev/null +++ b/tools/unsafe-finder/Cargo.toml @@ -0,0 +1,8 @@ +[package] +name = "unsafe-finder" +version = "0.1.0" +edition = "2024" + +[dependencies] +prettyplease = "0.2.32" +syn = {version = "2.0.101", features = ["full", "extra-traits", "visit"]} diff --git a/tools/unsafe-finder/LICENSE-APACHE b/tools/unsafe-finder/LICENSE-APACHE new file mode 100644 index 0000000000000..1b5ec8b78e237 --- /dev/null +++ b/tools/unsafe-finder/LICENSE-APACHE @@ -0,0 +1,176 @@ + Apache License + Version 2.0, January 2004 + http://www.apache.org/licenses/ + +TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION + +1. Definitions. + + "License" shall mean the terms and conditions for use, reproduction, + and distribution as defined by Sections 1 through 9 of this document. + + "Licensor" shall mean the copyright owner or entity authorized by + the copyright owner that is granting the License. + + "Legal Entity" shall mean the union of the acting entity and all + other entities that control, are controlled by, or are under common + control with that entity. For the purposes of this definition, + "control" means (i) the power, direct or indirect, to cause the + direction or management of such entity, whether by contract or + otherwise, or (ii) ownership of fifty percent (50%) or more of the + outstanding shares, or (iii) beneficial ownership of such entity. + + "You" (or "Your") shall mean an individual or Legal Entity + exercising permissions granted by this License. + + "Source" form shall mean the preferred form for making modifications, + including but not limited to software source code, documentation + source, and configuration files. + + "Object" form shall mean any form resulting from mechanical + transformation or translation of a Source form, including but + not limited to compiled object code, generated documentation, + and conversions to other media types. + + "Work" shall mean the work of authorship, whether in Source or + Object form, made available under the License, as indicated by a + copyright notice that is included in or attached to the work + (an example is provided in the Appendix below). + + "Derivative Works" shall mean any work, whether in Source or Object + form, that is based on (or derived from) the Work and for which the + editorial revisions, annotations, elaborations, or other modifications + represent, as a whole, an original work of authorship. For the purposes + of this License, Derivative Works shall not include works that remain + separable from, or merely link (or bind by name) to the interfaces of, + the Work and Derivative Works thereof. + + "Contribution" shall mean any work of authorship, including + the original version of the Work and any modifications or additions + to that Work or Derivative Works thereof, that is intentionally + submitted to Licensor for inclusion in the Work by the copyright owner + or by an individual or Legal Entity authorized to submit on behalf of + the copyright owner. For the purposes of this definition, "submitted" + means any form of electronic, verbal, or written communication sent + to the Licensor or its representatives, including but not limited to + communication on electronic mailing lists, source code control systems, + and issue tracking systems that are managed by, or on behalf of, the + Licensor for the purpose of discussing and improving the Work, but + excluding communication that is conspicuously marked or otherwise + designated in writing by the copyright owner as "Not a Contribution." + + "Contributor" shall mean Licensor and any individual or Legal Entity + on behalf of whom a Contribution has been received by Licensor and + subsequently incorporated within the Work. + +2. Grant of Copyright License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + copyright license to reproduce, prepare Derivative Works of, + publicly display, publicly perform, sublicense, and distribute the + Work and such Derivative Works in Source or Object form. + +3. Grant of Patent License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + (except as stated in this section) patent license to make, have made, + use, offer to sell, sell, import, and otherwise transfer the Work, + where such license applies only to those patent claims licensable + by such Contributor that are necessarily infringed by their + Contribution(s) alone or by combination of their Contribution(s) + with the Work to which such Contribution(s) was submitted. If You + institute patent litigation against any entity (including a + cross-claim or counterclaim in a lawsuit) alleging that the Work + or a Contribution incorporated within the Work constitutes direct + or contributory patent infringement, then any patent licenses + granted to You under this License for that Work shall terminate + as of the date such litigation is filed. + +4. Redistribution. You may reproduce and distribute copies of the + Work or Derivative Works thereof in any medium, with or without + modifications, and in Source or Object form, provided that You + meet the following conditions: + + (a) You must give any other recipients of the Work or + Derivative Works a copy of this License; and + + (b) You must cause any modified files to carry prominent notices + stating that You changed the files; and + + (c) You must retain, in the Source form of any Derivative Works + that You distribute, all copyright, patent, trademark, and + attribution notices from the Source form of the Work, + excluding those notices that do not pertain to any part of + the Derivative Works; and + + (d) If the Work includes a "NOTICE" text file as part of its + distribution, then any Derivative Works that You distribute must + include a readable copy of the attribution notices contained + within such NOTICE file, excluding those notices that do not + pertain to any part of the Derivative Works, in at least one + of the following places: within a NOTICE text file distributed + as part of the Derivative Works; within the Source form or + documentation, if provided along with the Derivative Works; or, + within a display generated by the Derivative Works, if and + wherever such third-party notices normally appear. The contents + of the NOTICE file are for informational purposes only and + do not modify the License. You may add Your own attribution + notices within Derivative Works that You distribute, alongside + or as an addendum to the NOTICE text from the Work, provided + that such additional attribution notices cannot be construed + as modifying the License. + + You may add Your own copyright statement to Your modifications and + may provide additional or different license terms and conditions + for use, reproduction, or distribution of Your modifications, or + for any such Derivative Works as a whole, provided Your use, + reproduction, and distribution of the Work otherwise complies with + the conditions stated in this License. + +5. Submission of Contributions. Unless You explicitly state otherwise, + any Contribution intentionally submitted for inclusion in the Work + by You to the Licensor shall be under the terms and conditions of + this License, without any additional terms or conditions. + Notwithstanding the above, nothing herein shall supersede or modify + the terms of any separate license agreement you may have executed + with Licensor regarding such Contributions. + +6. Trademarks. This License does not grant permission to use the trade + names, trademarks, service marks, or product names of the Licensor, + except as required for reasonable and customary use in describing the + origin of the Work and reproducing the content of the NOTICE file. + +7. Disclaimer of Warranty. Unless required by applicable law or + agreed to in writing, Licensor provides the Work (and each + Contributor provides its Contributions) on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or + implied, including, without limitation, any warranties or conditions + of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A + PARTICULAR PURPOSE. You are solely responsible for determining the + appropriateness of using or redistributing the Work and assume any + risks associated with Your exercise of permissions under this License. + +8. Limitation of Liability. In no event and under no legal theory, + whether in tort (including negligence), contract, or otherwise, + unless required by applicable law (such as deliberate and grossly + negligent acts) or agreed to in writing, shall any Contributor be + liable to You for damages, including any direct, indirect, special, + incidental, or consequential damages of any character arising as a + result of this License or out of the use or inability to use the + Work (including but not limited to damages for loss of goodwill, + work stoppage, computer failure or malfunction, or any and all + other commercial damages or losses), even if such Contributor + has been advised of the possibility of such damages. + +9. Accepting Warranty or Additional Liability. While redistributing + the Work or Derivative Works thereof, You may choose to offer, + and charge a fee for, acceptance of support, warranty, indemnity, + or other liability obligations and/or rights consistent with this + License. However, in accepting such obligations, You may act only + on Your own behalf and on Your sole responsibility, not on behalf + of any other Contributor, and only if You agree to indemnify, + defend, and hold each Contributor harmless for any liability + incurred by, or claims asserted against, such Contributor by reason + of your accepting any such warranty or additional liability. + +END OF TERMS AND CONDITIONS diff --git a/tools/unsafe-finder/LICENSE-MIT b/tools/unsafe-finder/LICENSE-MIT new file mode 100644 index 0000000000000..31aa79387f27e --- /dev/null +++ b/tools/unsafe-finder/LICENSE-MIT @@ -0,0 +1,23 @@ +Permission is hereby granted, free of charge, to any +person obtaining a copy of this software and associated +documentation files (the "Software"), to deal in the +Software without restriction, including without +limitation the rights to use, copy, modify, merge, +publish, distribute, sublicense, and/or sell copies of +the Software, and to permit persons to whom the Software +is furnished to do so, subject to the following +conditions: + +The above copyright notice and this permission notice +shall be included in all copies or substantial portions +of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF +ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED +TO THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A +PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT +SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY +CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION +OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR +IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER +DEALINGS IN THE SOFTWARE. diff --git a/tools/unsafe-finder/README.md b/tools/unsafe-finder/README.md new file mode 100644 index 0000000000000..e59ddb7bbe6c2 --- /dev/null +++ b/tools/unsafe-finder/README.md @@ -0,0 +1,49 @@ +# Unsafe finder + +This tool parses a Rust file and identifies three types of functions: +1. those that belong to impls and are `pub unsafe`; +2. those that belong to impls and are *not* `unsafe` but contain `unsafe` code; and, +3. those that are default functions belonging to traits and contain `unsafe` code. + +The clippy `missing_safety_doc` lint nags developers to add +(plain-text) safety comments to functions in category (1), with a +configuration option that will make clippy also complain about private +unsafe functions (default false). For the purpose of verifying the +Rust standard library, such functions should have contracts and be +verified against them. + +For categories (2) and (3), the unsafety is encapsulated in the +function; there must be some reason that the unsafe code in the +function is actually OK. (See +https://github.com/rust-lang/rust-clippy/issues/9330 for more +discussion on this issue). To verify the Rust standard library, one +must verify that there is no undefined behaviour triggered by the +unsafe code, probably by verifying the stated reason that the code is +OK. + +There are some related metrics which are automatically generated in the +`verify-rust-std` project. Those metrics live in `scritps/kani-std-analysis/metrics-data-core.json`. + +# Command-line arguments + +This tool takes a directory or a list of .rs files as input and prints +out a list of impls and traits that have functions in categories (1) +through (3), as well as the involved functions. + +``` +$ target/debug/unsafe-finder rc.rs +impl Rc {} +--- unsafe-containing fn inner +--- unsafe-containing fn into_inner_with_allocator + +impl Rc {} +--- unsafe-containing fn new +--- unsafe-containing fn new_uninit +--- unsafe-containing fn new_zeroed +--- unsafe-containing fn try_new +--- unsafe-containing fn try_new_uninit +--- unsafe-containing fn try_new_zeroed +--- unsafe-containing fn pin + +etc +``` diff --git a/tools/unsafe-finder/src/main.rs b/tools/unsafe-finder/src/main.rs new file mode 100644 index 0000000000000..0587a4bbf947e --- /dev/null +++ b/tools/unsafe-finder/src/main.rs @@ -0,0 +1,197 @@ +use syn::ImplItem; +use syn::Item::Impl; +use syn::ItemImpl; + +use syn::Item::Trait; +use syn::ItemTrait; +use syn::TraitItem; + +use syn::visit; +use syn::visit::Visit; + +use std::env; +use std::fs; +use std::io; +use std::process; +use std::path::Path; + +struct StmtVisitor { + found_unsafe: bool, +} + +impl<'ast> Visit<'ast> for StmtVisitor { + fn visit_expr_unsafe(&mut self, i: &'ast syn::ExprUnsafe) { + self.found_unsafe = true; + visit::visit_expr_unsafe(self, i); + } +} + +fn print_pub_unsafe_and_unsafe_containing_fns(ii: ItemImpl) { + let mut interesting = false; + let mut pub_unsafe_fns = Vec::new(); + let mut unsafe_containing_fns = Vec::new(); + for item in &ii.items { + match item { + ImplItem::Fn(f) => + { + // record all pub unsafe functions + if matches!(f.vis, syn::Visibility::Public(_)) && matches!(f.sig.unsafety, Some(_)) + { + interesting = true; + pub_unsafe_fns.push(format!("--- pub unsafe fn {}", f.sig.ident)); + } + // record functions that contain unsafe code in their bodies but that are not marked unsafe + else if matches!(f.sig.unsafety, None) { + let mut sv = StmtVisitor { + found_unsafe: false, + }; + sv.visit_block(&f.block); + if sv.found_unsafe { + interesting = true; + unsafe_containing_fns + .push(format!("--- unsafe-containing fn {}", f.sig.ident)); + } + } + } + _ => (), + } + } + if interesting { + // create an empty impl with the same name as ii + let mut i_copy = ii.clone(); + i_copy.items = Vec::new(); + let file = syn::File { + attrs: vec![], + items: vec![Impl(i_copy)], + shebang: None, + }; + print!("{}", prettyplease::unparse(&file)); + pub_unsafe_fns.iter().for_each(|s| { + println!("{}", s); + }); + unsafe_containing_fns.iter().for_each(|s| { + println!("{}", s); + }); + println!(); + } else { + // println!("--- nothing interesting here"); + } +} + +fn print_trait_unsafe_containing_fns(it: ItemTrait) { + let mut interesting = false; + let mut unsafe_containing_fns = Vec::new(); + for item in &it.items { + match item { + TraitItem::Fn(f) => + // record functions that contain unsafe code in their bodies but that are not marked unsafe + { + if matches!(f.sig.unsafety, None) { + let mut sv = StmtVisitor { + found_unsafe: false, + }; + if let Some(d) = &f.default { + sv.visit_block(&d); + } + if sv.found_unsafe { + interesting = true; + unsafe_containing_fns + .push(format!("--- unsafe-containing fn {}", f.sig.ident)); + } + } + } + _ => (), + } + } + if interesting { + let mut i_copy = it.clone(); + i_copy.items = Vec::new(); + let file = syn::File { + attrs: vec![], + items: vec![Trait(i_copy)], + shebang: None, + }; + print!("{}", prettyplease::unparse(&file)); + unsafe_containing_fns.iter().for_each(|s| { + println!("{}", s); + }); + println!(); + } else { + // println!("--- nothing interesting here"); + } +} + +fn handle_file(path:&Path) { + if !path.to_str().unwrap().ends_with(".rs") { + return; + } + + println!("# Unsafe usages in file {}", path.display()); + let src = fs::read_to_string(&path).expect("unable to read file"); + let syntax = syn::parse_file(&src).expect("unable to parse file"); + + for item in syntax.items { + match item { + Impl(im) => print_pub_unsafe_and_unsafe_containing_fns(im), + Trait(t) => print_trait_unsafe_containing_fns(t), + _ => (), + } + } +} + +fn handle_dir(path:&Path) -> io::Result<()> { + // https://users.rust-lang.org/t/testable-way-to-iterate-over-a-directory/81440 + let mut dirs = Vec::new(); + let mut dir_index = 0; + let mut dir_reader = fs::read_dir(path)?; + let mut had_files = false; + loop { + match dir_reader.next() { + Some(entry) => { + let cur_path = entry?.path(); + had_files = true; + if cur_path.is_dir() { + dirs.push(cur_path); + continue; + } + + if cur_path.is_file() { + handle_file(&cur_path); + continue; + } + } + _ => { + if !had_files && !dirs.is_empty() { + handle_file(&dirs[(dir_index - 1).max(0)].to_owned()); + } + if dir_index == dirs.len() { + break; + } + dir_reader = dirs[dir_index].read_dir()?; + had_files = false; + dir_index += 1; + } + } + } + + Ok(()) +} + +fn main() { + let mut args = env::args(); + let _ = args.next(); // executable name + + if args.len() == 0 { + eprintln!("Usage: unsafe-finder [directory | filename.rs]*"); + process::exit(1); + } + + for arg in args { + let path = Path::new(&arg); + if path.is_file() { + handle_file(&path); + } else if path.is_dir() { + handle_dir(&path).unwrap(); + } + } +} From 830be81e03ff284ee4b7901ea2d76153e8411da4 Mon Sep 17 00:00:00 2001 From: Patrick Lam Date: Tue, 23 Sep 2025 20:05:52 +1200 Subject: [PATCH 02/16] use output from std-analysis.sh to generate lists of unsafe functions instead of doing analysis from first principles --- tools/unsafe-finder/Cargo.toml | 4 + tools/unsafe-finder/src/main.rs | 333 ++++++++++++++++++++------------ 2 files changed, 216 insertions(+), 121 deletions(-) diff --git a/tools/unsafe-finder/Cargo.toml b/tools/unsafe-finder/Cargo.toml index 7c3284b8f73b0..1630694e76dd1 100644 --- a/tools/unsafe-finder/Cargo.toml +++ b/tools/unsafe-finder/Cargo.toml @@ -6,3 +6,7 @@ edition = "2024" [dependencies] prettyplease = "0.2.32" syn = {version = "2.0.101", features = ["full", "extra-traits", "visit"]} +csv = "1.1" +serde = { version = "1.0.55", features = ["derive"] } +regex = "1.11.2" +itertools = "0.14.0" diff --git a/tools/unsafe-finder/src/main.rs b/tools/unsafe-finder/src/main.rs index 0587a4bbf947e..dfdb8dc6263fc 100644 --- a/tools/unsafe-finder/src/main.rs +++ b/tools/unsafe-finder/src/main.rs @@ -1,142 +1,220 @@ -use syn::ImplItem; -use syn::Item::Impl; -use syn::ItemImpl; - -use syn::Item::Trait; -use syn::ItemTrait; -use syn::TraitItem; - -use syn::visit; -use syn::visit::Visit; - use std::env; use std::fs; +use std::error::Error; use std::io; use std::process; use std::path::Path; -struct StmtVisitor { - found_unsafe: bool, +use std::collections::HashMap; + +use itertools::Itertools; + +use serde::Serialize; +use serde::Deserialize; + +use regex::Regex; + +// from kani repo's tools/scanner/src/analysis.rs: +#[derive(Clone, Debug, Serialize, Deserialize)] +struct FnStats { + name: String, + is_unsafe: Option, + has_unsafe_ops: Option, + has_unsupported_input: Option, + has_loop_or_iterator: Option, + is_public: Option, +} + +#[derive(Clone)] +struct StructuredFnName { + krate: String, + module_path: Vec, + type_parameters: Vec, + item: String, } -impl<'ast> Visit<'ast> for StmtVisitor { - fn visit_expr_unsafe(&mut self, i: &'ast syn::ExprUnsafe) { - self.found_unsafe = true; - visit::visit_expr_unsafe(self, i); +#[derive(PartialOrd, Ord, Hash, Eq, PartialEq)] +struct CrateAndModules { + krate: String, + module_path: Vec +} + +fn split_by_double_colons(s:&str) -> Vec { + let mut bracket_level = 0; + let mut current_string = String::new(); + let mut previous_strings = vec![]; + let mut colons = 0; + for c in s.chars() { + current_string.push(c); + match c { + '<' => bracket_level += 1, + '>' => bracket_level -= 1, + ':' => { + if bracket_level > 0 { continue; } + colons += 1; + if colons == 2 { + colons = 0; + previous_strings.push(current_string[..current_string.len()-2].to_string()); + current_string.clear(); + }}, + _ => () + } } + previous_strings.push(current_string.clone()); + previous_strings } -fn print_pub_unsafe_and_unsafe_containing_fns(ii: ItemImpl) { - let mut interesting = false; - let mut pub_unsafe_fns = Vec::new(); - let mut unsafe_containing_fns = Vec::new(); - for item in &ii.items { - match item { - ImplItem::Fn(f) => - { - // record all pub unsafe functions - if matches!(f.vis, syn::Visibility::Public(_)) && matches!(f.sig.unsafety, Some(_)) - { - interesting = true; - pub_unsafe_fns.push(format!("--- pub unsafe fn {}", f.sig.ident)); - } - // record functions that contain unsafe code in their bodies but that are not marked unsafe - else if matches!(f.sig.unsafety, None) { - let mut sv = StmtVisitor { - found_unsafe: false, - }; - sv.visit_block(&f.block); - if sv.found_unsafe { - interesting = true; - unsafe_containing_fns - .push(format!("--- unsafe-containing fn {}", f.sig.ident)); - } - } - } - _ => (), - } - } - if interesting { - // create an empty impl with the same name as ii - let mut i_copy = ii.clone(); - i_copy.items = Vec::new(); - let file = syn::File { - attrs: vec![], - items: vec![Impl(i_copy)], - shebang: None, - }; - print!("{}", prettyplease::unparse(&file)); - pub_unsafe_fns.iter().for_each(|s| { - println!("{}", s); - }); - unsafe_containing_fns.iter().for_each(|s| { - println!("{}", s); - }); - println!(); - } else { - // println!("--- nothing interesting here"); +fn split_by_commas(s:&str) -> Vec { + let mut bracket_level = 0; + let mut parens_level = 0; + let mut current_string = String::new(); + let mut previous_strings = vec![]; + for c in s.chars() { + current_string.push(c); + match c { + '<' => bracket_level += 1, + '>' => bracket_level -= 1, + '(' => parens_level += 1, + ')' => parens_level -= 1, + ',' => { + if bracket_level > 0 || parens_level > 0 { continue; } + previous_strings.push(current_string[..current_string.len()-1].trim().to_string()); + current_string.clear(); + }, + _ => () + } } + previous_strings.push(current_string.trim().to_string().clone()); + previous_strings } -fn print_trait_unsafe_containing_fns(it: ItemTrait) { - let mut interesting = false; - let mut unsafe_containing_fns = Vec::new(); - for item in &it.items { - match item { - TraitItem::Fn(f) => - // record functions that contain unsafe code in their bodies but that are not marked unsafe - { - if matches!(f.sig.unsafety, None) { - let mut sv = StmtVisitor { - found_unsafe: false, - }; - if let Some(d) = &f.default { - sv.visit_block(&d); - } - if sv.found_unsafe { - interesting = true; - unsafe_containing_fns - .push(format!("--- unsafe-containing fn {}", f.sig.ident)); - } - } - } - _ => (), - } - } - if interesting { - let mut i_copy = it.clone(); - i_copy.items = Vec::new(); - let file = syn::File { - attrs: vec![], - items: vec![Trait(i_copy)], - shebang: None, - }; - print!("{}", prettyplease::unparse(&file)); - unsafe_containing_fns.iter().for_each(|s| { - println!("{}", s); - }); - println!(); - } else { - // println!("--- nothing interesting here"); +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn colons_singleton() { + let result = split_by_double_colons("a"); + assert_eq!(result, ["a"]); + } + + #[test] + fn colons_no_brackets() { + let result = split_by_double_colons("one::two"); + assert_eq!(result, ["one", "two"]); + } + + #[test] + fn colons_brackets_no_colons() { + let result = split_by_double_colons("one::::three"); + assert_eq!(result, ["one", "", "three"]); + } + + #[test] + fn colons_brackets_with_colons() { + let result = split_by_double_colons("one::::three"); + assert_eq!(result, ["one", "", "three"]); + } + + #[test] + fn commas_singleton() { + let result = split_by_commas("a"); + assert_eq!(result, ["a"]); + } + + #[test] + fn commas_brackets() { + let result = split_by_commas(""); + assert_eq!(result, [""]); + } + + #[test] + fn commas_no_brackets() { + let result = split_by_commas("a, b"); + assert_eq!(result, ["a","b"]); + } + + #[test] + fn commas_parens() { + let result = split_by_commas("(a,b)"); + assert_eq!(result, ["(a,b)"]); + } + + #[test] + fn commas_unmatched() { + let result = split_by_commas(" StructuredFnName { + let brackets_re = Regex::new(r"<(.+)>").unwrap(); + + let parts:Vec = split_by_double_colons(&raw_name).into_iter().rev().collect(); + let mut parts_index = 0; + let item = &parts[parts_index]; parts_index += 1; + let tp = &parts[parts_index].as_str(); + let type_parameters = if brackets_re.is_match(tp) { + let tp_commas = &brackets_re.captures(tp).unwrap(); + parts_index += 1; + split_by_commas(&tp_commas[1]).into_iter().map(|x| x.to_string()).collect() + } else { + vec![] + }; + let mut mp = vec![]; + while parts_index < parts.len() { + mp.push(parts[parts_index].to_string()); + parts_index += 1; + } + let kr = match mp.pop() { + Some(k) => k, + None => "".to_string() + }; + + StructuredFnName { + krate: kr, + module_path: mp.into_iter().rev().collect(), + type_parameters: type_parameters.into_iter().map(|x| x.to_string()).collect(), + item: item.to_string() } +} + +fn handle_file(path:&Path) -> Result<(), Box> { + let path_contents = fs::read_to_string(&path).expect("unable to read file"); + let mut rdr = csv::ReaderBuilder::new().delimiter(b';').from_reader(path_contents.as_bytes()); println!("# Unsafe usages in file {}", path.display()); - let src = fs::read_to_string(&path).expect("unable to read file"); - let syntax = syn::parse_file(&src).expect("unable to parse file"); - for item in syntax.items { - match item { - Impl(im) => print_pub_unsafe_and_unsafe_containing_fns(im), - Trait(t) => print_trait_unsafe_containing_fns(t), - _ => (), - } + let mut fns_by_crate_and_modules: HashMap> = HashMap::new(); + + for result in rdr.deserialize() { + let fn_stats: FnStats = result?; + if matches!(fn_stats.is_unsafe, Some(true)) { + let structured_fn_name = parse_fn_name(fn_stats.name); + let krate_and_module_path = CrateAndModules { + krate: structured_fn_name.krate.clone(), + module_path: structured_fn_name.module_path.clone() + }; + match fns_by_crate_and_modules.get_mut(&krate_and_module_path) { + Some(fns) => fns.push(structured_fn_name.clone()), + None => { fns_by_crate_and_modules.insert(krate_and_module_path, vec![structured_fn_name.clone()]); } + } + } } + + for krm in fns_by_crate_and_modules.keys().sorted() { + println!("crate {}, modules {:?}", krm.krate, krm.module_path); + if let Some(fns) = fns_by_crate_and_modules.get(krm) { + for structured_fn_name in fns { + println!("--- unsafe-containing fn {}", structured_fn_name.item); + if !structured_fn_name.type_parameters.is_empty() { + println!(" type parameters {:?}", structured_fn_name.type_parameters); + } + } + } + } + + Ok(()) } fn handle_dir(path:&Path) -> io::Result<()> { @@ -156,13 +234,20 @@ fn handle_dir(path:&Path) -> io::Result<()> { } if cur_path.is_file() { - handle_file(&cur_path); + if let Err(err) = handle_file(&cur_path) { + println!("error processing {}: {}", cur_path.display(), err); + process::exit(1); + } continue; } } _ => { if !had_files && !dirs.is_empty() { - handle_file(&dirs[(dir_index - 1).max(0)].to_owned()); + let target = dirs[(dir_index - 1).max(0)].to_owned(); + if let Err(err) = handle_file(&target) { + println!("error processing {}: {}", target.display(), err); + process::exit(1); + } } if dir_index == dirs.len() { break; @@ -182,16 +267,22 @@ fn main() { let _ = args.next(); // executable name if args.len() == 0 { - eprintln!("Usage: unsafe-finder [directory | filename.rs]*"); + // should we only handle files named "_scan_functions.csv"? + eprintln!("Usage: unsafe-finder [[prefix]_scan_functions.csv]*"); process::exit(1); } for arg in args { let path = Path::new(&arg); if path.is_file() { - handle_file(&path); + if let Err(err) = handle_file(&path) { + eprintln!("error processing {}: {}", arg, err); + process::exit(1); + } } else if path.is_dir() { handle_dir(&path).unwrap(); + } else { + eprintln!("could not open {}", arg); } } } From ab22aeac6e6e5ab5ae061afcbc90e6d943a57c76 Mon Sep 17 00:00:00 2001 From: Patrick Lam Date: Wed, 24 Sep 2025 21:01:32 +1200 Subject: [PATCH 03/16] also print fns with unsafe ops, and parse trait impls --- tools/unsafe-finder/src/main.rs | 28 +++++++++++++++++++++++----- 1 file changed, 23 insertions(+), 5 deletions(-) diff --git a/tools/unsafe-finder/src/main.rs b/tools/unsafe-finder/src/main.rs index dfdb8dc6263fc..9960578fc36fb 100644 --- a/tools/unsafe-finder/src/main.rs +++ b/tools/unsafe-finder/src/main.rs @@ -27,6 +27,7 @@ struct FnStats { #[derive(Clone)] struct StructuredFnName { + trait_impl: Option<(String, String)>, krate: String, module_path: Vec, type_parameters: Vec, @@ -148,9 +149,22 @@ mod tests { } fn parse_fn_name(raw_name:String) -> StructuredFnName { + let trait_impl_re = Regex::new(r"<(.+) as (.+)>").unwrap(); let brackets_re = Regex::new(r"<(.+)>").unwrap(); let parts:Vec = split_by_double_colons(&raw_name).into_iter().rev().collect(); + + if parts.len() == 2 && trait_impl_re.is_match(&parts[1]) { + let ti_captures = trait_impl_re.captures(&parts[1]).unwrap(); + return StructuredFnName { + trait_impl: Some((ti_captures[1].to_string(), ti_captures[2].to_string())), + krate: "".to_string(), + module_path: vec![], + type_parameters: vec![], + item: parts[0].to_string() + } + } + let mut parts_index = 0; let item = &parts[parts_index]; parts_index += 1; let tp = &parts[parts_index].as_str(); @@ -172,10 +186,11 @@ fn parse_fn_name(raw_name:String) -> StructuredFnName { }; StructuredFnName { - krate: kr, - module_path: mp.into_iter().rev().collect(), - type_parameters: type_parameters.into_iter().map(|x| x.to_string()).collect(), - item: item.to_string() + trait_impl: None, + krate: kr, + module_path: mp.into_iter().rev().collect(), + type_parameters: type_parameters.into_iter().map(|x| x.to_string()).collect(), + item: item.to_string() } } @@ -189,7 +204,7 @@ fn handle_file(path:&Path) -> Result<(), Box> { for result in rdr.deserialize() { let fn_stats: FnStats = result?; - if matches!(fn_stats.is_unsafe, Some(true)) { + if matches!(fn_stats.is_unsafe, Some(true)) || matches!(fn_stats.has_unsafe_ops, Some(true)) { let structured_fn_name = parse_fn_name(fn_stats.name); let krate_and_module_path = CrateAndModules { krate: structured_fn_name.krate.clone(), @@ -207,6 +222,9 @@ fn handle_file(path:&Path) -> Result<(), Box> { if let Some(fns) = fns_by_crate_and_modules.get(krm) { for structured_fn_name in fns { println!("--- unsafe-containing fn {}", structured_fn_name.item); + if let Some(ti) = &structured_fn_name.trait_impl { + println!(" trait {} as {}", ti.0, ti.1); + } else {} if !structured_fn_name.type_parameters.is_empty() { println!(" type parameters {:?}", structured_fn_name.type_parameters); } From dfb4dfd6df1a35b9038736e1b32efb5ee63f2d6a Mon Sep 17 00:00:00 2001 From: Patrick Lam Date: Thu, 25 Sep 2025 11:36:57 +1200 Subject: [PATCH 04/16] no krates --- tools/unsafe-finder/src/main.rs | 33 ++++++++------------------------- 1 file changed, 8 insertions(+), 25 deletions(-) diff --git a/tools/unsafe-finder/src/main.rs b/tools/unsafe-finder/src/main.rs index 9960578fc36fb..2515e639b7c0a 100644 --- a/tools/unsafe-finder/src/main.rs +++ b/tools/unsafe-finder/src/main.rs @@ -27,19 +27,12 @@ struct FnStats { #[derive(Clone)] struct StructuredFnName { - trait_impl: Option<(String, String)>, - krate: String, + trait_impl: Option<(String, String)>, // type as trait module_path: Vec, type_parameters: Vec, item: String, } -#[derive(PartialOrd, Ord, Hash, Eq, PartialEq)] -struct CrateAndModules { - krate: String, - module_path: Vec -} - fn split_by_double_colons(s:&str) -> Vec { let mut bracket_level = 0; let mut current_string = String::new(); @@ -158,7 +151,6 @@ fn parse_fn_name(raw_name:String) -> StructuredFnName { let ti_captures = trait_impl_re.captures(&parts[1]).unwrap(); return StructuredFnName { trait_impl: Some((ti_captures[1].to_string(), ti_captures[2].to_string())), - krate: "".to_string(), module_path: vec![], type_parameters: vec![], item: parts[0].to_string() @@ -180,14 +172,9 @@ fn parse_fn_name(raw_name:String) -> StructuredFnName { mp.push(parts[parts_index].to_string()); parts_index += 1; } - let kr = match mp.pop() { - Some(k) => k, - None => "".to_string() - }; StructuredFnName { trait_impl: None, - krate: kr, module_path: mp.into_iter().rev().collect(), type_parameters: type_parameters.into_iter().map(|x| x.to_string()).collect(), item: item.to_string() @@ -200,30 +187,26 @@ fn handle_file(path:&Path) -> Result<(), Box> { println!("# Unsafe usages in file {}", path.display()); - let mut fns_by_crate_and_modules: HashMap> = HashMap::new(); + let mut fns_by_modules: HashMap, Vec> = HashMap::new(); for result in rdr.deserialize() { let fn_stats: FnStats = result?; if matches!(fn_stats.is_unsafe, Some(true)) || matches!(fn_stats.has_unsafe_ops, Some(true)) { let structured_fn_name = parse_fn_name(fn_stats.name); - let krate_and_module_path = CrateAndModules { - krate: structured_fn_name.krate.clone(), - module_path: structured_fn_name.module_path.clone() - }; - match fns_by_crate_and_modules.get_mut(&krate_and_module_path) { + match fns_by_modules.get_mut(&structured_fn_name.module_path) { Some(fns) => fns.push(structured_fn_name.clone()), - None => { fns_by_crate_and_modules.insert(krate_and_module_path, vec![structured_fn_name.clone()]); } + None => { fns_by_modules.insert(structured_fn_name.module_path.clone(), vec![structured_fn_name.clone()]); } } } } - for krm in fns_by_crate_and_modules.keys().sorted() { - println!("crate {}, modules {:?}", krm.krate, krm.module_path); - if let Some(fns) = fns_by_crate_and_modules.get(krm) { + for mp in fns_by_modules.keys().sorted() { + println!("modules {:?}", mp); + if let Some(fns) = fns_by_modules.get(mp) { for structured_fn_name in fns { println!("--- unsafe-containing fn {}", structured_fn_name.item); if let Some(ti) = &structured_fn_name.trait_impl { - println!(" trait {} as {}", ti.0, ti.1); + println!(" trait impl: type {} as trait {}", ti.0, ti.1); } else {} if !structured_fn_name.type_parameters.is_empty() { println!(" type parameters {:?}", structured_fn_name.type_parameters); From 255e8f2c7839a871b89a83f5315f4f61dad48234 Mon Sep 17 00:00:00 2001 From: Patrick Lam Date: Thu, 25 Sep 2025 11:41:31 +1200 Subject: [PATCH 05/16] no directories --- tools/unsafe-finder/src/main.rs | 60 +++------------------------------ 1 file changed, 4 insertions(+), 56 deletions(-) diff --git a/tools/unsafe-finder/src/main.rs b/tools/unsafe-finder/src/main.rs index 2515e639b7c0a..55b251f62b5df 100644 --- a/tools/unsafe-finder/src/main.rs +++ b/tools/unsafe-finder/src/main.rs @@ -1,7 +1,6 @@ use std::env; use std::fs; use std::error::Error; -use std::io; use std::process; use std::path::Path; @@ -218,51 +217,6 @@ fn handle_file(path:&Path) -> Result<(), Box> { Ok(()) } -fn handle_dir(path:&Path) -> io::Result<()> { - // https://users.rust-lang.org/t/testable-way-to-iterate-over-a-directory/81440 - let mut dirs = Vec::new(); - let mut dir_index = 0; - let mut dir_reader = fs::read_dir(path)?; - let mut had_files = false; - loop { - match dir_reader.next() { - Some(entry) => { - let cur_path = entry?.path(); - had_files = true; - if cur_path.is_dir() { - dirs.push(cur_path); - continue; - } - - if cur_path.is_file() { - if let Err(err) = handle_file(&cur_path) { - println!("error processing {}: {}", cur_path.display(), err); - process::exit(1); - } - continue; - } - } - _ => { - if !had_files && !dirs.is_empty() { - let target = dirs[(dir_index - 1).max(0)].to_owned(); - if let Err(err) = handle_file(&target) { - println!("error processing {}: {}", target.display(), err); - process::exit(1); - } - } - if dir_index == dirs.len() { - break; - } - dir_reader = dirs[dir_index].read_dir()?; - had_files = false; - dir_index += 1; - } - } - } - - Ok(()) -} - fn main() { let mut args = env::args(); let _ = args.next(); // executable name @@ -274,16 +228,10 @@ fn main() { } for arg in args { - let path = Path::new(&arg); - if path.is_file() { - if let Err(err) = handle_file(&path) { - eprintln!("error processing {}: {}", arg, err); - process::exit(1); - } - } else if path.is_dir() { - handle_dir(&path).unwrap(); - } else { - eprintln!("could not open {}", arg); + let path = Path::new(&arg); + if let Err(err) = handle_file(&path) { + eprintln!("error processing {}: {}", arg, err); + process::exit(1); } } } From cfd4bca719a7b44ae34390492309a2bb5f2c108f Mon Sep 17 00:00:00 2001 From: Patrick Lam Date: Thu, 25 Sep 2025 12:06:19 +1200 Subject: [PATCH 06/16] print [pub] for public fns --- tools/unsafe-finder/src/main.rs | 13 ++++++++----- 1 file changed, 8 insertions(+), 5 deletions(-) diff --git a/tools/unsafe-finder/src/main.rs b/tools/unsafe-finder/src/main.rs index 55b251f62b5df..a3037a10dd443 100644 --- a/tools/unsafe-finder/src/main.rs +++ b/tools/unsafe-finder/src/main.rs @@ -30,6 +30,7 @@ struct StructuredFnName { module_path: Vec, type_parameters: Vec, item: String, + is_public: bool } fn split_by_double_colons(s:&str) -> Vec { @@ -140,7 +141,7 @@ mod tests { } } -fn parse_fn_name(raw_name:String) -> StructuredFnName { +fn parse_fn_name(raw_name:String, is_public:bool) -> StructuredFnName { let trait_impl_re = Regex::new(r"<(.+) as (.+)>").unwrap(); let brackets_re = Regex::new(r"<(.+)>").unwrap(); @@ -152,7 +153,8 @@ fn parse_fn_name(raw_name:String) -> StructuredFnName { trait_impl: Some((ti_captures[1].to_string(), ti_captures[2].to_string())), module_path: vec![], type_parameters: vec![], - item: parts[0].to_string() + item: parts[0].to_string(), + is_public: is_public } } @@ -176,7 +178,8 @@ fn parse_fn_name(raw_name:String) -> StructuredFnName { trait_impl: None, module_path: mp.into_iter().rev().collect(), type_parameters: type_parameters.into_iter().map(|x| x.to_string()).collect(), - item: item.to_string() + item: item.to_string(), + is_public: is_public } } @@ -191,7 +194,7 @@ fn handle_file(path:&Path) -> Result<(), Box> { for result in rdr.deserialize() { let fn_stats: FnStats = result?; if matches!(fn_stats.is_unsafe, Some(true)) || matches!(fn_stats.has_unsafe_ops, Some(true)) { - let structured_fn_name = parse_fn_name(fn_stats.name); + let structured_fn_name = parse_fn_name(fn_stats.name, fn_stats.is_public.is_some() && fn_stats.is_public.unwrap()); match fns_by_modules.get_mut(&structured_fn_name.module_path) { Some(fns) => fns.push(structured_fn_name.clone()), None => { fns_by_modules.insert(structured_fn_name.module_path.clone(), vec![structured_fn_name.clone()]); } @@ -203,7 +206,7 @@ fn handle_file(path:&Path) -> Result<(), Box> { println!("modules {:?}", mp); if let Some(fns) = fns_by_modules.get(mp) { for structured_fn_name in fns { - println!("--- unsafe-containing fn {}", structured_fn_name.item); + println!("--- unsafe-containing fn {} {}", structured_fn_name.item, if structured_fn_name.is_public { "[pub]" } else { "" } ); if let Some(ti) = &structured_fn_name.trait_impl { println!(" trait impl: type {} as trait {}", ti.0, ti.1); } else {} From 3856e98e26e221d57875a80b61297300f4302e32 Mon Sep 17 00:00:00 2001 From: Patrick Lam Date: Thu, 25 Sep 2025 12:10:16 +1200 Subject: [PATCH 07/16] say what unsupported input is --- tools/unsafe-finder/src/main.rs | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tools/unsafe-finder/src/main.rs b/tools/unsafe-finder/src/main.rs index a3037a10dd443..bd5f43e336815 100644 --- a/tools/unsafe-finder/src/main.rs +++ b/tools/unsafe-finder/src/main.rs @@ -19,7 +19,7 @@ struct FnStats { name: String, is_unsafe: Option, has_unsafe_ops: Option, - has_unsupported_input: Option, + has_unsupported_input: Option, // i.e. a function contains coroutines, floats, fn defs, fn ptrs, interior mut, raw pointers, recursive types, and mut refs has_loop_or_iterator: Option, is_public: Option, } From f1aecbd172ff4aeed2bff8d692b41418afe63ad8 Mon Sep 17 00:00:00 2001 From: Patrick Lam Date: Tue, 31 Mar 2026 11:26:17 -0400 Subject: [PATCH 08/16] resolve merge conflict --- scripts/kani-std-analysis/std-analysis.sh | 8 +++++--- 1 file changed, 5 insertions(+), 3 deletions(-) diff --git a/scripts/kani-std-analysis/std-analysis.sh b/scripts/kani-std-analysis/std-analysis.sh index 292233883801a..6f821bb835fc7 100755 --- a/scripts/kani-std-analysis/std-analysis.sh +++ b/scripts/kani-std-analysis/std-analysis.sh @@ -4,11 +4,13 @@ # Collect some metrics related to the crates that compose the standard library. # -# Files generates so far: +# Files generated so far: # # - ${crate}_scan_overall.csv: Summary of function metrics, such as safe vs unsafe. +# - ${crate}_scan_functions.csv: Function metrics including counts of unsafe functions and allegedly-safe abstractions. # - ${crate}_scan_input_tys.csv: Detailed information about the inputs' type of each # function found in this crate. +# - ... and others. # # How we collect metrics: # @@ -18,8 +20,8 @@ set -eu # Test for platform -PLATFORM=$(uname -sm) -if [[ $PLATFORM == "Linux x86_64" ]] +PLATFORM=$(uname -msp) +if [[ $PLATFORM == "Linux x86_64 unknown" ]] then TARGET="x86_64-unknown-linux-gnu" # 'env' necessary to avoid bash built-in 'time' From c0a18b96a6f13598139ddcbcbb900e202dbbb658 Mon Sep 17 00:00:00 2001 From: Patrick Lam Date: Mon, 31 Aug 2026 02:55:26 -0700 Subject: [PATCH 09/16] Address Copilot comments: update README.md and align with actual implementation; cache regexps; skip "impl" type parameters --- tools/unsafe-finder/Cargo.toml | 2 -- tools/unsafe-finder/README.md | 43 +++++++++++++------------- tools/unsafe-finder/src/main.rs | 53 +++++++++++++++++++++++---------- 3 files changed, 59 insertions(+), 39 deletions(-) diff --git a/tools/unsafe-finder/Cargo.toml b/tools/unsafe-finder/Cargo.toml index 1630694e76dd1..cfe938ed219d8 100644 --- a/tools/unsafe-finder/Cargo.toml +++ b/tools/unsafe-finder/Cargo.toml @@ -4,8 +4,6 @@ version = "0.1.0" edition = "2024" [dependencies] -prettyplease = "0.2.32" -syn = {version = "2.0.101", features = ["full", "extra-traits", "visit"]} csv = "1.1" serde = { version = "1.0.55", features = ["derive"] } regex = "1.11.2" diff --git a/tools/unsafe-finder/README.md b/tools/unsafe-finder/README.md index e59ddb7bbe6c2..1a0c919bcea1a 100644 --- a/tools/unsafe-finder/README.md +++ b/tools/unsafe-finder/README.md @@ -1,9 +1,8 @@ # Unsafe finder -This tool parses a Rust file and identifies three types of functions: -1. those that belong to impls and are `pub unsafe`; -2. those that belong to impls and are *not* `unsafe` but contain `unsafe` code; and, -3. those that are default functions belonging to traits and contain `unsafe` code. +This tool identifies two types of functions: +1. those that belong to impls and are `pub unsafe`; and, +2. those that belong to impls and are *not* `unsafe` but contain `unsafe` code. The clippy `missing_safety_doc` lint nags developers to add (plain-text) safety comments to functions in category (1), with a @@ -12,7 +11,7 @@ unsafe functions (default false). For the purpose of verifying the Rust standard library, such functions should have contracts and be verified against them. -For categories (2) and (3), the unsafety is encapsulated in the +For category (2), the unsafety is encapsulated in the function; there must be some reason that the unsafe code in the function is actually OK. (See https://github.com/rust-lang/rust-clippy/issues/9330 for more @@ -22,28 +21,30 @@ unsafe code, probably by verifying the stated reason that the code is OK. There are some related metrics which are automatically generated in the -`verify-rust-std` project. Those metrics live in `scritps/kani-std-analysis/metrics-data-core.json`. +`verify-rust-std` project. Those metrics live in `scripts/kani-std-analysis/metrics-data-core.json`. # Command-line arguments -This tool takes a directory or a list of .rs files as input and prints +This tool takes a _scan_functions.csv file generated by the tool in `scripts/kani-std-analysis/std-analysis.sh` and prints out a list of impls and traits that have functions in categories (1) -through (3), as well as the involved functions. +through (2), as well as the involved functions. ``` -$ target/debug/unsafe-finder rc.rs -impl Rc {} ---- unsafe-containing fn inner ---- unsafe-containing fn into_inner_with_allocator - -impl Rc {} ---- unsafe-containing fn new ---- unsafe-containing fn new_uninit ---- unsafe-containing fn new_zeroed ---- unsafe-containing fn try_new ---- unsafe-containing fn try_new_uninit ---- unsafe-containing fn try_new_zeroed ---- unsafe-containing fn pin +$ target/debug/unsafe-finder /tmp/std_lib_analysis/results/std_scan_functions.csv +# Unsafe usages in file /tmp/std_lib_analysis/results/std_scan_functions.csv +modules [] +--- unsafe-containing fn as_fd [pub] + trait impl: type std::io::stdio::Stderr as trait std::os::fd::owned::AsFd +--- unsafe-containing fn write_vectored [pub] + trait impl: type std::sys::stdio::unix::Stdout as trait core::io::Write +--- unsafe-containing fn deref_mut [pub] + trait impl: type std::sync::poison::rwlock::MappedRwLockWriteGuard<'_, T> as trait core::ops::DerefMut +--- unsafe-containing fn drop [pub] + trait impl: type std::backtrace_rs::symbolize::gimli::mmap::Mmap as trait core::ops::Drop +--- unsafe-containing fn drop [pub] + trait impl: type std::sync::poison::rwlock::MappedRwLockReadGuard<'_, T> as trait core::ops::Drop +--- unsafe fn pre_exec [pub] + trait impl: type std::process::Command as trait std::os::unix::process::CommandExt etc ``` diff --git a/tools/unsafe-finder/src/main.rs b/tools/unsafe-finder/src/main.rs index bd5f43e336815..5a5502e3f740a 100644 --- a/tools/unsafe-finder/src/main.rs +++ b/tools/unsafe-finder/src/main.rs @@ -5,6 +5,7 @@ use std::process; use std::path::Path; use std::collections::HashMap; +use std::sync::LazyLock; use itertools::Itertools; @@ -30,7 +31,8 @@ struct StructuredFnName { module_path: Vec, type_parameters: Vec, item: String, - is_public: bool + is_public: bool, + typ: String, } fn split_by_double_colons(s:&str) -> Vec { @@ -140,31 +142,37 @@ mod tests { assert_eq!(result, [" = LazyLock::new(|| { + Regex::new(r"<(.+) as (.+)>").expect("invalid regex") +}); +static BRACKETS: LazyLock = LazyLock::new(|| { + Regex::new(r"<(.+)>").expect("invalid regex") +}); -fn parse_fn_name(raw_name:String, is_public:bool) -> StructuredFnName { - let trait_impl_re = Regex::new(r"<(.+) as (.+)>").unwrap(); - let brackets_re = Regex::new(r"<(.+)>").unwrap(); +fn parse_fn_name(raw_name:String, is_public:bool, is_unsafe:bool) -> StructuredFnName { + let typ = if is_unsafe { "unsafe".to_string() } else { "unsafe-containing".to_string() }; let parts:Vec = split_by_double_colons(&raw_name).into_iter().rev().collect(); - if parts.len() == 2 && trait_impl_re.is_match(&parts[1]) { - let ti_captures = trait_impl_re.captures(&parts[1]).unwrap(); + if parts.len() == 2 && TRAIT_IMPL.is_match(&parts[1]) { + let ti_captures = TRAIT_IMPL.captures(&parts[1]).unwrap(); return StructuredFnName { trait_impl: Some((ti_captures[1].to_string(), ti_captures[2].to_string())), module_path: vec![], type_parameters: vec![], item: parts[0].to_string(), - is_public: is_public + is_public: is_public, + typ: typ, } } let mut parts_index = 0; let item = &parts[parts_index]; parts_index += 1; let tp = &parts[parts_index].as_str(); - let type_parameters = if brackets_re.is_match(tp) { - let tp_commas = &brackets_re.captures(tp).unwrap(); + let type_parameters = if BRACKETS.is_match(tp) { + let tp_commas = &BRACKETS.captures(tp).unwrap(); parts_index += 1; - split_by_commas(&tp_commas[1]).into_iter().map(|x| x.to_string()).collect() + split_by_commas(&tp_commas[1]).into_iter().map(|x| x.to_string()).filter(|x| !x.starts_with("impl")).collect() } else { vec![] }; @@ -179,12 +187,13 @@ fn parse_fn_name(raw_name:String, is_public:bool) -> StructuredFnName { module_path: mp.into_iter().rev().collect(), type_parameters: type_parameters.into_iter().map(|x| x.to_string()).collect(), item: item.to_string(), - is_public: is_public + is_public: is_public, + typ: typ, } } fn handle_file(path:&Path) -> Result<(), Box> { - let path_contents = fs::read_to_string(&path).expect("unable to read file"); + let path_contents = fs::read_to_string(&path)?; let mut rdr = csv::ReaderBuilder::new().delimiter(b';').from_reader(path_contents.as_bytes()); println!("# Unsafe usages in file {}", path.display()); @@ -192,9 +201,21 @@ fn handle_file(path:&Path) -> Result<(), Box> { let mut fns_by_modules: HashMap, Vec> = HashMap::new(); for result in rdr.deserialize() { - let fn_stats: FnStats = result?; - if matches!(fn_stats.is_unsafe, Some(true)) || matches!(fn_stats.has_unsafe_ops, Some(true)) { - let structured_fn_name = parse_fn_name(fn_stats.name, fn_stats.is_public.is_some() && fn_stats.is_public.unwrap()); + let fn_stats: FnStats = result?; + let is_unsafe = matches!(fn_stats.is_unsafe, Some(true)); + if is_unsafe { + let is_public = matches!(fn_stats.is_public, Some(true)); + if is_public { + let structured_fn_name = parse_fn_name(fn_stats.name, is_public, is_unsafe); + match fns_by_modules.get_mut(&structured_fn_name.module_path) { + Some(fns) => fns.push(structured_fn_name.clone()), + None => { fns_by_modules.insert(structured_fn_name.module_path.clone(), vec![structured_fn_name.clone()]); } + } + } + } + else if !is_unsafe && matches!(fn_stats.has_unsafe_ops, Some(true)) { + let is_public = matches!(fn_stats.is_public, Some(true)); + let structured_fn_name = parse_fn_name(fn_stats.name, is_public, is_unsafe); match fns_by_modules.get_mut(&structured_fn_name.module_path) { Some(fns) => fns.push(structured_fn_name.clone()), None => { fns_by_modules.insert(structured_fn_name.module_path.clone(), vec![structured_fn_name.clone()]); } @@ -206,7 +227,7 @@ fn handle_file(path:&Path) -> Result<(), Box> { println!("modules {:?}", mp); if let Some(fns) = fns_by_modules.get(mp) { for structured_fn_name in fns { - println!("--- unsafe-containing fn {} {}", structured_fn_name.item, if structured_fn_name.is_public { "[pub]" } else { "" } ); + println!("--- {} fn {} {}", structured_fn_name.typ, structured_fn_name.item, if structured_fn_name.is_public { "[pub]" } else { "" } ); if let Some(ti) = &structured_fn_name.trait_impl { println!(" trait impl: type {} as trait {}", ti.0, ti.1); } else {} From 3a6361c50825ecc93551420ff82fc36005e3c26f Mon Sep 17 00:00:00 2001 From: Patrick Lam Date: Wed, 2 Sep 2026 10:19:30 +1200 Subject: [PATCH 10/16] don't crash on fn_name without :: and run cargo --fmt --- tools/unsafe-finder/src/main.rs | 298 +++++++++++++++++++++----------- 1 file changed, 195 insertions(+), 103 deletions(-) diff --git a/tools/unsafe-finder/src/main.rs b/tools/unsafe-finder/src/main.rs index 5a5502e3f740a..b7fbb15164d13 100644 --- a/tools/unsafe-finder/src/main.rs +++ b/tools/unsafe-finder/src/main.rs @@ -1,16 +1,16 @@ use std::env; -use std::fs; use std::error::Error; -use std::process; +use std::fs; use std::path::Path; +use std::process; use std::collections::HashMap; use std::sync::LazyLock; use itertools::Itertools; -use serde::Serialize; use serde::Deserialize; +use serde::Serialize; use regex::Regex; @@ -25,7 +25,7 @@ struct FnStats { is_public: Option, } -#[derive(Clone)] +#[derive(Clone, Debug, PartialEq)] struct StructuredFnName { trait_impl: Option<(String, String)>, // type as trait module_path: Vec, @@ -35,50 +35,59 @@ struct StructuredFnName { typ: String, } -fn split_by_double_colons(s:&str) -> Vec { +fn split_by_double_colons(s: &str) -> Vec { let mut bracket_level = 0; let mut current_string = String::new(); let mut previous_strings = vec![]; let mut colons = 0; for c in s.chars() { - current_string.push(c); - match c { - '<' => bracket_level += 1, - '>' => bracket_level -= 1, - ':' => { - if bracket_level > 0 { continue; } - colons += 1; - if colons == 2 { - colons = 0; - previous_strings.push(current_string[..current_string.len()-2].to_string()); - current_string.clear(); - }}, - _ => () - } + current_string.push(c); + match c { + '<' => bracket_level += 1, + '>' => bracket_level -= 1, + ':' => { + if bracket_level > 0 { + continue; + } + colons += 1; + if colons == 2 { + colons = 0; + previous_strings.push(current_string[..current_string.len() - 2].to_string()); + current_string.clear(); + } + } + _ => (), + } } previous_strings.push(current_string.clone()); previous_strings } -fn split_by_commas(s:&str) -> Vec { +fn split_by_commas(s: &str) -> Vec { let mut bracket_level = 0; let mut parens_level = 0; let mut current_string = String::new(); let mut previous_strings = vec![]; for c in s.chars() { - current_string.push(c); - match c { - '<' => bracket_level += 1, - '>' => bracket_level -= 1, - '(' => parens_level += 1, - ')' => parens_level -= 1, - ',' => { - if bracket_level > 0 || parens_level > 0 { continue; } - previous_strings.push(current_string[..current_string.len()-1].trim().to_string()); - current_string.clear(); - }, - _ => () - } + current_string.push(c); + match c { + '<' => bracket_level += 1, + '>' => bracket_level -= 1, + '(' => parens_level += 1, + ')' => parens_level -= 1, + ',' => { + if bracket_level > 0 || parens_level > 0 { + continue; + } + previous_strings.push( + current_string[..current_string.len() - 1] + .trim() + .to_string(), + ); + current_string.clear(); + } + _ => (), + } } previous_strings.push(current_string.trim().to_string().clone()); previous_strings @@ -87,72 +96,126 @@ fn split_by_commas(s:&str) -> Vec { #[cfg(test)] mod tests { use super::*; - + #[test] fn colons_singleton() { - let result = split_by_double_colons("a"); - assert_eq!(result, ["a"]); + let result = split_by_double_colons("a"); + assert_eq!(result, ["a"]); } #[test] fn colons_no_brackets() { - let result = split_by_double_colons("one::two"); - assert_eq!(result, ["one", "two"]); + let result = split_by_double_colons("one::two"); + assert_eq!(result, ["one", "two"]); } #[test] fn colons_brackets_no_colons() { - let result = split_by_double_colons("one::::three"); - assert_eq!(result, ["one", "", "three"]); + let result = split_by_double_colons("one::::three"); + assert_eq!(result, ["one", "", "three"]); } #[test] fn colons_brackets_with_colons() { - let result = split_by_double_colons("one::::three"); - assert_eq!(result, ["one", "", "three"]); + let result = split_by_double_colons("one::::three"); + assert_eq!(result, ["one", "", "three"]); } #[test] fn commas_singleton() { - let result = split_by_commas("a"); - assert_eq!(result, ["a"]); + let result = split_by_commas("a"); + assert_eq!(result, ["a"]); } #[test] fn commas_brackets() { - let result = split_by_commas(""); - assert_eq!(result, [""]); + let result = split_by_commas(""); + assert_eq!(result, [""]); } #[test] fn commas_no_brackets() { - let result = split_by_commas("a, b"); - assert_eq!(result, ["a","b"]); + let result = split_by_commas("a, b"); + assert_eq!(result, ["a", "b"]); } #[test] fn commas_parens() { - let result = split_by_commas("(a,b)"); - assert_eq!(result, ["(a,b)"]); + let result = split_by_commas("(a,b)"); + assert_eq!(result, ["(a,b)"]); } #[test] fn commas_unmatched() { - let result = split_by_commas("::from_mode".to_string(), + false, + false, + ); + assert_eq!( + result, + StructuredFnName { + trait_impl: Some(( + "std::fs::Permissions".to_string(), + "std::os::unix::fs::PermissionsExt".to_string() + )), + module_path: Vec::new(), + type_parameters: Vec::new(), + item: "from_mode".to_string(), + is_public: false, + typ: "unsafe-containing".to_string() + } + ); + } + + #[test] + fn parse_fn_name_insufficient_segments() { + let result = parse_fn_name("foo".to_string(), false, false); + assert_eq!( + result, + StructuredFnName { + trait_impl: None, + module_path: Vec::new(), + type_parameters: Vec::new(), + item: "foo".to_string(), + is_public: false, + typ: "unsafe-containing".to_string() + } + ); } } -static TRAIT_IMPL: LazyLock = LazyLock::new(|| { - Regex::new(r"<(.+) as (.+)>").expect("invalid regex") -}); -static BRACKETS: LazyLock = LazyLock::new(|| { - Regex::new(r"<(.+)>").expect("invalid regex") -}); - -fn parse_fn_name(raw_name:String, is_public:bool, is_unsafe:bool) -> StructuredFnName { - let typ = if is_unsafe { "unsafe".to_string() } else { "unsafe-containing".to_string() }; - - let parts:Vec = split_by_double_colons(&raw_name).into_iter().rev().collect(); +static TRAIT_IMPL: LazyLock = + LazyLock::new(|| Regex::new(r"<(.+) as (.+)>").expect("invalid regex")); +static BRACKETS: LazyLock = LazyLock::new(|| Regex::new(r"<(.+)>").expect("invalid regex")); + +fn parse_fn_name(raw_name: String, is_public: bool, is_unsafe: bool) -> StructuredFnName { + let typ = if is_unsafe { + "unsafe".to_string() + } else { + "unsafe-containing".to_string() + }; + + let parts: Vec = split_by_double_colons(&raw_name) + .into_iter() + .rev() + .collect(); + + if parts.len() == 1 { + return StructuredFnName { + trait_impl: None, + module_path: Vec::new(), + type_parameters: Vec::new(), + item: raw_name, + is_public: is_public, + typ: typ, + }; + } if parts.len() == 2 && TRAIT_IMPL.is_match(&parts[1]) { let ti_captures = TRAIT_IMPL.captures(&parts[1]).unwrap(); @@ -163,23 +226,28 @@ fn parse_fn_name(raw_name:String, is_public:bool, is_unsafe:bool) -> StructuredF item: parts[0].to_string(), is_public: is_public, typ: typ, - } + }; } let mut parts_index = 0; - let item = &parts[parts_index]; parts_index += 1; + let item = &parts[parts_index]; + parts_index += 1; let tp = &parts[parts_index].as_str(); let type_parameters = if BRACKETS.is_match(tp) { let tp_commas = &BRACKETS.captures(tp).unwrap(); - parts_index += 1; - split_by_commas(&tp_commas[1]).into_iter().map(|x| x.to_string()).filter(|x| !x.starts_with("impl")).collect() + parts_index += 1; + split_by_commas(&tp_commas[1]) + .into_iter() + .map(|x| x.to_string()) + .filter(|x| !x.starts_with("impl")) + .collect() } else { vec![] }; let mut mp = vec![]; while parts_index < parts.len() { - mp.push(parts[parts_index].to_string()); - parts_index += 1; + mp.push(parts[parts_index].to_string()); + parts_index += 1; } StructuredFnName { @@ -192,52 +260,76 @@ fn parse_fn_name(raw_name:String, is_public:bool, is_unsafe:bool) -> StructuredF } } -fn handle_file(path:&Path) -> Result<(), Box> { +fn handle_file(path: &Path) -> Result<(), Box> { let path_contents = fs::read_to_string(&path)?; - let mut rdr = csv::ReaderBuilder::new().delimiter(b';').from_reader(path_contents.as_bytes()); + let mut rdr = csv::ReaderBuilder::new() + .delimiter(b';') + .from_reader(path_contents.as_bytes()); println!("# Unsafe usages in file {}", path.display()); let mut fns_by_modules: HashMap, Vec> = HashMap::new(); - + for result in rdr.deserialize() { - let fn_stats: FnStats = result?; - let is_unsafe = matches!(fn_stats.is_unsafe, Some(true)); - if is_unsafe { + let fn_stats: FnStats = result?; + let is_unsafe = matches!(fn_stats.is_unsafe, Some(true)); + if is_unsafe { let is_public = matches!(fn_stats.is_public, Some(true)); if is_public { - let structured_fn_name = parse_fn_name(fn_stats.name, is_public, is_unsafe); - match fns_by_modules.get_mut(&structured_fn_name.module_path) { - Some(fns) => fns.push(structured_fn_name.clone()), - None => { fns_by_modules.insert(structured_fn_name.module_path.clone(), vec![structured_fn_name.clone()]); } - } + let structured_fn_name = parse_fn_name(fn_stats.name, is_public, is_unsafe); + match fns_by_modules.get_mut(&structured_fn_name.module_path) { + Some(fns) => fns.push(structured_fn_name.clone()), + None => { + fns_by_modules.insert( + structured_fn_name.module_path.clone(), + vec![structured_fn_name.clone()], + ); + } + } } - } - else if !is_unsafe && matches!(fn_stats.has_unsafe_ops, Some(true)) { + } else if !is_unsafe && matches!(fn_stats.has_unsafe_ops, Some(true)) { let is_public = matches!(fn_stats.is_public, Some(true)); - let structured_fn_name = parse_fn_name(fn_stats.name, is_public, is_unsafe); - match fns_by_modules.get_mut(&structured_fn_name.module_path) { - Some(fns) => fns.push(structured_fn_name.clone()), - None => { fns_by_modules.insert(structured_fn_name.module_path.clone(), vec![structured_fn_name.clone()]); } - } - } + let structured_fn_name = parse_fn_name(fn_stats.name, is_public, is_unsafe); + match fns_by_modules.get_mut(&structured_fn_name.module_path) { + Some(fns) => fns.push(structured_fn_name.clone()), + None => { + fns_by_modules.insert( + structured_fn_name.module_path.clone(), + vec![structured_fn_name.clone()], + ); + } + } + } } for mp in fns_by_modules.keys().sorted() { - println!("modules {:?}", mp); - if let Some(fns) = fns_by_modules.get(mp) { - for structured_fn_name in fns { - println!("--- {} fn {} {}", structured_fn_name.typ, structured_fn_name.item, if structured_fn_name.is_public { "[pub]" } else { "" } ); + println!("modules {:?}", mp); + if let Some(fns) = fns_by_modules.get(mp) { + for structured_fn_name in fns { + println!( + "--- {} fn {} {}", + structured_fn_name.typ, + structured_fn_name.item, + if structured_fn_name.is_public { + "[pub]" + } else { + "" + } + ); if let Some(ti) = &structured_fn_name.trait_impl { println!(" trait impl: type {} as trait {}", ti.0, ti.1); - } else {} - if !structured_fn_name.type_parameters.is_empty() { - println!(" type parameters {:?}", structured_fn_name.type_parameters); - } - } - } + } else { + } + if !structured_fn_name.type_parameters.is_empty() { + println!( + " type parameters {:?}", + structured_fn_name.type_parameters + ); + } + } + } } - + Ok(()) } @@ -246,7 +338,7 @@ fn main() { let _ = args.next(); // executable name if args.len() == 0 { - // should we only handle files named "_scan_functions.csv"? + // should we only handle files named "_scan_functions.csv"? eprintln!("Usage: unsafe-finder [[prefix]_scan_functions.csv]*"); process::exit(1); } @@ -254,8 +346,8 @@ fn main() { for arg in args { let path = Path::new(&arg); if let Err(err) = handle_file(&path) { - eprintln!("error processing {}: {}", arg, err); - process::exit(1); - } + eprintln!("error processing {}: {}", arg, err); + process::exit(1); + } } } From 0c5544dc3f5834e9096939449b35e288e9fc5d99 Mon Sep 17 00:00:00 2001 From: Patrick Lam Date: Wed, 2 Sep 2026 10:48:53 +1200 Subject: [PATCH 11/16] revert change to uname invocation --- scripts/kani-std-analysis/std-analysis.sh | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/scripts/kani-std-analysis/std-analysis.sh b/scripts/kani-std-analysis/std-analysis.sh index 6f821bb835fc7..8ab0f40708829 100755 --- a/scripts/kani-std-analysis/std-analysis.sh +++ b/scripts/kani-std-analysis/std-analysis.sh @@ -10,7 +10,6 @@ # - ${crate}_scan_functions.csv: Function metrics including counts of unsafe functions and allegedly-safe abstractions. # - ${crate}_scan_input_tys.csv: Detailed information about the inputs' type of each # function found in this crate. -# - ... and others. # # How we collect metrics: # @@ -20,8 +19,8 @@ set -eu # Test for platform -PLATFORM=$(uname -msp) -if [[ $PLATFORM == "Linux x86_64 unknown" ]] +PLATFORM=$(uname -sm) +if [[ $PLATFORM == "Linux x86_64" ]] then TARGET="x86_64-unknown-linux-gnu" # 'env' necessary to avoid bash built-in 'time' From ae492c20bee2a2733e7f896a51ea44950765729b Mon Sep 17 00:00:00 2001 From: Patrick Lam Date: Wed, 2 Sep 2026 10:54:42 +1200 Subject: [PATCH 12/16] apply clippy fixes --- tools/unsafe-finder/src/main.rs | 17 ++++++++--------- 1 file changed, 8 insertions(+), 9 deletions(-) diff --git a/tools/unsafe-finder/src/main.rs b/tools/unsafe-finder/src/main.rs index b7fbb15164d13..9f2fd260e285b 100644 --- a/tools/unsafe-finder/src/main.rs +++ b/tools/unsafe-finder/src/main.rs @@ -212,8 +212,8 @@ fn parse_fn_name(raw_name: String, is_public: bool, is_unsafe: bool) -> Structur module_path: Vec::new(), type_parameters: Vec::new(), item: raw_name, - is_public: is_public, - typ: typ, + is_public, + typ, }; } @@ -224,8 +224,8 @@ fn parse_fn_name(raw_name: String, is_public: bool, is_unsafe: bool) -> Structur module_path: vec![], type_parameters: vec![], item: parts[0].to_string(), - is_public: is_public, - typ: typ, + is_public, + typ, }; } @@ -255,13 +255,13 @@ fn parse_fn_name(raw_name: String, is_public: bool, is_unsafe: bool) -> Structur module_path: mp.into_iter().rev().collect(), type_parameters: type_parameters.into_iter().map(|x| x.to_string()).collect(), item: item.to_string(), - is_public: is_public, - typ: typ, + is_public, + typ, } } fn handle_file(path: &Path) -> Result<(), Box> { - let path_contents = fs::read_to_string(&path)?; + let path_contents = fs::read_to_string(path)?; let mut rdr = csv::ReaderBuilder::new() .delimiter(b';') .from_reader(path_contents.as_bytes()); @@ -318,7 +318,6 @@ fn handle_file(path: &Path) -> Result<(), Box> { ); if let Some(ti) = &structured_fn_name.trait_impl { println!(" trait impl: type {} as trait {}", ti.0, ti.1); - } else { } if !structured_fn_name.type_parameters.is_empty() { println!( @@ -345,7 +344,7 @@ fn main() { for arg in args { let path = Path::new(&arg); - if let Err(err) = handle_file(&path) { + if let Err(err) = handle_file(path) { eprintln!("error processing {}: {}", arg, err); process::exit(1); } From 80f6d689b34884dd283f1a90b0e69bc664cdc813 Mon Sep 17 00:00:00 2001 From: Patrick Lam Date: Wed, 2 Sep 2026 11:00:41 +1200 Subject: [PATCH 13/16] add tests for parse_fn --- tools/unsafe-finder/src/main.rs | 39 ++++++++++++++++++++++++++++----- 1 file changed, 33 insertions(+), 6 deletions(-) diff --git a/tools/unsafe-finder/src/main.rs b/tools/unsafe-finder/src/main.rs index 9f2fd260e285b..d9b53827cce68 100644 --- a/tools/unsafe-finder/src/main.rs +++ b/tools/unsafe-finder/src/main.rs @@ -152,7 +152,23 @@ mod tests { } #[test] - fn parse_fn_name_normal() { + fn parse_fn_name_insufficient_segments() { + let result = parse_fn_name("foo".to_string(), false, false); + assert_eq!( + result, + StructuredFnName { + trait_impl: None, + module_path: Vec::new(), + type_parameters: Vec::new(), + item: "foo".to_string(), + is_public: false, + typ: "unsafe-containing".to_string() + } + ); + } + + #[test] + fn parse_fn_name_trait_impl() { let result = parse_fn_name( "::from_mode".to_string(), false, @@ -175,15 +191,26 @@ mod tests { } #[test] - fn parse_fn_name_insufficient_segments() { - let result = parse_fn_name("foo".to_string(), false, false); + fn parse_fn_name_with_generics() { + let result = parse_fn_name( + "std::sync::mpmc::list::Channel::::len".to_string(), + false, + false, + ); assert_eq!( result, StructuredFnName { trait_impl: None, - module_path: Vec::new(), - type_parameters: Vec::new(), - item: "foo".to_string(), + module_path: [ + "std".to_string(), + "sync".to_string(), + "mpmc".to_string(), + "list".to_string(), + "Channel".to_string() + ] + .to_vec(), + type_parameters: ["T".to_string()].to_vec(), + item: "len".to_string(), is_public: false, typ: "unsafe-containing".to_string() } From 1d5a9585d817add8549e82cb4ac4d3a65aed29a9 Mon Sep 17 00:00:00 2001 From: Patrick Lam Date: Wed, 2 Sep 2026 11:12:48 +1200 Subject: [PATCH 14/16] special-case -> in split_by_commas and split_by_double_colons --- tools/unsafe-finder/src/main.rs | 283 ++++++++++++++++++-------------- 1 file changed, 157 insertions(+), 126 deletions(-) diff --git a/tools/unsafe-finder/src/main.rs b/tools/unsafe-finder/src/main.rs index d9b53827cce68..a1a153c0d596a 100644 --- a/tools/unsafe-finder/src/main.rs +++ b/tools/unsafe-finder/src/main.rs @@ -37,6 +37,7 @@ struct StructuredFnName { fn split_by_double_colons(s: &str) -> Vec { let mut bracket_level = 0; + let mut prev_was_minus = false; let mut current_string = String::new(); let mut previous_strings = vec![]; let mut colons = 0; @@ -44,7 +45,11 @@ fn split_by_double_colons(s: &str) -> Vec { current_string.push(c); match c { '<' => bracket_level += 1, - '>' => bracket_level -= 1, + '>' => { + if !prev_was_minus { + bracket_level -= 1; + } + } ':' => { if bracket_level > 0 { continue; @@ -58,6 +63,10 @@ fn split_by_double_colons(s: &str) -> Vec { } _ => (), } + prev_was_minus = false; + if c == '-' { + prev_was_minus = true + } } previous_strings.push(current_string.clone()); previous_strings @@ -66,13 +75,18 @@ fn split_by_double_colons(s: &str) -> Vec { fn split_by_commas(s: &str) -> Vec { let mut bracket_level = 0; let mut parens_level = 0; + let mut prev_was_minus = false; let mut current_string = String::new(); let mut previous_strings = vec![]; for c in s.chars() { current_string.push(c); match c { '<' => bracket_level += 1, - '>' => bracket_level -= 1, + '>' => { + if !prev_was_minus { + bracket_level -= 1; + } + } '(' => parens_level += 1, ')' => parens_level -= 1, ',' => { @@ -88,135 +102,15 @@ fn split_by_commas(s: &str) -> Vec { } _ => (), } + prev_was_minus = false; + if c == '-' { + prev_was_minus = true + } } previous_strings.push(current_string.trim().to_string().clone()); previous_strings } -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn colons_singleton() { - let result = split_by_double_colons("a"); - assert_eq!(result, ["a"]); - } - - #[test] - fn colons_no_brackets() { - let result = split_by_double_colons("one::two"); - assert_eq!(result, ["one", "two"]); - } - - #[test] - fn colons_brackets_no_colons() { - let result = split_by_double_colons("one::::three"); - assert_eq!(result, ["one", "", "three"]); - } - - #[test] - fn colons_brackets_with_colons() { - let result = split_by_double_colons("one::::three"); - assert_eq!(result, ["one", "", "three"]); - } - - #[test] - fn commas_singleton() { - let result = split_by_commas("a"); - assert_eq!(result, ["a"]); - } - - #[test] - fn commas_brackets() { - let result = split_by_commas(""); - assert_eq!(result, [""]); - } - - #[test] - fn commas_no_brackets() { - let result = split_by_commas("a, b"); - assert_eq!(result, ["a", "b"]); - } - - #[test] - fn commas_parens() { - let result = split_by_commas("(a,b)"); - assert_eq!(result, ["(a,b)"]); - } - - #[test] - fn commas_unmatched() { - let result = split_by_commas("::from_mode".to_string(), - false, - false, - ); - assert_eq!( - result, - StructuredFnName { - trait_impl: Some(( - "std::fs::Permissions".to_string(), - "std::os::unix::fs::PermissionsExt".to_string() - )), - module_path: Vec::new(), - type_parameters: Vec::new(), - item: "from_mode".to_string(), - is_public: false, - typ: "unsafe-containing".to_string() - } - ); - } - - #[test] - fn parse_fn_name_with_generics() { - let result = parse_fn_name( - "std::sync::mpmc::list::Channel::::len".to_string(), - false, - false, - ); - assert_eq!( - result, - StructuredFnName { - trait_impl: None, - module_path: [ - "std".to_string(), - "sync".to_string(), - "mpmc".to_string(), - "list".to_string(), - "Channel".to_string() - ] - .to_vec(), - type_parameters: ["T".to_string()].to_vec(), - item: "len".to_string(), - is_public: false, - typ: "unsafe-containing".to_string() - } - ); - } -} static TRAIT_IMPL: LazyLock = LazyLock::new(|| Regex::new(r"<(.+) as (.+)>").expect("invalid regex")); static BRACKETS: LazyLock = LazyLock::new(|| Regex::new(r"<(.+)>").expect("invalid regex")); @@ -377,3 +271,140 @@ fn main() { } } } + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn colons_singleton() { + let result = split_by_double_colons("a"); + assert_eq!(result, ["a"]); + } + + #[test] + fn colons_no_brackets() { + let result = split_by_double_colons("one::two"); + assert_eq!(result, ["one", "two"]); + } + + #[test] + fn colons_brackets_no_colons() { + let result = split_by_double_colons("one::::three"); + assert_eq!(result, ["one", "", "three"]); + } + + #[test] + fn colons_brackets_with_colons() { + let result = split_by_double_colons("one::::three"); + assert_eq!(result, ["one", "", "three"]); + } + + #[test] + fn colons_arrow() { + let result = split_by_double_colons("mymod::bar::Baz>::the_item"); + assert_eq!(result, ["mymod", "bar::Baz>", "the_item"]); + } + + #[test] + fn commas_singleton() { + let result = split_by_commas("a"); + assert_eq!(result, ["a"]); + } + + #[test] + fn commas_brackets() { + let result = split_by_commas(""); + assert_eq!(result, [""]); + } + + #[test] + fn commas_no_brackets() { + let result = split_by_commas("a, b"); + assert_eq!(result, ["a", "b"]); + } + + #[test] + fn commas_parens() { + let result = split_by_commas("(a,b)"); + assert_eq!(result, ["(a,b)"]); + } + + #[test] + fn commas_unmatched() { + let result = split_by_commas("bar::Baz>, the_item"); + assert_eq!(result, ["mymod", "bar::Baz>", "the_item"]); + } + + #[test] + fn parse_fn_name_insufficient_segments() { + let result = parse_fn_name("foo".to_string(), false, false); + assert_eq!( + result, + StructuredFnName { + trait_impl: None, + module_path: Vec::new(), + type_parameters: Vec::new(), + item: "foo".to_string(), + is_public: false, + typ: "unsafe-containing".to_string() + } + ); + } + + #[test] + fn parse_fn_name_trait_impl() { + let result = parse_fn_name( + "::from_mode".to_string(), + false, + false, + ); + assert_eq!( + result, + StructuredFnName { + trait_impl: Some(( + "std::fs::Permissions".to_string(), + "std::os::unix::fs::PermissionsExt".to_string() + )), + module_path: Vec::new(), + type_parameters: Vec::new(), + item: "from_mode".to_string(), + is_public: false, + typ: "unsafe-containing".to_string() + } + ); + } + + #[test] + fn parse_fn_name_with_generics() { + let result = parse_fn_name( + "std::sync::mpmc::list::Channel::::len".to_string(), + false, + false, + ); + assert_eq!( + result, + StructuredFnName { + trait_impl: None, + module_path: [ + "std".to_string(), + "sync".to_string(), + "mpmc".to_string(), + "list".to_string(), + "Channel".to_string() + ] + .to_vec(), + type_parameters: ["T".to_string()].to_vec(), + item: "len".to_string(), + is_public: false, + typ: "unsafe-containing".to_string() + } + ); + } +} From c281e4a000d3f9ea2b24cb5e1b09fb45d609a54f Mon Sep 17 00:00:00 2001 From: Patrick Lam Date: Wed, 2 Sep 2026 11:32:57 +1200 Subject: [PATCH 15/16] print out non-pub unsafe functions, reject input files that do not end with scan_functions.csv --- tools/unsafe-finder/README.md | 4 ++-- tools/unsafe-finder/src/main.rs | 40 ++++++++++++++++----------------- 2 files changed, 22 insertions(+), 22 deletions(-) diff --git a/tools/unsafe-finder/README.md b/tools/unsafe-finder/README.md index 1a0c919bcea1a..891d37e1cf897 100644 --- a/tools/unsafe-finder/README.md +++ b/tools/unsafe-finder/README.md @@ -1,7 +1,7 @@ # Unsafe finder This tool identifies two types of functions: -1. those that belong to impls and are `pub unsafe`; and, +1. those that belong to impls and are marked `unsafe`; and, 2. those that belong to impls and are *not* `unsafe` but contain `unsafe` code. The clippy `missing_safety_doc` lint nags developers to add @@ -9,7 +9,7 @@ The clippy `missing_safety_doc` lint nags developers to add configuration option that will make clippy also complain about private unsafe functions (default false). For the purpose of verifying the Rust standard library, such functions should have contracts and be -verified against them. +verified against them, whether they are public or not. For category (2), the unsafety is encapsulated in the function; there must be some reason that the unsafe code in the diff --git a/tools/unsafe-finder/src/main.rs b/tools/unsafe-finder/src/main.rs index a1a153c0d596a..bcc8a8de4321b 100644 --- a/tools/unsafe-finder/src/main.rs +++ b/tools/unsafe-finder/src/main.rs @@ -194,23 +194,9 @@ fn handle_file(path: &Path) -> Result<(), Box> { for result in rdr.deserialize() { let fn_stats: FnStats = result?; let is_unsafe = matches!(fn_stats.is_unsafe, Some(true)); - if is_unsafe { - let is_public = matches!(fn_stats.is_public, Some(true)); - if is_public { - let structured_fn_name = parse_fn_name(fn_stats.name, is_public, is_unsafe); - match fns_by_modules.get_mut(&structured_fn_name.module_path) { - Some(fns) => fns.push(structured_fn_name.clone()), - None => { - fns_by_modules.insert( - structured_fn_name.module_path.clone(), - vec![structured_fn_name.clone()], - ); - } - } - } - } else if !is_unsafe && matches!(fn_stats.has_unsafe_ops, Some(true)) { - let is_public = matches!(fn_stats.is_public, Some(true)); - let structured_fn_name = parse_fn_name(fn_stats.name, is_public, is_unsafe); + let is_public = matches!(fn_stats.is_public, Some(true)); + let structured_fn_name = parse_fn_name(fn_stats.name, is_public, is_unsafe); + if is_unsafe || (!is_unsafe && matches!(fn_stats.has_unsafe_ops, Some(true))) { match fns_by_modules.get_mut(&structured_fn_name.module_path) { Some(fns) => fns.push(structured_fn_name.clone()), None => { @@ -224,7 +210,15 @@ fn handle_file(path: &Path) -> Result<(), Box> { } for mp in fns_by_modules.keys().sorted() { - println!("modules {:?}", mp); + println!( + "modules {:?} {}", + mp, + if mp.is_empty() { + "(including trait impls)" + } else { + "" + } + ); if let Some(fns) = fns_by_modules.get(mp) { for structured_fn_name in fns { println!( @@ -255,15 +249,21 @@ fn handle_file(path: &Path) -> Result<(), Box> { fn main() { let mut args = env::args(); - let _ = args.next(); // executable name + let _ = args.next(); // skip executable name if args.len() == 0 { - // should we only handle files named "_scan_functions.csv"? eprintln!("Usage: unsafe-finder [[prefix]_scan_functions.csv]*"); process::exit(1); } for arg in args { + if !arg.ends_with("_scan_functions.csv") { + eprintln!( + "error: filename {} does not end with _scan_functions.csv", + arg + ); + process::exit(1); + } let path = Path::new(&arg); if let Err(err) = handle_file(path) { eprintln!("error processing {}: {}", arg, err); From 34e7d72db751b60b7a87f156c491b01a311e47a9 Mon Sep 17 00:00:00 2001 From: Patrick Lam Date: Wed, 2 Sep 2026 11:46:14 +1200 Subject: [PATCH 16/16] edition 2021 is fine --- tools/unsafe-finder/Cargo.toml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tools/unsafe-finder/Cargo.toml b/tools/unsafe-finder/Cargo.toml index cfe938ed219d8..e1f44fb2a991f 100644 --- a/tools/unsafe-finder/Cargo.toml +++ b/tools/unsafe-finder/Cargo.toml @@ -1,7 +1,7 @@ [package] name = "unsafe-finder" version = "0.1.0" -edition = "2024" +edition = "2021" [dependencies] csv = "1.1"