From 96aa75d955ce7a2dfb2f1fcd9e061d1a9d6590d6 Mon Sep 17 00:00:00 2001 From: Zack Maril Date: Wed, 7 Oct 2026 12:47:53 +0000 Subject: [PATCH 01/25] A patch that checks reuse inside rustc, on in every fuzz and replay build docs/hunt/verify-reuse.patch: under RUSTC_VERIFY_REUSE, reused metadata is encoded again and compared, and at the end of the session every green cached query value is computed again and compared by stable hash, Debug text and encoding, plus which allocations values share. With each of the three fixes reverted it reports that bug; with all three it is quiet. Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_018Mnrg9JXj9X1ht6Qkz2ybh --- docs/hunt/verify-reuse.patch | 573 +++++++++++++++++++++++++++++++++++ docs/scale.md | 4 + docs/shadow-mode.md | 71 ++++- rustc/fuzz.py | 24 ++ rustc/replay.py | 11 + 5 files changed, 677 insertions(+), 6 deletions(-) create mode 100644 docs/hunt/verify-reuse.patch diff --git a/docs/hunt/verify-reuse.patch b/docs/hunt/verify-reuse.patch new file mode 100644 index 0000000..8136775 --- /dev/null +++ b/docs/hunt/verify-reuse.patch @@ -0,0 +1,573 @@ +diff --git a/compiler/rustc_incremental/src/persist/save.rs b/compiler/rustc_incremental/src/persist/save.rs +index 46f47d6c8..69ea0af7a 100644 +--- a/compiler/rustc_incremental/src/persist/save.rs ++++ b/compiler/rustc_incremental/src/persist/save.rs +@@ -41,6 +41,7 @@ pub(crate) fn save_dep_graph(tcx: TyCtxt<'_>) { + + sess.time("assert_dep_graph", || assert_dep_graph(tcx)); + sess.time("check_clean", || clean::check_clean_annotations(tcx)); ++ sess.time("verify_reused_values", || tcx.verify_reused_values()); + + par_join( + move || { +diff --git a/compiler/rustc_metadata/src/rmeta/encoder.rs b/compiler/rustc_metadata/src/rmeta/encoder.rs +index 03493d0e1..1bb732d9a 100644 +--- a/compiler/rustc_metadata/src/rmeta/encoder.rs ++++ b/compiler/rustc_metadata/src/rmeta/encoder.rs +@@ -52,6 +52,9 @@ + pub(super) struct EncodeContext<'a, 'tcx> { + opaque: FileEncoder<'a>, + metadata_hasher: Arc>, ++ /// Set for a shadow encoding (see `verify_reused_metadata`), which keeps the crate hash it ++ /// computes here instead of setting the session's, which the reused metadata already set. ++ shadow_hash: Option>, + tcx: TyCtxt<'tcx>, + feat: &'tcx rustc_feature::Features, + tables: TableBuilders, +@@ -817,7 +820,11 @@ macro_rules! stat { + } else { + tcx.crate_hash(LOCAL_CRATE) + }; +- tcx.untracked().local_crate_hash.set(hash).expect("local_crate_hash set twice"); ++ if let Some(shadow_hash) = &mut self.shadow_hash { ++ *shadow_hash = Some(hash); ++ } else { ++ tcx.untracked().local_crate_hash.set(hash).expect("local_crate_hash set twice"); ++ } + + let unhashed = stat!("final", || { + // Indexed by dependency `CrateNum`, matching the numbering `encode_crate_deps` uses. +@@ -2561,6 +2568,10 @@ pub fn encode_metadata(tcx: TyCtxt<'_>, path: &Path, ref_path: Option<&Path>) { + let hash = blob.expect("file already created").get_crate_hash(); + tcx.untracked().local_crate_hash.set(hash).expect("local_crate_hash set twice"); + ++ if std::env::var_os("RUSTC_VERIFY_REUSE").is_some() { ++ verify_reused_metadata(tcx, path); ++ } ++ + // Generate the metadata stub manually, as that is a small file compared to full metadata. + if let Some(ref_path) = ref_path { + let _prof_timer = tcx.prof.verbose_generic_activity("generate_crate_metadata_stub"); +@@ -2638,6 +2649,15 @@ fn with_encode_metadata_header( + tcx: TyCtxt<'_>, + path: &Path, + f: impl FnOnce(&mut EncodeContext<'_, '_>) -> (usize, usize), ++) { ++ with_encode_metadata_header_shadow(tcx, path, false, f) ++} ++ ++fn with_encode_metadata_header_shadow( ++ tcx: TyCtxt<'_>, ++ path: &Path, ++ shadow: bool, ++ f: impl FnOnce(&mut EncodeContext<'_, '_>) -> (usize, usize), + ) { + // By default the crate hash (SVH) is computed from the bytes of the encoded metadata, + // Under `-Z metadata-crate-hash=no` the SVH comes from the legacy `crate_hash` query instead and +@@ -2694,6 +2714,7 @@ fn with_encode_metadata_header( + let mut ecx = EncodeContext { + opaque: encoder, + metadata_hasher: Arc::clone(&metadata_hasher), ++ shadow_hash: shadow.then_some(None), + tcx, + feat: tcx.features(), + tables: Default::default(), +@@ -2733,12 +2754,15 @@ fn with_encode_metadata_header( + tcx.dcx().emit_fatal(FailWriteFile { path: ecx.opaque.path(), err }); + } + +- let hash = tcx +- .untracked() +- .local_crate_hash +- .get() +- .copied() +- .expect("local_crate_hash set during encoding"); ++ let hash = match ecx.shadow_hash { ++ Some(shadow_hash) => shadow_hash.expect("crate hash computed during the shadow encoding"), ++ None => tcx ++ .untracked() ++ .local_crate_hash ++ .get() ++ .copied() ++ .expect("local_crate_hash set during encoding"), ++ }; + if let Err(err) = encode_crate_hash(file, hash) { + tcx.dcx().emit_fatal(FailWriteFile { path: ecx.opaque.path(), err }); + } +@@ -2748,6 +2772,50 @@ fn with_encode_metadata_header( + } + } + ++/// Shadow verification, for testing incremental compilation: metadata was just reused from the ++/// incremental cache because its dep-node is green. Encode it again from this session's ++/// queries, outside dependency tracking, and check that the bytes are the same. A difference ++/// means the reuse was wrong: something the metadata depends on is not tracked. ++/// ++/// On a difference this prints a line starting `rustc-verify-reuse:` and keeps the fresh ++/// encoding beside the reused file, as `.fresh`; otherwise it removes it. ++fn verify_reused_metadata(tcx: TyCtxt<'_>, path: &Path) { ++ let fresh = path.with_extension("rmeta.fresh"); ++ // `encode_metadata` already runs with dependency tracking ignored. ++ with_encode_metadata_header_shadow(tcx, &fresh, true, |ecx| { ++ let (root, unhashed) = ecx.encode_crate_root(); ++ ecx.opaque.flush(); ++ (root.position.get(), unhashed.position.get()) ++ }); ++ let reused = std::fs::read(path).unwrap_or_default(); ++ let encoded = std::fs::read(&fresh).unwrap_or_default(); ++ if std::env::var_os("RUSTC_VERIFY_REUSE").is_some_and(|v| v == "verbose") { ++ eprintln!("rustc-verify-reuse-checked: metadata of `{}`", tcx.crate_name(LOCAL_CRATE)); ++ } ++ if reused == encoded { ++ let _ = std::fs::remove_file(&fresh); ++ } else { ++ // Keep it next to the output, outside the temporary directory. ++ let fresh = match tcx.output_filenames(()).path(rustc_session::config::OutputType::Metadata) { ++ rustc_session::config::OutFileName::Real(out) ++ if std::fs::rename(&fresh, out.with_extension("rmeta.fresh")).is_ok() => ++ { ++ out.with_extension("rmeta.fresh") ++ } ++ _ => fresh, ++ }; ++ let at = reused.iter().zip(&encoded).position(|(a, b)| a != b).unwrap_or(reused.len().min(encoded.len())); ++ eprintln!( ++ "rustc-verify-reuse: metadata of `{}` reused from the incremental cache differs from a fresh encoding ({} and {} bytes, first difference at byte {}); kept as {}", ++ tcx.crate_name(LOCAL_CRATE), ++ reused.len(), ++ encoded.len(), ++ at, ++ fresh.display(), ++ ); ++ } ++} ++ + fn encode_root_position(mut file: &File, pos: usize) -> Result<(), std::io::Error> { + file.seek(SeekFrom::Start(ROOT_POS_OFFSET as u64))?; + file.write_all(&pos.to_le_bytes())?; +diff --git a/compiler/rustc_middle/src/hooks.rs b/compiler/rustc_middle/src/hooks.rs +index 7bccc34db..71f2341b1 100644 +--- a/compiler/rustc_middle/src/hooks.rs ++++ b/compiler/rustc_middle/src/hooks.rs +@@ -106,6 +106,10 @@ fn clone(&self) -> Self { *self } + + hook verify_query_key_hashes() -> (); + ++ /// Under `RUSTC_VERIFY_REUSE`, computes reused query values again and reports any that ++ /// differ. For testing incremental compilation. ++ hook verify_reused_values() -> (); ++ + /// Ensure the given scalar is valid for the given type. + /// This checks non-recursive runtime validity. + hook validate_scalar_in_layout(scalar: crate::ty::ScalarInt, ty: Ty<'tcx>) -> bool; +diff --git a/compiler/rustc_middle/src/query/on_disk_cache.rs b/compiler/rustc_middle/src/query/on_disk_cache.rs +index 956c59012..957ddcb20 100644 +--- a/compiler/rustc_middle/src/query/on_disk_cache.rs ++++ b/compiler/rustc_middle/src/query/on_disk_cache.rs +@@ -756,6 +756,51 @@ pub struct CacheEncoder<'tcx> { + side_effects_index: Vec<(SerializedDepNodeIndex, AbsoluteBytePos)>, + } + ++/// For testing incremental compilation (`RUSTC_VERIFY_REUSE`): `value` as the cache would ++/// encode it, alone, in a fresh encoder, so that two values can be compared by the bytes a later ++/// session would read, and the allocations it refers to in the order it first refers to them. ++/// `path` is a scratch file. ++pub fn encode_alone<'tcx, V: Encodable>>( ++ tcx: TyCtxt<'tcx>, ++ path: &std::path::Path, ++ value: &V, ++) -> (Vec, Vec) { ++ let file_to_file_index = tcx ++ .sess ++ .source_map() ++ .files() ++ .iter() ++ .enumerate() ++ .map(|(index, file)| (&raw const **file, SourceFileIndex(index as u32))) ++ .collect(); ++ let Ok(file) = FileEncoder::new(path) else { return Default::default() }; ++ let mut encoder = CacheEncoder { ++ tcx, ++ encoder: file, ++ type_shorthands: Default::default(), ++ predicate_shorthands: Default::default(), ++ interpret_allocs: Default::default(), ++ caching_source_map_view: CachingSourceMapView::new(tcx.sess.source_map()), ++ file_to_file_index, ++ hygiene_context: Default::default(), ++ symbol_index_table: Default::default(), ++ source_span_cache: Default::default(), ++ query_values_index: Default::default(), ++ side_effects_index: Default::default(), ++ }; ++ value.encode(&mut encoder); ++ // The allocations it refers to, as `serialize` encodes them, so that their contents are ++ // compared too. ++ let mut n = 0; ++ while n < encoder.interpret_allocs.len() { ++ let id = encoder.interpret_allocs[n]; ++ interpret::specialized_encode_alloc_id(&mut encoder, tcx, id); ++ n += 1; ++ } ++ let _ = encoder.encoder.finish(); ++ (std::fs::read(path).unwrap_or_default(), encoder.interpret_allocs.into_iter().collect()) ++} ++ + impl<'tcx> fmt::Debug for CacheEncoder<'tcx> { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + // Add more details here if/when necessary. +diff --git a/compiler/rustc_query_impl/src/incremental.rs b/compiler/rustc_query_impl/src/incremental.rs +index a030fe71e..f868ab5b2 100644 +--- a/compiler/rustc_query_impl/src/incremental.rs ++++ b/compiler/rustc_query_impl/src/incremental.rs +@@ -1,17 +1,20 @@ + use rustc_data_structures::fingerprint::{Fingerprint, PackedFingerprint}; ++use rustc_data_structures::fx::FxHashMap; + use rustc_data_structures::unord::UnordMap; + #[expect(unused_imports, reason = "used by doc comments")] + use rustc_middle::dep_graph::DepKindVTable; + use rustc_middle::dep_graph::{ + DepGraphData, DepNode, DepNodeIndex, DepNodeKey, SerializedDepNodeIndex, + }; ++use rustc_middle::mir::interpret::AllocId; + use rustc_middle::query::erase::{Erasable, Erased}; +-use rustc_middle::query::on_disk_cache::{CacheDecoder, CacheEncoder}; +-use rustc_middle::query::{QueryCache, QueryState, QueryVTable, erase}; ++use rustc_middle::query::on_disk_cache::{self, CacheDecoder, CacheEncoder}; ++use rustc_middle::query::{QueryCache, QueryKey, QueryState, QueryVTable, erase}; + use rustc_middle::ty::TyCtxt; + use rustc_middle::verify_ich::incremental_verify_ich; + use rustc_serialize::{Decodable, Encodable}; + use rustc_span::bug; ++use rustc_span::def_id::LocalDefId; + + use crate::query_vtables::for_each_query_vtable; + +@@ -25,6 +28,317 @@ pub(crate) fn encode_query_values<'tcx>(tcx: TyCtxt<'tcx>, encoder: &mut CacheEn + }); + } + ++/// Shadow verification, for testing incremental compilation, under `RUSTC_VERIFY_REUSE`: at ++/// the end of the session, before the dependency graph and the cache are saved, compute every ++/// green value of a cached query again with its provider, outside dependency tracking, and ++/// compare: by stable hash; by `Debug` text, which also sees fields the stable hash ignores; ++/// and by the bytes the cache would encode, which also see the order of hash maps. A green ++/// value was reused from the previous session; a difference means it was stale (its ++/// computation read something that is not tracked), or that it did not survive the round trip ++/// through the cache. ++/// ++/// It runs here because no query is executing, so recomputing cannot cycle back into the ++/// query being checked, and the previous session's cache is still readable. A difference ++/// prints a line starting `rustc-verify-reuse:`. `RUSTC_VERIFY_REUSE_SKIP` lists more queries, comma-separated, not to recompute. ++pub(crate) fn verify_reused_values<'tcx>(tcx: TyCtxt<'tcx>) { ++ if std::env::var_os("RUSTC_VERIFY_REUSE").is_none() { ++ return; ++ } ++ // Load every green value that was not used this session, so it is checked too. ++ tcx.dep_graph.exec_cache_promotions(tcx); ++ // Printing values must not use trimmed paths, which assume a diagnostic is emitted. ++ let mut allocs = Allocs::default(); ++ rustc_middle::ty::print::with_no_trimmed_paths!(for_each_query_vtable!( ++ CACHE_ON_DISK, ++ tcx, ++ |query| verify_reused_values_inner(tcx, query, &mut allocs) ++ )); ++} ++ ++/// `Debug` text in a form two equal values print the same: the contents of `OnceLock`s are ++/// removed, since they are caches filled on demand and a value that has been used shows more ++/// than a fresh one; the elements of `UnordMap`s and `UnordSet`s are sorted, since their ++/// order is unspecified; and the numbers of `AllocId`s (`alloc12`) are removed, since a ++/// decoded allocation and a computed one get different ones. (The encoding comparison sees ++/// the allocations' contents.) ++fn normalized(text: &str) -> String { ++ normalized_inner(&without_alloc_ids(text)) ++} ++ ++fn without_alloc_ids(text: &str) -> String { ++ let mut out = String::with_capacity(text.len()); ++ let mut rest = text; ++ while let Some(at) = rest.find("alloc") { ++ let after = &rest[at + "alloc".len()..]; ++ let digits = after.len() - after.trim_start_matches(|c: char| c.is_ascii_digit()).len(); ++ let word = rest[..at].chars().next_back().is_some_and(|c| c.is_alphanumeric() || c == '_'); ++ out.push_str(&rest[..at + "alloc".len()]); ++ if digits > 0 && !word { ++ out.push('_'); ++ } else { ++ out.push_str(&after[..digits]); ++ } ++ rest = &after[digits..]; ++ } ++ out.push_str(rest); ++ out ++} ++ ++fn normalized_inner(text: &str) -> String { ++ // The index of the bracket closing the one just before `text[from..]`. ++ fn closing(text: &str, from: usize) -> usize { ++ let mut depth = 1; ++ for (i, c) in text[from..].char_indices() { ++ match c { ++ '(' | '[' | '{' => depth += 1, ++ ')' | ']' | '}' => depth -= 1, ++ _ => {} ++ } ++ if depth == 0 { ++ return from + i; ++ } ++ } ++ text.len() ++ } ++ // `text` split at top-level `, `. ++ fn elements(text: &str) -> Vec<&str> { ++ let (mut out, mut depth, mut start) = (Vec::new(), 0, 0); ++ let bytes = text.as_bytes(); ++ for (i, c) in text.char_indices() { ++ match c { ++ '(' | '[' | '{' => depth += 1, ++ ')' | ']' | '}' => depth -= 1, ++ ',' if depth == 0 && bytes.get(i + 1) == Some(&b' ') => { ++ out.push(&text[start..i]); ++ start = i + 2; ++ } ++ _ => {} ++ } ++ } ++ if start < text.len() { ++ out.push(&text[start..]); ++ } ++ out ++ } ++ let mut out = String::with_capacity(text.len()); ++ let mut rest = text; ++ loop { ++ let cache = rest.find("OnceLock(").map(|at| (at, "OnceLock(", false)); ++ let unord = ["UnordMap { inner: {", "UnordSet { inner: {"] ++ .into_iter() ++ .filter_map(|m| rest.find(m).map(|at| (at, m, true))) ++ .min(); ++ let Some((at, marker, sort)) = [cache, unord].into_iter().flatten().min() else { ++ break; ++ }; ++ let open = at + marker.len(); ++ let end = closing(rest, open); ++ out.push_str(&rest[..open]); ++ if sort { ++ let mut items: Vec = ++ elements(&rest[open..end]).into_iter().map(normalized_inner).collect(); ++ items.sort(); ++ out.push_str(&items.join(", ")); ++ } else { ++ out.push('_'); ++ } ++ rest = &rest[end..]; ++ } ++ out.push_str(rest); ++ out ++} ++ ++/// Which allocations values share. For every value of the queries in `SHARING`, each ++/// allocation a fresh computation refers to is mapped to the one the value in use refers to in ++/// the same place (a value computed this session maps its allocations to themselves). Two ++/// different allocations for one fresh one mean the reused value does not share an allocation ++/// that a clean session would share: the round trip through the cache lost a deduplication. ++#[derive(Default)] ++struct Allocs { ++ /// Fresh allocation → the allocation in use, and where it was first seen. ++ seen: FxHashMap, ++} ++ ++impl Allocs { ++ fn check(&mut self, fresh: AllocId, used: AllocId, here: impl Fn() -> String) { ++ match self.seen.get(&fresh) { ++ None => { ++ self.seen.insert(fresh, (used, here())); ++ } ++ Some((other_used, other)) if *other_used != used => eprintln!( ++ "rustc-verify-reuse: allocation shared differently: {} uses {used:?} where a fresh computation uses {fresh:?}, but {other} uses {other_used:?} for it", ++ here(), ++ ), ++ Some(_) => {} ++ } ++ } ++} ++ ++/// Queries whose values refer to allocations that may be shared with other values. ++const SHARING: &[&str] = &[ ++ "eval_static_initializer", ++ "eval_to_allocation_raw", ++ "eval_to_const_value_raw", ++ "mir_for_ctfe", ++ "optimized_mir", ++ "promoted_mir", ++ "trivial_const", ++]; ++ ++/// Queries whose providers read the MIR or THIR of the definition they are given, which is ++/// stolen once it has been used: they are recomputed only if none of it has been stolen. ++const READING_BODIES: &[&str] = &[ ++ "check_liveness", ++ "check_match", ++ "check_tail_calls", ++ "check_unsafety", ++ "has_ffi_unwind_calls", ++ "mir_const_qualif", ++ "mir_coroutine_witnesses", ++ "mir_for_ctfe", ++ "mir_inliner_callees", ++ "optimized_mir", ++ "promoted_mir", ++ "thir_abstract_const", ++ "trivial_const", ++]; ++ ++/// Queries never recomputed: `mir_borrowck` reads the MIR of nested bodies too, and ++/// `coroutine_by_move_body_def_id` makes a definition. ++const NOT_RECOMPUTED: &[&str] = &["coroutine_by_move_body_def_id", "mir_borrowck"]; ++ ++/// Whether any MIR or THIR of `def` computed this session has been stolen. ++fn body_stolen<'tcx>(tcx: TyCtxt<'tcx>, def: LocalDefId) -> bool { ++ use rustc_middle::queries::{ ++ mir_built, mir_drops_elaborated_and_const_checked as mir_elaborated, mir_promoted, ++ thir_body, ++ }; ++ let vtables = &tcx.query_system.query_vtables; ++ let built = vtables.mir_built.cache.lookup(&def).is_some_and(|(body, _)| { ++ erase::restore_val::>(body).is_stolen() ++ }); ++ let elaborated = vtables.mir_drops_elaborated_and_const_checked.cache.lookup(&def).is_some_and( ++ |(body, _)| erase::restore_val::>(body).is_stolen(), ++ ); ++ let promoted = vtables.mir_promoted.cache.lookup(&def).is_some_and(|(value, _)| { ++ let (body, promoted) = erase::restore_val::>(value); ++ body.is_stolen() || promoted.is_stolen() ++ }); ++ let thir = vtables.thir_body.cache.lookup(&def).is_some_and(|(value, _)| { ++ erase::restore_val::>(value).is_ok_and(|(thir, _)| thir.is_stolen()) ++ }); ++ built || elaborated || promoted || thir ++} ++ ++fn verify_reused_values_inner<'tcx, C, V>( ++ tcx: TyCtxt<'tcx>, ++ query: &'tcx QueryVTable<'tcx, C>, ++ allocs: &mut Allocs, ++) where ++ C: QueryCache>, ++ V: Erasable + Encodable>, ++{ ++ if NOT_RECOMPUTED.contains(&query.name) { ++ return; ++ } ++ let reads_body = READING_BODIES.contains(&query.name); ++ if let Some(skip) = std::env::var_os("RUSTC_VERIFY_REUSE_SKIP") ++ && skip.to_string_lossy().split(',').any(|name| name == query.name) ++ { ++ return; ++ } ++ let mut entries = Vec::new(); ++ query.cache.for_each(&mut |key, value, _| entries.push((*key, *value))); ++ let hash = |value: &C::Value| { ++ query.hash_value_fn.map_or(Fingerprint::ZERO, |f| { ++ tcx.with_stable_hashing_context(|mut hcx| f(&mut hcx, value)) ++ }) ++ }; ++ let scratch = std::env::temp_dir().join(format!("rustc-verify-reuse-{}", std::process::id())); ++ let mut checked = 0; ++ for (key, value) in entries { ++ if !query.will_cache_on_disk_for_key(key) { ++ continue; ++ } ++ let sharing = SHARING.contains(&query.name); ++ let dep_node = DepNode::construct(tcx, query.dep_kind, &key); ++ if !tcx.dep_graph.is_green(&dep_node) { ++ if sharing { ++ let (_, ids) = ++ on_disk_cache::encode_alone(tcx, &scratch, &erase::restore_val::(value)); ++ let here = || format!("query `{}` for {:?}, computed this session", query.name, key); ++ for id in ids { ++ allocs.check(id, id, here); ++ } ++ } ++ continue; ++ } ++ // A feedable query's value for a definition the compiler made up is set by whatever ++ // made it, and the provider may not accept the key: the associated type for an ++ // `impl Trait` in a trait (a synthetic HIR node), an elided lifetime that lowering ++ // adds as a parameter, or the type of a const argument, set when it is lowered. ++ if query.feedable ++ && key.key_as_def_id().and_then(|id| id.as_local()).is_some_and(|id| { ++ matches!( ++ tcx.def_kind(id), ++ rustc_hir::def::DefKind::LifetimeParam | rustc_hir::def::DefKind::AnonConst ++ ) || matches!(tcx.hir_node_by_def_id(id), rustc_hir::Node::Synthetic) ++ }) ++ { ++ continue; ++ } ++ if reads_body ++ && key ++ .key_as_def_id() ++ .and_then(|id| id.as_local()) ++ .is_none_or(|id| body_stolen(tcx, id) || body_stolen(tcx, tcx.typeck_root_def_id_local(id))) ++ { ++ continue; ++ } ++ checked += 1; ++ let fresh = tcx.dep_graph.with_ignore(|| (query.invoke_provider_fn)(tcx, key)); ++ let text = (query.format_value)(&value); ++ let loaded = normalized(&text); ++ let computed = normalized(&(query.format_value)(&fresh)); ++ let same_hash = hash(&value) == hash(&fresh); ++ let same_text = loaded == computed; ++ // `UnordMap`s and `UnordSet`s encode in their iteration order, which differs between a ++ // decoded value and a computed one and which nothing may observe, so the encoding of a ++ // value containing one is not compared. ++ let (loaded_bytes, loaded_ids) = ++ on_disk_cache::encode_alone(tcx, &scratch, &erase::restore_val::(value)); ++ let (fresh_bytes, fresh_ids) = ++ on_disk_cache::encode_alone(tcx, &scratch, &erase::restore_val::(fresh)); ++ let same_bytes = ++ text.contains("UnordMap {") || text.contains("UnordSet {") || loaded_bytes == fresh_bytes; ++ if sharing && loaded_ids.len() == fresh_ids.len() { ++ let here = || format!("query `{}` for {:?}, green", query.name, key); ++ for (fresh_id, loaded_id) in fresh_ids.into_iter().zip(loaded_ids) { ++ allocs.check(fresh_id, loaded_id, here); ++ } ++ } ++ if !(same_hash && same_text && same_bytes) { ++ let short = |s: &str| s.chars().take(2000).collect::(); ++ let what = [(same_hash, "stable hash"), (same_text, "Debug text"), (same_bytes, "encoding")] ++ .into_iter() ++ .filter_map(|(same, what)| (!same).then_some(what)) ++ .collect::>() ++ .join(", "); ++ eprintln!( ++ "rustc-verify-reuse: query `{}` for {:?}, green, differs from a fresh computation ({what})\n reused: {}\n fresh: {}", ++ query.name, ++ key, ++ short(&loaded), ++ short(&computed), ++ ); ++ } ++ } ++ let _ = std::fs::remove_file(&scratch); ++ if checked > 0 && std::env::var_os("RUSTC_VERIFY_REUSE").is_some_and(|v| v == "verbose") { ++ eprintln!("rustc-verify-reuse-checked: {} {checked}", query.name); ++ } ++} ++ + fn encode_query_values_inner<'tcx, C, V>( + tcx: TyCtxt<'tcx>, + query: &'tcx QueryVTable<'tcx, C>, +diff --git a/compiler/rustc_query_impl/src/lib.rs b/compiler/rustc_query_impl/src/lib.rs +index eed772cae..5eecf938d 100644 +--- a/compiler/rustc_query_impl/src/lib.rs ++++ b/compiler/rustc_query_impl/src/lib.rs +@@ -52,4 +52,5 @@ pub fn provide(providers: &mut rustc_middle::util::Providers) { + self_profile::alloc_self_profile_query_strings; + providers.hooks.verify_query_key_hashes = incremental::verify_query_key_hashes; + providers.hooks.encode_query_values = incremental::encode_query_values; ++ providers.hooks.verify_reused_values = incremental::verify_reused_values; + } diff --git a/docs/scale.md b/docs/scale.md index 4b958e9..77bd463 100644 --- a/docs/scale.md +++ b/docs/scale.md @@ -127,6 +127,10 @@ compared, and ICEs, hangs and one-sided failures are reported. Every 40 kept edi starts again from the pristine fixture. A finding keeps every edit since the last reset, and `rustc/fuzz-replay.py` replays it exactly. +Since [`shadow-mode.md`](shadow-mode.md), the fuzzer and the replay also run every build +with the compiler's own check of what it reused (`RUSTC_VERIFY_REUSE`, on a compiler with +[`hunt/verify-reuse.patch`](hunt/verify-reuse.patch)), and report what it prints. + Throughput on this 16-core machine: about 2 edits a second with six workers, about 170,000 a day, while other work shared the machine. diff --git a/docs/shadow-mode.md b/docs/shadow-mode.md index 74544bf..412f261 100644 --- a/docs/shadow-mode.md +++ b/docs/shadow-mode.md @@ -43,11 +43,70 @@ Recomputing everything doubles a build's cost, so the flag would usually sample, `-Zincremental-verify-ich` does: a deterministic subset per session, rotating so that every reused item is checked over many sessions, with an option to check everything. -## Where it would go first +## The patch -Metadata is the cheapest to start with and has the most recent bug: the reuse decision is -one function (`encode_metadata` in `rustc_metadata/src/rmeta/encoder.rs`), and encoding -again into a temporary file and comparing bytes is a small change. A failure would name -the first differing byte, which `-Zmeta-stats`'s sections place in a table. +[`hunt/verify-reuse.patch`](hunt/verify-reuse.patch), against the pinned rustc, does this for +metadata and for query results. It is on when `RUSTC_VERIFY_REUSE` is set (`verbose` also +counts what was checked), and every fuzzer and replay build now runs with it: a line +starting `rustc-verify-reuse:` is a finding of kind `verify-reuse` (`reuse` in the replay). -Not started. This page is the record of what was looked at. +**Metadata.** When `encode_metadata` reuses the saved `.rmeta`, it also encodes the metadata +again, into a file next to the output, and compares the bytes. A difference prints the +first differing byte and keeps the fresh file as `.rmeta.fresh`. + +**Query results.** At the end of the session, just before the dependency graph and the cache +are saved, every value of a query cached on disk whose node is green is computed again by +its provider, outside dependency tracking, and compared with the value in use three ways: + +- the stable hash; +- the `Debug` text, which sees fields the stable hash ignores, with on-demand caches + (`OnceLock`) blanked, `UnordMap`/`UnordSet` elements sorted and `AllocId` numbers removed; +- the bytes the cache would encode for the value alone, allocations included, which see the + order of hash maps (not compared for values containing an `UnordMap` or `UnordSet`, which + encode in an order nothing may observe). + +And across values: for queries whose values refer to allocations (MIR, const evaluation), +each allocation a fresh computation refers to is mapped to the one the value in use refers +to. One fresh allocation mapped to two different ones means the values in use do not share an +allocation that a clean session would share. + +The first version recomputed a value as it was loaded. That runs the provider while the query +that asked for the value is still executing, and it cycled (`E0391`, "cycle detected when +finding item bounds"). At the end of the session nothing is executing, and the previous +session's cache can still be read. + +Not recomputed: + +- queries whose provider reads MIR or THIR that has already been stolen + (`optimized_mir`, `mir_for_ctfe` and others, for a definition whose bodies were built this + session; the same definitions are checked when their bodies were not built); +- `mir_borrowck`, which reads the MIR of nested bodies too, and + `coroutine_by_move_body_def_id`, which makes a definition; +- values of feedable queries for definitions the compiler made up (the associated type of + an `impl Trait` in a trait, an elided lifetime added by lowering, the type of a const + argument), which are set rather than computed; +- anything named in `RUSTC_VERIFY_REUSE_SKIP` (comma-separated query names). + +Object files and replayed diagnostics are not checked yet. + +## Does it find the known bugs? + +Each fix reverted in turn on the patched compiler, with the reproduction from +[`hunt/`](hunt) and an edit to `fixtures/sink`: + +| bug | fix reverted | the check prints | +|---|---|---| +| `param_def_id_to_index` order ([report](hunt/issue-generics-order.md)) | `generics-index-map.patch` | ``query `generics_of` for DefId(0:11 ~ lib[ab28]::{impl#0}), green, differs from a fresh computation (encoding)`` | +| literal allocation deduplication ([report](hunt/issue-literal-dedup.md)) | `alloc-dedup-on-decode.patch` | ``allocation shared differently: query `optimized_mir` for DefId(0:4 ~ lib[ab28]::b), computed this session uses alloc1 where a fresh computation uses alloc1, but query `optimized_mir` for DefId(0:3 ~ lib[ab28]::a), green uses alloc2 for it`` | +| stale metadata reuse ([report](hunt/issue-stale-metadata-reuse.md)) | `metadata-source-files.patch` | ``metadata of `sink_core` reused from the incremental cache differs from a fresh encoding (237382 and 237383 bytes, first difference at byte 8)`` | + +With all three fixes, the same builds print nothing. Neither of the first two is visible to +the stable hash, so `-Zincremental-verify-ich` cannot see them. + +**Cost.** An incremental rebuild of `fixtures/sink` after an edit recomputes about 11,800 green +values and takes 2.65 s instead of 2.33 s. + +**Noise removed on the way.** Before the `Debug` text and encoding comparisons were +normalized, they reported values that were equal: lazily filled caches in MIR bodies, the +iteration order of `UnordMap`s in `typeck_root` and `specialization_graph_of`, and the +numbers of allocations in const-evaluation results. diff --git a/rustc/fuzz.py b/rustc/fuzz.py index a2ec444..ec5d4f6 100755 --- a/rustc/fuzz.py +++ b/rustc/fuzz.py @@ -14,6 +14,8 @@ exe the binary's bytes diag the diagnostics each crate printed run the binaries' output and exit status + reuse the compiler's own check of what it reused (RUSTC_VERIFY_REUSE, + docs/hunt/verify-reuse.patch) found nothing stale ICE neither build crashed the compiler split both builds succeed or both fail @@ -54,6 +56,8 @@ p.add_argument("--rustflags", default="-Zincremental-verify-ich") p.add_argument("--seed", type=int, default=0) p.add_argument("--timeout", type=int, default=180, help="seconds before a build counts as hung") +p.add_argument("--no-verify-reuse", action="store_true", + help="do not set RUSTC_VERIFY_REUSE (needs a compiler with docs/hunt/verify-reuse.patch)") args = p.parse_args() WORK = Path(args.work).resolve() @@ -262,6 +266,8 @@ def env(): e = dict(os.environ) e.update(RUSTC=args.rustc, RUSTC_WRAPPER="", CARGO_INCREMENTAL="1", RUSTFLAGS=args.rustflags, CARGO_TERM_COLOR="never") + if not args.no_verify_reuse: + e["RUSTC_VERIFY_REUSE"] = "1" return e @@ -312,11 +318,26 @@ def build(src, target): "ok": r.returncode == 0, "ice": "internal compiler error" in log or "the compiler unexpectedly panicked" in log, "hang": hang, + "reuse": reuse_checks(log), "secs": time.time() - t, "log": log, "rmetas": rmetas, "exe": exe, "art": artifacts.collect(r.stdout, target), } +def reuse_checks(log): + """What the compiler's own check of reused results (docs/hunt/verify-reuse.patch) found + stale: `query `, `metadata` or `allocation sharing`, once each.""" + found = set() + for line in log.splitlines(): + if line.startswith("rustc-verify-reuse: query `"): + found.add("query " + line.split("`")[1]) + elif line.startswith("rustc-verify-reuse: metadata"): + found.add("metadata") + elif line.startswith("rustc-verify-reuse: allocation shared differently"): + found.add("allocation sharing") + return sorted(found) + + def run_exe(exe): if not exe: return None @@ -410,6 +431,9 @@ def report(kind, detail, inc, clean, extra=None): report("ICE", [], inc, None) if inc["hang"]: report("hang", [], inc, None) + if inc["reuse"]: + lines = [l[:3000] for l in inc["log"].splitlines() if l.startswith("rustc-verify-reuse")] + report("verify-reuse", inc["reuse"], inc, None, lines[:20]) if not inc["ok"]: stats["failed"] += 1 path.write_text(old) diff --git a/rustc/replay.py b/rustc/replay.py index 2619f2a..be5cebe 100755 --- a/rustc/replay.py +++ b/rustc/replay.py @@ -8,6 +8,8 @@ P6 every .rmeta Cargo reports for a workspace member is identical rlib every rlib's members are identical, object code included diag both builds printed the same diagnostics + reuse the compiler's own check of what it reused (RUSTC_VERIFY_REUSE, + docs/hunt/verify-reuse.patch) found nothing stale ICE neither build crashed the compiler split both builds succeed or both fail @@ -46,6 +48,8 @@ p.add_argument("--keep", type=int, default=3, help="differences kept per crate") p.add_argument("--from", dest="start", type=int, default=0, help="first commit index to replay") p.add_argument("--to", dest="end", type=int, default=None, help="last commit index to replay") +p.add_argument("--no-verify-reuse", action="store_true", + help="do not set RUSTC_VERIFY_REUSE (needs a compiler with docs/hunt/verify-reuse.patch)") args = p.parse_args() work = Path(args.work).resolve() @@ -65,6 +69,8 @@ RUSTFLAGS="--cap-lints=warn", CARGO_TERM_COLOR="never", ) +if not args.no_verify_reuse: + env["RUSTC_VERIFY_REUSE"] = "1" cargo = ["cargo", f"+{args.toolchain}"] @@ -106,6 +112,7 @@ def build(target_dir): "ice": "internal compiler error" in log or "the compiler unexpectedly panicked" in log, "secs": round(time.time() - t, 1), "log": log[-6000:], + "reuse": [line[:3000] for line in log.splitlines() if line.startswith("rustc-verify-reuse:")], "rmetas": rmetas, "fresh": fresh, "art": artifacts.collect(r.stdout, target_dir), @@ -160,6 +167,10 @@ def keep_dir(i, commit): problems.append("ICE") if inc["ok"] != clean["ok"]: problems.append("split") + if inc["reuse"]: + # The compiler's own check found something it reused stale. + problems.append("reuse") + (keep_dir(i, commit) / "reuse.txt").write_text("\n".join(inc["reuse"])) if clean["fresh"]: # A member the clean build did not compile again would be compared with itself. problems.append("stale") From 09f3de0e5035dac3aae251ddc807d96d40d5bea2 Mon Sep 17 00:00:00 2001 From: Zack Maril Date: Wed, 7 Oct 2026 13:13:03 +0000 Subject: [PATCH 02/25] Finding 6: reused object code keeps an edited file's old checksum and embedded source Found by a short fuzz run at -Copt-level=2: binaries differed from clean builds only in ThinLTO's .llvm. suffixes, because a reused codegen unit's DIFile has the previous MD5 of the edited file. With -Zembed-source the object embeds the previous file. Report drafted, reproduction added to repro.sh. The fuzzer now names allocation-sharing findings by their queries. Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_018Mnrg9JXj9X1ht6Qkz2ybh --- docs/hunt.md | 7 +- docs/hunt/issue-stale-debuginfo-source.md | 126 ++++++++++++++++++++++ docs/hunt/repro.sh | 14 ++- docs/shadow-mode.md | 4 +- rustc/fuzz.py | 8 +- 5 files changed, 153 insertions(+), 6 deletions(-) create mode 100644 docs/hunt/issue-stale-debuginfo-source.md diff --git a/docs/hunt.md b/docs/hunt.md index d0d01f5..8c3dbf2 100644 --- a/docs/hunt.md +++ b/docs/hunt.md @@ -29,6 +29,7 @@ with `-Zthreads=8`. `rustc/check.sh wide` runs the ordinary checks. | 3 | with `-Zthreads=8`, two traits with `-> impl Trait` methods give different metadata from run to run | known: [#162202](https://github.com/rust-lang/rust/issues/162202) | | 4 | incremental rebuilds republish the previous session's metadata when an edit moves no span, so its source map describes old files | **looks new**; found later by the fuzzer and the history replay ([`scale.md`](scale.md)); root cause found, regression from #114669 (1.90); fix and regression test written | | 5 | `-Zemit-stack-sizes`, `-Zcodegen-source-order` and `-Zbuild-sdylib-interface` are untracked but change output that incremental compilation reuses | **looks new**; found by a query written from a closed bug and an option audit ([`ur-queries.md`](ur-queries.md)); report drafted | +| 6 | reused object code keeps the previous checksum of an edited source file in its debuginfo, and with `-Zembed-source` the previous file; with optimizations, ThinLTO symbol names then differ from a clean build | **looks new**; found by the fuzzer at `-Copt-level=2`; root cause found, since 1.44 (#69718); report drafted, no fix | Findings 1 and 2 are single-threaded: an ordinary `cargo build`, an edit, another `cargo build`, and the metadata differs from a clean build of the edited source. Both come @@ -37,12 +38,14 @@ incremental cache unchanged. All three reproduce with the official `nightly-2026 without mirth: `docs/hunt/repro.sh` runs them. None was searched for: P5 and P6 reported them on the first run of the new fixture. -Draft bug reports for 1, 2, 4 and 5, written to be filed upstream, are +Draft bug reports for 1, 2, 4, 5 and 6, written to be filed upstream, are [`hunt/issue-generics-order.md`](hunt/issue-generics-order.md), [`hunt/issue-literal-dedup.md`](hunt/issue-literal-dedup.md) and [`hunt/issue-stale-metadata-reuse.md`](hunt/issue-stale-metadata-reuse.md) and [`hunt/issue-untracked-options.md`](hunt/issue-untracked-options.md), the last a comment for -rust-lang/rust#84232. Each has a candidate fix +rust-lang/rust#84232; for 6 it is +[`hunt/issue-stale-debuginfo-source.md`](hunt/issue-stale-debuginfo-source.md), which has no +fix, since every fix costs codegen reuse and the choice is the maintainers'. Each has a candidate fix (`hunt/*.patch`) and a regression test in the style of rustc's `tests/run-make` (`hunt/tests/`), which fails on the pinned compiler and passes with the fix. diff --git a/docs/hunt/issue-stale-debuginfo-source.md b/docs/hunt/issue-stale-debuginfo-source.md new file mode 100644 index 0000000..484a525 --- /dev/null +++ b/docs/hunt/issue-stale-debuginfo-source.md @@ -0,0 +1,126 @@ +# Incremental rebuilds reuse object code whose debuginfo has the previous checksum (and embedded source) of an edited file + + + +A codegen unit's debuginfo names each source file with a checksum of its contents +(`DIFile`, added in #69718 so that "a debugger can verify that the source code matches the +executable"), and with `-Zembed-source` the contents themselves. Both are read straight from +the session's source map (`file_metadata` in `rustc_codegen_llvm/src/debuginfo/metadata.rs`), +which nothing tracks. When an edit leaves a codegen unit green (a comment added at the end +of a file, for instance), the incremental rebuild reuses the unit's object code, and with +it the previous session's checksum and source. A clean build of the same source has the +new ones. + +### Reproduction + +With `-Zembed-source`, the stale file is visible directly: + +```sh +printf 'pub fn f(x: u32) -> u32 {\n x ^ 7\n}\n' > lib.rs +F="--crate-type lib -Cdebuginfo=2 -Zdwarf-version=5 -Zembed-source=yes -Cembed-bitcode=no" +rustc $F -C incremental=incr --out-dir rebuilt lib.rs +echo "// added after the first build" >> lib.rs +rustc $F -C incremental=incr --out-dir rebuilt lib.rs +rustc $F -C incremental=clean --out-dir clean lib.rs +for d in rebuilt clean; do (mkdir $d/x && cd $d/x && ar x ../liblib.rlib && llvm-dwarfdump --debug-line *.o | grep source:); done +``` + +On `nightly-2026-10-06` the rebuilt object embeds `lib.rs` as it was before the edit: + +```text +rebuilt: source: "pub fn f(x: u32) -> u32 {\n x ^ 7\n}\n" +clean: source: "pub fn f(x: u32) -> u32 {\n x ^ 7\n}\n// added after the first build\n" +``` + +Without `-Zembed-source` only the checksum is stale. It is in the LLVM IR, so it shows in the +bitcode `rustc` embeds in rlib objects by default: the same steps without +`-Zdwarf-version=5 -Zembed-source=yes -Cembed-bitcode=no` give objects whose embedded +bitcode, disassembled, differs only in the MD5 of `lib.rs` (the old one in the rebuilt +object, the current one in the clean build) and in the module hash computed from it. (DWARF line tables on Linux carry no MD5s, because the compile unit's own file entry +has none and LLVM emits them for all files or none.) + +It also changes executables built with optimizations, through ThinLTO: the `.llvm.` +suffix of a promoted symbol is a hash of its module's bitcode, checksum included. With this +`main.rs`: + +```rust +mod a { + #[inline(never)] + pub fn get(v: &[u32], i: usize) -> u32 { + v[i] * 3 + } +} + +mod b { + pub fn run(v: &[u32]) -> u32 { + crate::a::get(v, 1) + v[0] + } +} + +fn main() { + let v: Vec = std::env::args().map(|a| a.len() as u32).collect(); + println!("{}", b::run(&v)); +} +``` + +```sh +F="--edition 2021 -Copt-level=2 -Cdebuginfo=2 -Ccodegen-units=4" +rustc $F -C incremental=incr -o rebuilt main.rs +echo "// a comment at the end" >> main.rs +rustc $F -C incremental=incr -o rebuilt main.rs +rustc $F -C incremental=clean -o clean main.rs +diff <(nm rebuilt | awk '{print $3}' | sort) <(nm clean | awk '{print $3}' | sort) +``` + +```text +< anon.a976517f1c32298f1e683e92c2a4747a.0.llvm.15415811769069898958 +< anon.a976517f1c32298f1e683e92c2a4747a.1.llvm.15415811769069898958 +--- +> anon.a976517f1c32298f1e683e92c2a4747a.0.llvm.4724874333102837038 +> anon.a976517f1c32298f1e683e92c2a4747a.1.llvm.4724874333102837038 +``` + +The code, data and debug sections are identical; only those names differ. Two clean builds +agree, and with `-Cdebuginfo=0` the rebuild agrees with the clean build. + +I expected the rebuilt objects to equal the clean build's, as they do when the same edit +is made with debuginfo off. + +### Consequences + +- With `-Zembed-source`, a debugger that shows embedded source shows a file that no longer + exists. With several codegen units, one binary can embed different versions of the same + file, depending on which units were reused. +- The checksum exists so a debugger can tell whether the source on disk matches the binary. + A reused unit claims the old file, so a debugger that checks it would reject the current + file for code built from it. CodeView (MSVC targets) carries these checksums + (#113707 made them SHA256); I have not tried this on Windows. +- Incremental builds are not reproducible against clean builds once ThinLTO runs (any + `opt-level` above 0 with more than one codegen unit), which is the default for a profile + with optimizations and incremental compilation on. + +### Since when + +Checksums came with #69718 (1.44). Bitcode embedded in rlibs by default came in 1.45; +from 1.45 on, the reproduction above without `-Zembed-source` gives a rebuilt object with the +old MD5 (checked on 1.44.0, 1.45.0, 1.46.0, 1.47.0, 1.55.0, 1.90.0 and the nightly). The +executable-level difference depends on what ThinLTO promotes; it appears on 1.47.0, 1.51.0, +1.60.0 to 1.90.0 and the nightly, and not on 1.55.0. `-Zembed-source` came with #126985. + +### Possible fixes + +The codegen unit's dependency node would have to depend on the contents of each file its +debuginfo names. That makes every unit with code from a file red whenever the file changes, +comments included, which costs codegen reuse; with `-Zembed-source` there is no way around +it. For the checksum alone, a cheaper option might be to leave it out of debuginfo in +incremental sessions, or to fix it up when a unit is reused, but either changes what +incremental builds emit. A query fingerprinting one source file's contents, read by +`file_metadata`, would make the dependency explicit, as the candidate fix for the same +problem in reused metadata does ([report](issue-stale-metadata-reuse.md)). + +### How it was found + +[mirth](https://github.com/PowderworksCode/mirth) fuzzes edits on a five-crate workspace +and compares every incremental rebuild's artifacts with a clean build's. At `-Copt-level=2` +a third of the rebuilds gave binaries that differed only in `.llvm.` suffixes; the +modules behind them had pre-LTO bitcode that differed only in a `DIFile` checksum. diff --git a/docs/hunt/repro.sh b/docs/hunt/repro.sh index 51b6bb3..a67c14a 100755 --- a/docs/hunt/repro.sh +++ b/docs/hunt/repro.sh @@ -1,5 +1,5 @@ #!/usr/bin/env bash -# The three reproductions in hunt.md, with plain rustc (RUSTC, or the pinned +# The reproductions in hunt.md, with plain rustc (RUSTC, or the pinned # nightly). Each prints "same" or "DIFFER". set -u here=$(cd "$(dirname "$0")" && pwd) @@ -42,3 +42,15 @@ rc --crate-type lib --crate-name dep --emit=metadata,link -Cincremental="$d/i6" if cmp -s "$d/s1/libdep.rmeta" "$d/s2/libdep.rmeta"; then echo same elif cmp -s "$d/s1/libdep.rmeta" "$d/before.rmeta"; then echo "DIFFER (the previous session's metadata, republished)" else echo DIFFER; fi + +echo -n "stale-debuginfo, embedded source after a comment at the end, incremental vs clean: " +mkdir -p "$d/e1" "$d/e2" +printf 'pub fn f(x: u32) -> u32 {\n x ^ 7\n}\n' > "$d/e.rs" +ef="--crate-type lib --crate-name e -Cdebuginfo=2 -Zdwarf-version=5 -Zembed-source=yes -Cembed-bitcode=no" +rc $ef -Cincremental="$d/i7" --out-dir "$d/e1" "$d/e.rs" +echo "// added after the first build" >> "$d/e.rs" +rc $ef -Cincremental="$d/i7" --out-dir "$d/e1" "$d/e.rs" +rc $ef -Cincremental="$d/i8" --out-dir "$d/e2" "$d/e.rs" +n1=$(cd "$d/e1" && ar p libe.rlib | grep -a -c "added after the first build") +n2=$(cd "$d/e2" && ar p libe.rlib | grep -a -c "added after the first build") +if [ "$n1" = "$n2" ]; then echo same; else echo "DIFFER (the rebuilt object embeds the file as it was)"; fi diff --git a/docs/shadow-mode.md b/docs/shadow-mode.md index 412f261..b202c50 100644 --- a/docs/shadow-mode.md +++ b/docs/shadow-mode.md @@ -87,7 +87,9 @@ Not recomputed: argument), which are set rather than computed; - anything named in `RUSTC_VERIFY_REUSE_SKIP` (comma-separated query names). -Object files and replayed diagnostics are not checked yet. +Object files and replayed diagnostics are not checked yet. Finding 6 in [`hunt.md`](hunt.md), +reused object code whose debuginfo names the previous version of an edited file, is the kind +of bug a check of reused object files would catch on the spot. ## Does it find the known bugs? diff --git a/rustc/fuzz.py b/rustc/fuzz.py index ec5d4f6..3e0254a 100755 --- a/rustc/fuzz.py +++ b/rustc/fuzz.py @@ -326,7 +326,7 @@ def build(src, target): def reuse_checks(log): """What the compiler's own check of reused results (docs/hunt/verify-reuse.patch) found - stale: `query `, `metadata` or `allocation sharing`, once each.""" + stale: `query `, `metadata` or `allocation sharing `, once each.""" found = set() for line in log.splitlines(): if line.startswith("rustc-verify-reuse: query `"): @@ -334,7 +334,11 @@ def reuse_checks(log): elif line.startswith("rustc-verify-reuse: metadata"): found.add("metadata") elif line.startswith("rustc-verify-reuse: allocation shared differently"): - found.add("allocation sharing") + # Named by the two queries and typing modes, so that a new pattern is kept apart + # from a known one. + queries = re.findall(r"query `(\w+)`", line) + modes = re.findall(r"TypingModeEqWrapper\((\w+)\)", line) + found.add("allocation sharing " + " / ".join(queries + sorted(set(modes)))) return sorted(found) From 522acf739dad441de8c738581e8fbd6d3dee2da4 Mon Sep 17 00:00:00 2001 From: Zack Maril Date: Wed, 7 Oct 2026 13:39:59 +0000 Subject: [PATCH 03/25] Finding 6 as an Ur query; the Cranelift backend has the same bug sourceContentRead flags reads of a source file's contents or hash. Over the compiler it finds finding 6, the same reads in rustc_codegen_cranelift (confirmed: with -Zembed-source its rebuilt object embeds the old file), and the metadata site of finding 4. Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_018Mnrg9JXj9X1ht6Qkz2ybh --- docs/hunt/debuginfo-checksum-stopgap.patch | 19 +++++++++++++++++++ docs/hunt/issue-stale-debuginfo-source.md | 5 +++++ docs/ur-queries.md | 1 + ur/rustc/RoundTrip.rsc | 11 +++++++++++ 4 files changed, 36 insertions(+) create mode 100644 docs/hunt/debuginfo-checksum-stopgap.patch diff --git a/docs/hunt/debuginfo-checksum-stopgap.patch b/docs/hunt/debuginfo-checksum-stopgap.patch new file mode 100644 index 0000000..3560af1 --- /dev/null +++ b/docs/hunt/debuginfo-checksum-stopgap.patch @@ -0,0 +1,19 @@ +diff --git a/compiler/rustc_codegen_llvm/src/debuginfo/metadata.rs b/compiler/rustc_codegen_llvm/src/debuginfo/metadata.rs +index a3bcce345..ae8647888 100644 +--- a/compiler/rustc_codegen_llvm/src/debuginfo/metadata.rs ++++ b/compiler/rustc_codegen_llvm/src/debuginfo/metadata.rs +@@ -618,6 +618,14 @@ fn alloc_new_file_metadata<'ll>( + rustc_span::SourceFileHashAlgorithm::Blake3 => llvm::ChecksumKind::None, + }; + let hash_value = hex_encode(source_file.src_hash.hash_bytes()); ++ // mirth: testing only. A reused codegen unit keeps the checksum of the file as it was ++ // when the unit was compiled (hunt.md, finding 6); leave it out of incremental sessions ++ // so that the difference does not hide others. ++ let (hash_kind, hash_value) = if cx.sess().opts.incremental.is_some() { ++ (llvm::ChecksumKind::None, String::new()) ++ } else { ++ (hash_kind, hash_value) ++ }; + + let mut source = None; + let external_src; diff --git a/docs/hunt/issue-stale-debuginfo-source.md b/docs/hunt/issue-stale-debuginfo-source.md index 484a525..2ce7750 100644 --- a/docs/hunt/issue-stale-debuginfo-source.md +++ b/docs/hunt/issue-stale-debuginfo-source.md @@ -86,6 +86,11 @@ agree, and with `-Cdebuginfo=0` the rebuild agrees with the clean build. I expected the rebuilt objects to equal the clean build's, as they do when the same edit is made with debuginfo off. +The Cranelift backend reads the same fields (`debuginfo/line_info.rs` in +`rustc_codegen_cranelift`) and gives the same result: with `-Zcodegen-backend=cranelift` added +to the first reproduction, the rebuilt object embeds the old `lib.rs` and the clean one the +current file. + ### Consequences - With `-Zembed-source`, a debugger that shows embedded source shows a file that no longer diff --git a/docs/ur-queries.md b/docs/ur-queries.md index 765289e..ee511a5 100644 --- a/docs/ur-queries.md +++ b/docs/ur-queries.md @@ -21,6 +21,7 @@ Ur `ba84ca7`. | `hashOrderAlias` | a type alias for such a collection, which anything encoding a value of it writes in iteration order | #159677 (`DocLinkResMap`), from before its fix | 14 | none reach an encoder today | | `decodedFresh` | a decoder calling a function that reserves a fresh identity (`reserve*`, `fresh*`, `*next_id*`) where neither its name nor its definition deduplicates | the string literal decoded twice ([report](hunt/issue-literal-dedup.md)) | 1 | the bug | | `untrackedWhileEncoding` | inside an encoder (an `Encode*` impl or a function named `encode*`), a read of the session, the environment or the clock | stale metadata from the untracked source map ([report](hunt/issue-stale-metadata-reuse.md)) | 23 | the bug, and one more read of the same data; the rest benign | +| `sourceContentRead` | a read of a source file's contents or their hash (`src`, `src_hash`, `external_src`, `checksum_hash`, `unnormalized_source_len`) | reused codegen units keep an edited file's old checksum and embedded source ([report](hunt/issue-stale-debuginfo-source.md)) | 58 | the bug, the same bug in the Cranelift backend, and the metadata site of the stale-metadata bug; the rest are the source map itself, the lexer, dep-info written fresh each session, debugger visualizers (an edited script reaches the rebuild), and fields named `src` that are not source files | **Each query finds the bug it came from.** `hashOrderAlias` also finds #159677, a bug fixed in July 2026: with `DocLinkResMap` put back to the `UnordMap` it was before #159718, the diff --git a/ur/rustc/RoundTrip.rsc b/ur/rustc/RoundTrip.rsc index 95bb897..1a1fc69 100644 --- a/ur/rustc/RoundTrip.rsc +++ b/ur/rustc/RoundTrip.rsc @@ -11,6 +11,7 @@ classify("hashOrderEncoded", "a hash-ordered collection in a type that derives a classify("hashOrderAlias", "a type alias for a hash-ordered collection: anything encoding a value of it writes iteration order"); classify("decodedFresh", "a decoder that reserves a fresh identity, where creation may have deduplicated"); classify("untrackedWhileEncoding", "an encoder reading untracked state: the session, its source map, the environment, the clock"); +classify("sourceContentRead", "a read of a source file's contents or their hash, which nothing tracks"); data Rewrite = rewrite(str function, str description); rewrite("never", "rewrites nothing"); @@ -61,3 +62,13 @@ str untrackedWhileEncoding(Call c) = callee when let callee = unparse(c.operand), endsWith(callee, "env::var") || endsWith(callee, "env::var_os") || callee == "SystemTime::now" || callee == "Instant::now", inEncoder(c); + +// A source file's contents or their hash, read outside the code that tracks them: output +// built from it is reused when the file changes without changing anything tracked (a reused +// codegen unit's debuginfo keeps the old checksum, hunt.md finding 6; reused metadata kept the +// old source map, finding 4). +set[str] sourceContents = {"src_hash", "src", "external_src", "checksum_hash", "unnormalized_source_len"}; + +str sourceContentRead(Field f) = unparse(f.operand) + "." + name + when let name = unparse(f.field), + name in sourceContents; From 63ec7982f919f5ef7241f97d130e0394b43672ed Mon Sep 17 00:00:00 2001 From: Zack Maril Date: Wed, 7 Oct 2026 13:46:07 +0000 Subject: [PATCH 04/25] Finding 6: a run-make test, failing on the pinned compiler Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_018Mnrg9JXj9X1ht6Qkz2ybh --- docs/hunt.md | 2 +- docs/hunt/issue-stale-debuginfo-source.md | 11 +++ .../incr-debuginfo-embedded-source/rmake.rs | 78 +++++++++++++++++++ 3 files changed, 90 insertions(+), 1 deletion(-) create mode 100644 docs/hunt/tests/incr-debuginfo-embedded-source/rmake.rs diff --git a/docs/hunt.md b/docs/hunt.md index 8c3dbf2..3eab1f0 100644 --- a/docs/hunt.md +++ b/docs/hunt.md @@ -29,7 +29,7 @@ with `-Zthreads=8`. `rustc/check.sh wide` runs the ordinary checks. | 3 | with `-Zthreads=8`, two traits with `-> impl Trait` methods give different metadata from run to run | known: [#162202](https://github.com/rust-lang/rust/issues/162202) | | 4 | incremental rebuilds republish the previous session's metadata when an edit moves no span, so its source map describes old files | **looks new**; found later by the fuzzer and the history replay ([`scale.md`](scale.md)); root cause found, regression from #114669 (1.90); fix and regression test written | | 5 | `-Zemit-stack-sizes`, `-Zcodegen-source-order` and `-Zbuild-sdylib-interface` are untracked but change output that incremental compilation reuses | **looks new**; found by a query written from a closed bug and an option audit ([`ur-queries.md`](ur-queries.md)); report drafted | -| 6 | reused object code keeps the previous checksum of an edited source file in its debuginfo, and with `-Zembed-source` the previous file; with optimizations, ThinLTO symbol names then differ from a clean build | **looks new**; found by the fuzzer at `-Copt-level=2`; root cause found, since 1.44 (#69718); report drafted, no fix | +| 6 | reused object code keeps the previous checksum of an edited source file in its debuginfo, and with `-Zembed-source` the previous file; with optimizations, ThinLTO symbol names then differ from a clean build | **looks new**; found by the fuzzer at `-Copt-level=2`; root cause found, since 1.44 (#69718); report drafted and a regression test (`hunt/tests/incr-debuginfo-embedded-source`, failing: no fix) | Findings 1 and 2 are single-threaded: an ordinary `cargo build`, an edit, another `cargo build`, and the metadata differs from a clean build of the edited source. Both come diff --git a/docs/hunt/issue-stale-debuginfo-source.md b/docs/hunt/issue-stale-debuginfo-source.md index 2ce7750..68f2b80 100644 --- a/docs/hunt/issue-stale-debuginfo-source.md +++ b/docs/hunt/issue-stale-debuginfo-source.md @@ -123,6 +123,17 @@ incremental builds emit. A query fingerprinting one source file's contents, read `file_metadata`, would make the dependency explicit, as the candidate fix for the same problem in reused metadata does ([report](issue-stale-metadata-reuse.md)). +### A test + +`tests/run-make/incr-debuginfo-embedded-source` (in mirth at +`docs/hunt/tests/incr-debuginfo-embedded-source/rmake.rs`) builds a binary with +`-Zembed-source`, adds a comment at the end of `main.rs`, rebuilds incrementally and builds +clean, and checks that both embed the edited file. On `ea137335b` it fails: + +```text +the incremental rebuild embeds a different main.rs from a clean build: ["fn main() {\n println!(\"{}\", 7);\n}\n"] +``` + ### How it was found [mirth](https://github.com/PowderworksCode/mirth) fuzzes edits on a five-crate workspace diff --git a/docs/hunt/tests/incr-debuginfo-embedded-source/rmake.rs b/docs/hunt/tests/incr-debuginfo-embedded-source/rmake.rs new file mode 100644 index 0000000..a0bd310 --- /dev/null +++ b/docs/hunt/tests/incr-debuginfo-embedded-source/rmake.rs @@ -0,0 +1,78 @@ +//@ needs-target-std +//@ ignore-windows +//@ ignore-apple +//@ ignore-wasm (`object` doesn't handle wasm object files) +//@ ignore-cross-compile +//@ ignore-backends: gcc +// +// An incremental rebuild must embed the same source as a clean build of the same source. +// A codegen unit's debuginfo names each file with its checksum and, under `-Zembed-source`, +// its contents, read from the session's source map. Adding a comment at the end of the file +// changes nothing the codegen unit depends on, so the rebuild reused the unit's object code +// and with it the file as it was before the edit. + +use std::path::Path; +use std::rc::Rc; + +use gimli::{EndianRcSlice, Reader, RunTimeEndian}; +use object::{Object, ObjectSection}; +use run_make_support::{gimli, object, rfs, rustc}; + +const SOURCE: &str = "fn main() {\n println!(\"{}\", 7);\n}\n"; +const EDITED: &str = "fn main() {\n println!(\"{}\", 7);\n}\n// a comment at the end\n"; + +fn build(src: &str, incremental: &str, output: &str) { + rfs::write("main.rs", src); + rustc() + .input("main.rs") + .output(output) + .incremental(incremental) + .arg("-g") + .arg("-Zembed-source=yes") + .arg("-Cdwarf-version=5") + .run(); +} + +/// The source embedded for `main.rs`, in every unit that embeds it. +fn embedded(output: &str) -> Vec { + let data = rfs::read(Path::new(output)); + let obj = object::File::parse(data.as_slice()).unwrap(); + let endian = if obj.is_little_endian() { RunTimeEndian::Little } else { RunTimeEndian::Big }; + let dwarf = gimli::Dwarf::load(|section| -> Result<_, ()> { + let data = obj.section_by_name(section.name()).map(|s| s.uncompressed_data().unwrap()); + Ok(EndianRcSlice::new(Rc::from(data.unwrap_or_default().as_ref()), endian)) + }) + .unwrap(); + let mut found = Vec::new(); + let mut units = dwarf.units(); + while let Some(header) = units.next().unwrap() { + let unit = dwarf.unit(header).unwrap(); + let unit = unit.unit_ref(&dwarf); + let Some(program) = &unit.line_program else { continue }; + for file in program.header().file_names() { + let name = unit.attr_string(file.path_name()).unwrap(); + if name.to_string_lossy().unwrap() != "main.rs" { + continue; + } + if let Some(source) = file.source() { + let source = unit.attr_string(source).unwrap(); + found.push(source.to_string_lossy().unwrap().to_string()); + } + } + } + found +} + +fn main() { + build(SOURCE, "incr", "rebuilt"); + build(EDITED, "incr", "rebuilt"); + build(EDITED, "clean-incr", "clean"); + + let clean = embedded("clean"); + assert!(!clean.is_empty() && clean.iter().all(|s| s == EDITED), "clean build: {clean:?}"); + let rebuilt = embedded("rebuilt"); + assert!( + rebuilt.iter().all(|s| s == EDITED), + "the incremental rebuild embeds a different main.rs from a clean build: {rebuilt:?}" + ); +} From 7939a36b5731b5a2aab84131cf0340191ac7f7f3 Mon Sep 17 00:00:00 2001 From: Zack Maril Date: Wed, 7 Oct 2026 13:46:59 +0000 Subject: [PATCH 05/25] replay.py: --rustflags, for replaying with optimizations Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_018Mnrg9JXj9X1ht6Qkz2ybh --- rustc/replay.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/rustc/replay.py b/rustc/replay.py index be5cebe..548eb0d 100755 --- a/rustc/replay.py +++ b/rustc/replay.py @@ -48,6 +48,7 @@ p.add_argument("--keep", type=int, default=3, help="differences kept per crate") p.add_argument("--from", dest="start", type=int, default=0, help="first commit index to replay") p.add_argument("--to", dest="end", type=int, default=None, help="last commit index to replay") +p.add_argument("--rustflags", default="", help="more flags for every build, such as -Copt-level=2") p.add_argument("--no-verify-reuse", action="store_true", help="do not set RUSTC_VERIFY_REUSE (needs a compiler with docs/hunt/verify-reuse.patch)") args = p.parse_args() @@ -66,7 +67,7 @@ RUSTC_WRAPPER="", CARGO_INCREMENTAL="1", # A commit that denies warnings would otherwise stop building with a newer compiler. - RUSTFLAGS="--cap-lints=warn", + RUSTFLAGS=f"--cap-lints=warn {args.rustflags}".strip(), CARGO_TERM_COLOR="never", ) if not args.no_verify_reuse: From 1d6fed68d3229878a9e4cb4afa8781e74c9927f8 Mon Sep 17 00:00:00 2001 From: Zack Maril Date: Wed, 7 Oct 2026 14:19:35 +0000 Subject: [PATCH 06/25] Check reused codegen units; finding 7: asm warnings vanish on reuse verify-reuse.patch now also keeps each generated codegen unit's unoptimized IR and, when a later session reuses the unit, generates it again and compares (srcloc cookies left out). It flags finding 6 under -Zembed-source and is quiet on sink. Looking at its srcloc noise led to finding 7: warnings LLVM reports for inline assembly are not shown again when the unit is reused (1.60 through the nightly). Report, repro and a run-make test that fails on the pinned compiler. Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_018Mnrg9JXj9X1ht6Qkz2ybh --- docs/hunt.md | 3 +- docs/hunt/issue-asm-warnings-reused-cgu.md | 71 +++ docs/hunt/repro.sh | 11 + .../tests/incr-asm-warning-reused/rmake.rs | 37 ++ docs/hunt/verify-reuse.patch | 472 +++++++++++++++++- docs/shadow-mode.md | 26 +- rustc/fuzz.py | 5 +- 7 files changed, 617 insertions(+), 8 deletions(-) create mode 100644 docs/hunt/issue-asm-warnings-reused-cgu.md create mode 100644 docs/hunt/tests/incr-asm-warning-reused/rmake.rs diff --git a/docs/hunt.md b/docs/hunt.md index 3eab1f0..49d1470 100644 --- a/docs/hunt.md +++ b/docs/hunt.md @@ -30,6 +30,7 @@ with `-Zthreads=8`. `rustc/check.sh wide` runs the ordinary checks. | 4 | incremental rebuilds republish the previous session's metadata when an edit moves no span, so its source map describes old files | **looks new**; found later by the fuzzer and the history replay ([`scale.md`](scale.md)); root cause found, regression from #114669 (1.90); fix and regression test written | | 5 | `-Zemit-stack-sizes`, `-Zcodegen-source-order` and `-Zbuild-sdylib-interface` are untracked but change output that incremental compilation reuses | **looks new**; found by a query written from a closed bug and an option audit ([`ur-queries.md`](ur-queries.md)); report drafted | | 6 | reused object code keeps the previous checksum of an edited source file in its debuginfo, and with `-Zembed-source` the previous file; with optimizations, ThinLTO symbol names then differ from a clean build | **looks new**; found by the fuzzer at `-Copt-level=2`; root cause found, since 1.44 (#69718); report drafted and a regression test (`hunt/tests/incr-debuginfo-embedded-source`, failing: no fix) | +| 7 | warnings from inline assembly are not shown again when an incremental rebuild reuses the codegen unit | **looks new**; found while checking reused codegen units ([`shadow-mode.md`](shadow-mode.md)); on 1.60.0 through the nightly; report drafted ([draft](hunt/issue-asm-warnings-reused-cgu.md)) and a regression test (`hunt/tests/incr-asm-warning-reused`, failing: no fix) | Findings 1 and 2 are single-threaded: an ordinary `cargo build`, an edit, another `cargo build`, and the metadata differs from a clean build of the edited source. Both come @@ -38,7 +39,7 @@ incremental cache unchanged. All three reproduce with the official `nightly-2026 without mirth: `docs/hunt/repro.sh` runs them. None was searched for: P5 and P6 reported them on the first run of the new fixture. -Draft bug reports for 1, 2, 4, 5 and 6, written to be filed upstream, are +Draft bug reports for 1, 2, 4, 5, 6 and 7, written to be filed upstream, are [`hunt/issue-generics-order.md`](hunt/issue-generics-order.md), [`hunt/issue-literal-dedup.md`](hunt/issue-literal-dedup.md) and [`hunt/issue-stale-metadata-reuse.md`](hunt/issue-stale-metadata-reuse.md) and diff --git a/docs/hunt/issue-asm-warnings-reused-cgu.md b/docs/hunt/issue-asm-warnings-reused-cgu.md new file mode 100644 index 0000000..354b0cf --- /dev/null +++ b/docs/hunt/issue-asm-warnings-reused-cgu.md @@ -0,0 +1,71 @@ +# Warnings from inline assembly disappear when an incremental rebuild reuses the codegen unit + + + +The assembler's warnings about an `asm!` block are reported while LLVM compiles the codegen +unit that contains it. When an incremental rebuild reuses that unit's object code, LLVM does +not compile it, and the warning is not shown. Unlike warnings from queries, which incremental +compilation stores and shows again, warnings from LLVM are not stored. A clean build of the +same source shows the warning. + +### Reproduction + +```rust +// main.rs +mod m; +fn main() { + m::f(); +} +``` + +```rust +// m.rs +pub fn f() { + unsafe { std::arch::asm!(".warning \"from the assembler\"") } +} +``` + +```sh +rustc --edition 2021 -C codegen-units=4 -C incremental=incr -o rebuilt main.rs # the warning +echo "// a comment at the end" >> m.rs +rustc --edition 2021 -C codegen-units=4 -C incremental=incr -o rebuilt main.rs # nothing +rustc --edition 2021 -C codegen-units=4 -C incremental=clean-incr -o clean main.rs # the warning +``` + +The first build and the clean build print: + +```text +warning: from the assembler + --> m.rs:2:31 + | +2 | unsafe { std::arch::asm!(".warning \"from the assembler\"") } + | ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ +``` + +The rebuild prints nothing. I expected the rebuild to print the same warning as the clean +build, as it does for warnings from rustc itself. + +The same happens on 1.60.0, 1.65.0, 1.70.0, 1.75.0 and 1.90.0, and at `-C opt-level=2`. + +### Why + +Diagnostics emitted inside a query are saved as side effects and replayed when the query's +result is reused. Diagnostics from the LLVM backend are emitted through the codegen +coordinator's `SharedEmitter` while a module is compiled, and a work product (the `.o`, and +the `.bc` for ThinLTO) records nothing of them. A reused unit is never compiled, so its +warnings are lost, for as long as the unit stays reused. + +Inline assembly is the case found here; any other warning LLVM reports while compiling a +module would be lost the same way. + +### Possible fixes + +Record the diagnostics emitted while compiling a module with its work product and emit them +again when the work product is reused, or treat a unit whose compilation warned as not +reusable. + +### How it was found + +[mirth](https://github.com/PowderworksCode/mirth) checks reused codegen units against a fresh +codegen of the same unit. The only differences on its test workspace were `srcloc` cookies on +inline assembly, which led to looking at what inline-assembly diagnostics do across a reuse. diff --git a/docs/hunt/repro.sh b/docs/hunt/repro.sh index a67c14a..3b3e424 100755 --- a/docs/hunt/repro.sh +++ b/docs/hunt/repro.sh @@ -54,3 +54,14 @@ rc $ef -Cincremental="$d/i8" --out-dir "$d/e2" "$d/e.rs" n1=$(cd "$d/e1" && ar p libe.rlib | grep -a -c "added after the first build") n2=$(cd "$d/e2" && ar p libe.rlib | grep -a -c "added after the first build") if [ "$n1" = "$n2" ]; then echo same; else echo "DIFFER (the rebuilt object embeds the file as it was)"; fi + +echo -n "asm-warning, a warning from inline assembly after a comment at the end, incremental vs clean: " +mkdir -p "$d/w" +printf 'mod m;\nfn main() {\n m::f();\n}\n' > "$d/w/main.rs" +printf 'pub fn f() {\n unsafe { std::arch::asm!(".warning \\"from the assembler\\"") }\n}\n' > "$d/w/m.rs" +wf="--crate-type bin -Ccodegen-units=4" +"$rustc" --edition 2024 $wf -Cincremental="$d/i9" -o "$d/w/a" "$d/w/main.rs" 2> /dev/null +echo "// a comment at the end" >> "$d/w/m.rs" +n1=$("$rustc" --edition 2024 $wf -Cincremental="$d/i9" -o "$d/w/a" "$d/w/main.rs" 2>&1 | grep -c "from the assembler") +n2=$("$rustc" --edition 2024 $wf -Cincremental="$d/i10" -o "$d/w/b" "$d/w/main.rs" 2>&1 | grep -c "from the assembler") +if [ "$n1" = "$n2" ]; then echo same; else echo "DIFFER (the rebuild shows no warning)"; fi diff --git a/docs/hunt/tests/incr-asm-warning-reused/rmake.rs b/docs/hunt/tests/incr-asm-warning-reused/rmake.rs new file mode 100644 index 0000000..bac9e88 --- /dev/null +++ b/docs/hunt/tests/incr-asm-warning-reused/rmake.rs @@ -0,0 +1,37 @@ +//@ needs-target-std +//@ only-x86_64 +//@ ignore-cross-compile +// +// An incremental rebuild must show the same warnings as a clean build of the same source. A +// warning from inline assembly is reported by LLVM while it compiles the codegen unit, and +// nothing records it with the unit's work product, so a rebuild that reused the unit showed +// nothing. + +use run_make_support::{rfs, rustc}; + +const MAIN: &str = "mod m;\nfn main() {\n m::f();\n}\n"; +const M: &str = "pub fn f() {\n unsafe { std::arch::asm!(\".warning \\\"from the assembler\\\"\") }\n}\n"; + +fn build(incremental: &str, output: &str) -> String { + rustc() + .input("main.rs") + .output(output) + .incremental(incremental) + .codegen_units(4) + .run() + .stderr_utf8() +} + +fn main() { + rfs::write("main.rs", MAIN); + rfs::write("m.rs", M); + assert!(build("incr", "rebuilt").contains("from the assembler")); + rfs::write("m.rs", format!("{M}// a comment at the end\n")); + let rebuilt = build("incr", "rebuilt"); + let clean = build("clean-incr", "clean"); + assert!(clean.contains("from the assembler"), "clean build: {clean}"); + assert!( + rebuilt.contains("from the assembler"), + "the incremental rebuild shows no warning from the assembler; the clean build shows:\n{clean}" + ); +} diff --git a/docs/hunt/verify-reuse.patch b/docs/hunt/verify-reuse.patch index 8136775..a83eba4 100644 --- a/docs/hunt/verify-reuse.patch +++ b/docs/hunt/verify-reuse.patch @@ -1,3 +1,469 @@ +diff --git a/compiler/rustc_codegen_llvm/src/back/llvm_backend.rs b/compiler/rustc_codegen_llvm/src/back/llvm_backend.rs +index 8780af2f6..374c78070 100644 +--- a/compiler/rustc_codegen_llvm/src/back/llvm_backend.rs ++++ b/compiler/rustc_codegen_llvm/src/back/llvm_backend.rs +@@ -74,6 +74,39 @@ fn compile_codegen_unit( + ) -> (ModuleCodegen, u64) { + base::compile_codegen_unit(tcx, cgu_name, bitcode_needed) + } ++ fn codegen_unit_again( ++ &self, ++ tcx: TyCtxt<'_>, ++ cgu_name: Symbol, ++ bitcode_needed: bool, ++ ) -> Option> { ++ Some(base::codegen_unit_again(tcx, cgu_name, bitcode_needed)) ++ } ++ fn unoptimized_code(&self, module: &ModuleCodegen) -> Option> { ++ // The module as text, without inline assembly's `srcloc` cookies: they are raw byte ++ // positions, which an edit before the code moves without changing it, and they are ++ // only emitted where a reused module never goes through LLVM again (see `asm.rs`). ++ let text = unsafe { ++ let raw = llvm::LLVMPrintModuleToString(module.module_llvm.llmod()); ++ let text = CStr::from_ptr(raw).to_string_lossy().into_owned(); ++ llvm::LLVMDisposeMessage(raw); ++ text ++ }; ++ let srclocs: Vec<&str> = text ++ .lines() ++ .filter_map(|line| line.split_once(", !srcloc ").map(|(_, node)| node)) ++ .collect(); ++ let mut out = String::with_capacity(text.len()); ++ for line in text.lines() { ++ let line = line.split_once(", !srcloc ").map_or(line, |(code, _)| code); ++ if srclocs.iter().any(|node| line.starts_with(&format!("{node} = "))) { ++ continue; ++ } ++ out.push_str(line); ++ out.push('\n'); ++ } ++ Some(out.into_bytes()) ++ } + } + + impl WriteBackendMethods for LlvmCodegenBackend { +diff --git a/compiler/rustc_codegen_llvm/src/base.rs b/compiler/rustc_codegen_llvm/src/base.rs +index 77cb37ede..4bd2556f3 100644 +--- a/compiler/rustc_codegen_llvm/src/base.rs ++++ b/compiler/rustc_codegen_llvm/src/base.rs +@@ -80,145 +80,155 @@ pub(crate) fn compile_codegen_unit( + // the time we needed for codegenning it. + let cost = time_to_codegen.as_nanos() as u64; + +- fn module_codegen( +- tcx: TyCtxt<'_>, +- cgu_name: Symbol, +- needs_bitcode: bool, +- ) -> ModuleCodegen { +- let cgu = tcx.codegen_unit(cgu_name); +- let _prof_timer = +- tcx.prof.generic_activity_with_arg_recorder("codegen_module", |recorder| { +- recorder.record_arg(cgu_name.to_string()); +- recorder.record_arg(cgu.size_estimate().to_string()); +- }); +- // Instantiate monomorphizations without filling out definitions yet... +- let llvm_module = ModuleLlvm::new(tcx, cgu_name.as_str()); +- { +- let mut cx = CodegenCx::new(tcx, cgu, &llvm_module, needs_bitcode); ++ (module, cost) ++} + +- // Declare and store globals shared by all offload kernels +- // +- // These globals are left in the LLVM-IR host module so all kernels can access them. +- // They are necessary for correct offload execution. We do this here to simplify the +- // `offload` intrinsic, avoiding the need for tracking whether it's the first +- // intrinsic call or not. +- let has_host_offload = cx +- .sess() +- .opts +- .unstable_opts +- .offload +- .iter() +- .any(|o| matches!(o, Offload::Host(_) | Offload::Test)); +- if has_host_offload && !cx.sess().target.is_like_gpu { +- cx.offload_globals.replace(Some(OffloadGlobals::declare(&cx))); +- } ++fn module_codegen( ++ tcx: TyCtxt<'_>, ++ cgu_name: Symbol, ++ needs_bitcode: bool, ++) -> ModuleCodegen { ++ let cgu = tcx.codegen_unit(cgu_name); ++ let _prof_timer = ++ tcx.prof.generic_activity_with_arg_recorder("codegen_module", |recorder| { ++ recorder.record_arg(cgu_name.to_string()); ++ recorder.record_arg(cgu.size_estimate().to_string()); ++ }); ++ // Instantiate monomorphizations without filling out definitions yet... ++ let llvm_module = ModuleLlvm::new(tcx, cgu_name.as_str()); ++ { ++ let mut cx = CodegenCx::new(tcx, cgu, &llvm_module, needs_bitcode); + +- let mono_items = cx.codegen_unit.items_in_deterministic_order(cx.tcx); +- for &(mono_item, data) in &mono_items { +- mono_item.predefine::>( +- &mut cx, +- cgu_name.as_str(), +- data.linkage, +- data.visibility, +- ); +- } ++ // Declare and store globals shared by all offload kernels ++ // ++ // These globals are left in the LLVM-IR host module so all kernels can access them. ++ // They are necessary for correct offload execution. We do this here to simplify the ++ // `offload` intrinsic, avoiding the need for tracking whether it's the first ++ // intrinsic call or not. ++ let has_host_offload = cx ++ .sess() ++ .opts ++ .unstable_opts ++ .offload ++ .iter() ++ .any(|o| matches!(o, Offload::Host(_) | Offload::Test)); ++ if has_host_offload && !cx.sess().target.is_like_gpu { ++ cx.offload_globals.replace(Some(OffloadGlobals::declare(&cx))); ++ } + +- // ... and now that we have everything pre-defined, fill out those definitions. +- for &(mono_item, item_data) in &mono_items { +- mono_item.define::>(&mut cx, cgu_name.as_str(), item_data); +- } ++ let mono_items = cx.codegen_unit.items_in_deterministic_order(cx.tcx); ++ for &(mono_item, data) in &mono_items { ++ mono_item.predefine::>( ++ &mut cx, ++ cgu_name.as_str(), ++ data.linkage, ++ data.visibility, ++ ); ++ } + +- // If this codegen unit contains the main function, also create the +- // wrapper here +- if let Some(entry) = +- maybe_create_entry_wrapper::>(&cx, cx.codegen_unit) +- { +- let mut attrs = +- attributes::sanitize_attrs(&cx, tcx, SanitizerFnAttrs::default(), None, None); +- // When pointer authentication is enabled, ensure that the ptrauth-* attributes are +- // also attached to the entry wrapper. +- // +- // FIXME(jchlanda) If it ever becomes necessary to ensure that all compiler +- // generated functions receive the ptrauth-* attributes, `declare_fn` or +- // `declare_raw_fn` could be used to provide those. +- if cx.sess().pointer_authentication() { +- let cfg = cx.sess().pointer_auth_config.as_ref().unwrap(); +- for ptrauth_attr in cfg.fn_attrs() { +- attrs.push(llvm::CreateAttrString(cx.llcx, ptrauth_attr)); +- } +- } +- attributes::apply_to_llfn(entry, llvm::AttributePlace::Function, &attrs); +- } ++ // ... and now that we have everything pre-defined, fill out those definitions. ++ for &(mono_item, item_data) in &mono_items { ++ mono_item.define::>(&mut cx, cgu_name.as_str(), item_data); ++ } + +- // Define Objective-C module info and module flags. Note, the module info will +- // also be added to the `llvm.compiler.used` variable, created later. ++ // If this codegen unit contains the main function, also create the ++ // wrapper here ++ if let Some(entry) = ++ maybe_create_entry_wrapper::>(&cx, cx.codegen_unit) ++ { ++ let mut attrs = ++ attributes::sanitize_attrs(&cx, tcx, SanitizerFnAttrs::default(), None, None); ++ // When pointer authentication is enabled, ensure that the ptrauth-* attributes are ++ // also attached to the entry wrapper. + // +- // These are only necessary when we need the linker to do its Objective-C-specific +- // magic. We could theoretically do it unconditionally, but at a slight cost to linker +- // performance in the common case where it's unnecessary. +- if !cx.objc_classrefs.borrow().is_empty() || !cx.objc_selrefs.borrow().is_empty() { +- if cx.objc_abi_version() == 1 { +- cx.define_objc_module_info(); +- } +- cx.add_objc_module_flags(); +- } +- ++ // FIXME(jchlanda) If it ever becomes necessary to ensure that all compiler ++ // generated functions receive the ptrauth-* attributes, `declare_fn` or ++ // `declare_raw_fn` could be used to provide those. + if cx.sess().pointer_authentication() { + let cfg = cx.sess().pointer_auth_config.as_ref().unwrap(); +- +- let aarch64_elf_pauthabi_version = +- cfg.calculate_pauth_abi_version(&cx.sess().target); +- if aarch64_elf_pauthabi_version != 0 { +- cx.add_ptrauth_pauthabi_version_and_platform_flags( +- aarch64_elf_pauthabi_version, +- ); +- } +- if cfg.elf_got { +- cx.add_ptrauth_elf_got_flag(); +- } +- if cx.sess().pointer_authentication_functions().is_some() { +- cx.add_ptrauth_sign_personality_flag(); ++ for ptrauth_attr in cfg.fn_attrs() { ++ attrs.push(llvm::CreateAttrString(cx.llcx, ptrauth_attr)); + } + } ++ attributes::apply_to_llfn(entry, llvm::AttributePlace::Function, &attrs); ++ } + +- // Finalize code coverage by injecting the coverage map. Note, the coverage map will +- // also be added to the `llvm.compiler.used` variable, created next. +- if cx.sess().instrument_coverage() { +- cx.coverageinfo_finalize(); ++ // Define Objective-C module info and module flags. Note, the module info will ++ // also be added to the `llvm.compiler.used` variable, created later. ++ // ++ // These are only necessary when we need the linker to do its Objective-C-specific ++ // magic. We could theoretically do it unconditionally, but at a slight cost to linker ++ // performance in the common case where it's unnecessary. ++ if !cx.objc_classrefs.borrow().is_empty() || !cx.objc_selrefs.borrow().is_empty() { ++ if cx.objc_abi_version() == 1 { ++ cx.define_objc_module_info(); + } ++ cx.add_objc_module_flags(); ++ } + +- // Create the llvm.used variable. +- if !cx.used_statics.is_empty() { +- cx.create_used_variable_impl(c"llvm.used", &cx.used_statics); +- } ++ if cx.sess().pointer_authentication() { ++ let cfg = cx.sess().pointer_auth_config.as_ref().unwrap(); + +- // Create the llvm.compiler.used variable. +- { +- let compiler_used_statics = cx.compiler_used_statics.borrow(); +- if !compiler_used_statics.is_empty() { +- cx.create_used_variable_impl(c"llvm.compiler.used", &compiler_used_statics); +- } ++ let aarch64_elf_pauthabi_version = ++ cfg.calculate_pauth_abi_version(&cx.sess().target); ++ if aarch64_elf_pauthabi_version != 0 { ++ cx.add_ptrauth_pauthabi_version_and_platform_flags( ++ aarch64_elf_pauthabi_version, ++ ); ++ } ++ if cfg.elf_got { ++ cx.add_ptrauth_elf_got_flag(); + } ++ if cx.sess().pointer_authentication_functions().is_some() { ++ cx.add_ptrauth_sign_personality_flag(); ++ } ++ } + +- // Run replace-all-uses-with for statics that need it. This must +- // happen after the llvm.used variables are created. +- for &(old_g, new_g) in cx.statics_to_rauw().borrow().iter() { +- unsafe { +- llvm::LLVMReplaceAllUsesWith(old_g, new_g); +- llvm::LLVMDeleteGlobal(old_g); +- } ++ // Finalize code coverage by injecting the coverage map. Note, the coverage map will ++ // also be added to the `llvm.compiler.used` variable, created next. ++ if cx.sess().instrument_coverage() { ++ cx.coverageinfo_finalize(); ++ } ++ ++ // Create the llvm.used variable. ++ if !cx.used_statics.is_empty() { ++ cx.create_used_variable_impl(c"llvm.used", &cx.used_statics); ++ } ++ ++ // Create the llvm.compiler.used variable. ++ { ++ let compiler_used_statics = cx.compiler_used_statics.borrow(); ++ if !compiler_used_statics.is_empty() { ++ cx.create_used_variable_impl(c"llvm.compiler.used", &compiler_used_statics); + } ++ } + +- // Finalize debuginfo +- if cx.sess().opts.debuginfo != DebugInfo::None { +- cx.debuginfo_finalize(); ++ // Run replace-all-uses-with for statics that need it. This must ++ // happen after the llvm.used variables are created. ++ for &(old_g, new_g) in cx.statics_to_rauw().borrow().iter() { ++ unsafe { ++ llvm::LLVMReplaceAllUsesWith(old_g, new_g); ++ llvm::LLVMDeleteGlobal(old_g); + } + } + +- ModuleCodegen::new_regular(cgu_name.to_string(), llvm_module) ++ // Finalize debuginfo ++ if cx.sess().opts.debuginfo != DebugInfo::None { ++ cx.debuginfo_finalize(); ++ } + } + +- (module, cost) ++ ModuleCodegen::new_regular(cgu_name.to_string(), llvm_module) ++} ++ ++/// For testing incremental compilation (`RUSTC_VERIFY_REUSE`): the codegen unit generated ++/// again, outside dependency tracking, as a check of one reused from the incremental cache. ++pub(crate) fn codegen_unit_again( ++ tcx: TyCtxt<'_>, ++ cgu_name: Symbol, ++ bitcode_needed: bool, ++) -> ModuleCodegen { ++ tcx.dep_graph.with_ignore(|| module_codegen(tcx, cgu_name, bitcode_needed)) + } + + pub(crate) fn set_link_section(llval: &Value, attrs: &CodegenFnAttrs) { +diff --git a/compiler/rustc_codegen_llvm/src/llvm/ffi.rs b/compiler/rustc_codegen_llvm/src/llvm/ffi.rs +index f576f29a1..bc9ce9580 100644 +--- a/compiler/rustc_codegen_llvm/src/llvm/ffi.rs ++++ b/compiler/rustc_codegen_llvm/src/llvm/ffi.rs +@@ -1649,6 +1649,7 @@ pub(crate) fn LLVMBuildFence<'a>( + pub(crate) fn LLVMGetHostCPUFeatures() -> *mut c_char; + + pub(crate) fn LLVMDisposeMessage(message: *mut c_char); ++ pub(crate) fn LLVMPrintModuleToString(M: &Module) -> *mut c_char; + + pub(crate) fn LLVMIsMultithreaded() -> Bool; + +diff --git a/compiler/rustc_codegen_ssa/src/base.rs b/compiler/rustc_codegen_ssa/src/base.rs +index f9dfd04fa..43aa5424c 100644 +--- a/compiler/rustc_codegen_ssa/src/base.rs ++++ b/compiler/rustc_codegen_ssa/src/base.rs +@@ -1,4 +1,5 @@ + use std::collections::BTreeSet; ++use std::path::PathBuf; + use std::sync::Arc; + use std::time::{Duration, Instant}; + use std::{cmp, iter}; +@@ -847,6 +848,7 @@ pub fn codegen_crate< + FxHashMap::default() + }; + ++ let verify_reuse = std::env::var_os("RUSTC_VERIFY_REUSE").is_some(); + for (i, cgu) in codegen_units.iter().enumerate() { + ongoing_codegen.wait_for_signal_to_codegen_item(); + ongoing_codegen.check_for_errors(tcx.sess); +@@ -868,9 +870,15 @@ pub fn codegen_crate< + // compilation hang on post-monomorphization errors. + tcx.dcx().abort_if_errors(); + ++ if verify_reuse { ++ record_codegen_unit(&backend, tcx, cgu.name(), &module); ++ } + submit_codegened_module_to_llvm(&ongoing_codegen.coordinator, module, cost); + } + CguReuse::PreLto => { ++ if verify_reuse { ++ verify_reused_codegen_unit(&backend, tcx, cgu.name(), bitcode_needed); ++ } + submit_pre_lto_module_to_llvm( + tcx, + &ongoing_codegen.coordinator, +@@ -885,6 +893,9 @@ pub fn codegen_crate< + tcx.dcx().abort_if_errors(); + } + CguReuse::PostLto => { ++ if verify_reuse { ++ verify_reused_codegen_unit(&backend, tcx, cgu.name(), bitcode_needed); ++ } + submit_post_lto_module_to_llvm( + &ongoing_codegen.coordinator, + CachedModuleCodegen { +@@ -1253,6 +1264,61 @@ pub(crate) fn provide(providers: &mut Providers) { + }; + } + ++/// For testing incremental compilation (`RUSTC_VERIFY_REUSE`): where a codegen unit's ++/// unoptimized code is kept, from the session that generated it, to check the unit when a ++/// later session reuses it. In the crate's incremental directory, outside any one session's. ++fn recorded_codegen_unit(tcx: TyCtxt<'_>, cgu_name: Symbol) -> Option { ++ let session: &std::path::Path = &tcx.incr_comp_session?.new_session_directory; ++ Some(session.parent()?.join("verify-reuse").join(format!("{cgu_name}.ll"))) ++} ++ ++fn record_codegen_unit( ++ backend: &B, ++ tcx: TyCtxt<'_>, ++ cgu_name: Symbol, ++ module: &ModuleCodegen, ++) { ++ let (Some(path), Some(code)) = ++ (recorded_codegen_unit(tcx, cgu_name), backend.unoptimized_code(module)) ++ else { ++ return; ++ }; ++ if let Some(dir) = path.parent() { ++ let _ = std::fs::create_dir_all(dir); ++ } ++ let _ = std::fs::write(path, code); ++} ++ ++/// Generates a reused codegen unit again and compares its unoptimized code with that of the ++/// session that generated the reused one. A difference prints a line starting ++/// `rustc-verify-reuse:` and keeps the fresh code next to the recorded one. ++fn verify_reused_codegen_unit( ++ backend: &B, ++ tcx: TyCtxt<'_>, ++ cgu_name: Symbol, ++ bitcode_needed: bool, ++) { ++ let Some(path) = recorded_codegen_unit(tcx, cgu_name) else { return }; ++ let Ok(recorded) = std::fs::read(&path) else { return }; ++ let Some(module) = backend.codegen_unit_again(tcx, cgu_name, bitcode_needed) else { return }; ++ let Some(fresh) = backend.unoptimized_code(&module) else { return }; ++ if std::env::var_os("RUSTC_VERIFY_REUSE").is_some_and(|v| v == "verbose") { ++ eprintln!("rustc-verify-reuse-checked: codegen unit `{cgu_name}`"); ++ } ++ if fresh != recorded { ++ let kept = path.with_extension("fresh.ll"); ++ let _ = std::fs::write(&kept, &fresh); ++ eprintln!( ++ "rustc-verify-reuse: codegen unit `{cgu_name}` of `{}` reused from the incremental cache differs from a fresh codegen (unoptimized code, {} and {} bytes); kept as {} next to {}", ++ tcx.crate_name(LOCAL_CRATE), ++ recorded.len(), ++ fresh.len(), ++ kept.display(), ++ path.display(), ++ ); ++ } ++} ++ + pub fn determine_cgu_reuse<'tcx>(tcx: TyCtxt<'tcx>, cgu: &CodegenUnit<'tcx>) -> CguReuse { + if !tcx.dep_graph.is_fully_enabled() + || tcx.sess.opts.unstable_opts.disable_incr_comp_backend_caching +diff --git a/compiler/rustc_codegen_ssa/src/traits/backend.rs b/compiler/rustc_codegen_ssa/src/traits/backend.rs +index e06ac36fd..1fea1d7c8 100644 +--- a/compiler/rustc_codegen_ssa/src/traits/backend.rs ++++ b/compiler/rustc_codegen_ssa/src/traits/backend.rs +@@ -167,4 +167,23 @@ fn compile_codegen_unit( + cgu_name: Symbol, + bitcode_needed: bool, + ) -> (ModuleCodegen, u64); ++ ++ /// For testing incremental compilation (`RUSTC_VERIFY_REUSE`): the codegen unit generated ++ /// again, outside dependency tracking, as a check of one reused from the incremental cache. ++ /// `None` if the backend cannot. ++ fn codegen_unit_again( ++ &self, ++ _tcx: TyCtxt<'_>, ++ _cgu_name: Symbol, ++ _bitcode_needed: bool, ++ ) -> Option> { ++ None ++ } ++ ++ /// For testing incremental compilation (`RUSTC_VERIFY_REUSE`): a generated module's code ++ /// before optimization, as bytes that are equal when the code is. `None` if the backend ++ /// cannot say. ++ fn unoptimized_code(&self, _module: &ModuleCodegen) -> Option> { ++ None ++ } + } diff --git a/compiler/rustc_incremental/src/persist/save.rs b/compiler/rustc_incremental/src/persist/save.rs index 46f47d6c8..69ea0af7a 100644 --- a/compiler/rustc_incremental/src/persist/save.rs @@ -217,7 +683,7 @@ index 956c59012..957ddcb20 100644 fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { // Add more details here if/when necessary. diff --git a/compiler/rustc_query_impl/src/incremental.rs b/compiler/rustc_query_impl/src/incremental.rs -index a030fe71e..f868ab5b2 100644 +index a030fe71e..4ce05e8e3 100644 --- a/compiler/rustc_query_impl/src/incremental.rs +++ b/compiler/rustc_query_impl/src/incremental.rs @@ -1,17 +1,20 @@ @@ -243,7 +709,7 @@ index a030fe71e..f868ab5b2 100644 use crate::query_vtables::for_each_query_vtable; -@@ -25,6 +28,317 @@ pub(crate) fn encode_query_values<'tcx>(tcx: TyCtxt<'tcx>, encoder: &mut CacheEn +@@ -25,6 +28,319 @@ pub(crate) fn encode_query_values<'tcx>(tcx: TyCtxt<'tcx>, encoder: &mut CacheEn }); } @@ -427,6 +893,8 @@ index a030fe71e..f868ab5b2 100644 +const NOT_RECOMPUTED: &[&str] = &["coroutine_by_move_body_def_id", "mir_borrowck"]; + +/// Whether any MIR or THIR of `def` computed this session has been stolen. ++// Only decides whether to check a value, outside dependency tracking. ++#[allow(rustc::untracked_query_information)] +fn body_stolen<'tcx>(tcx: TyCtxt<'tcx>, def: LocalDefId) -> bool { + use rustc_middle::queries::{ + mir_built, mir_drops_elaborated_and_const_checked as mir_elaborated, mir_promoted, diff --git a/docs/shadow-mode.md b/docs/shadow-mode.md index b202c50..769f1b4 100644 --- a/docs/shadow-mode.md +++ b/docs/shadow-mode.md @@ -46,7 +46,7 @@ reused item is checked over many sessions, with an option to check everything. ## The patch [`hunt/verify-reuse.patch`](hunt/verify-reuse.patch), against the pinned rustc, does this for -metadata and for query results. It is on when `RUSTC_VERIFY_REUSE` is set (`verbose` also +metadata, query results and codegen units. It is on when `RUSTC_VERIFY_REUSE` is set (`verbose` also counts what was checked), and every fuzzer and replay build now runs with it: a line starting `rustc-verify-reuse:` is a finding of kind `verify-reuse` (`reuse` in the replay). @@ -87,9 +87,27 @@ Not recomputed: argument), which are set rather than computed; - anything named in `RUSTC_VERIFY_REUSE_SKIP` (comma-separated query names). -Object files and replayed diagnostics are not checked yet. Finding 6 in [`hunt.md`](hunt.md), -reused object code whose debuginfo names the previous version of an edited file, is the kind -of bug a check of reused object files would catch on the spot. +**Codegen units.** When a codegen unit is generated, its unoptimized LLVM IR is kept in the +crate's incremental directory (`verify-reuse/.ll`, outside any one session's +directory). When a later session reuses the unit's object code, the unit is generated again, +outside dependency tracking, and its IR compared with the kept one; a difference keeps the +fresh IR as `.fresh.ll`. Inline assembly's `srcloc` cookies are left out of both: +they are raw byte positions that an edit earlier in the source map moves, and rustc emits +them only where a reused module never goes through LLVM again. Both sessions need the check +on. This is the check that would have caught [finding 6](hunt.md) on the spot: with +`-Zembed-source`, its reproduction prints + +```text +rustc-verify-reuse: codegen unit `21p0vejx42dvnj8u08b23lumy` of `lib` reused from the incremental cache differs from a fresh codegen (unoptimized code, 2102 and 2135 bytes) +``` + +and on `fixtures/sink` an edit and rebuild checks 257 reused units and prints nothing (with +finding 6's checksum left out of incremental sessions by +[`hunt/debuginfo-checksum-stopgap.patch`](hunt/debuginfo-checksum-stopgap.patch), a testing aid, +not a fix). It costs more than the rest: about 40% on that rebuild. + +Replayed diagnostics are not checked yet, and diagnostics LLVM emits while compiling a unit +are not replayed at all ([finding 7](hunt.md)). ## Does it find the known bugs? diff --git a/rustc/fuzz.py b/rustc/fuzz.py index 3e0254a..8a28ec5 100755 --- a/rustc/fuzz.py +++ b/rustc/fuzz.py @@ -326,13 +326,16 @@ def build(src, target): def reuse_checks(log): """What the compiler's own check of reused results (docs/hunt/verify-reuse.patch) found - stale: `query `, `metadata` or `allocation sharing `, once each.""" + stale: `query `, `metadata`, `codegen unit` or `allocation sharing `, + once each.""" found = set() for line in log.splitlines(): if line.startswith("rustc-verify-reuse: query `"): found.add("query " + line.split("`")[1]) elif line.startswith("rustc-verify-reuse: metadata"): found.add("metadata") + elif line.startswith("rustc-verify-reuse: codegen unit"): + found.add("codegen unit") elif line.startswith("rustc-verify-reuse: allocation shared differently"): # Named by the two queries and typing modes, so that a new pattern is kept apart # from a known one. From 29854d0a962a05133a325f63fb25225a27b24ad2 Mon Sep 17 00:00:00 2001 From: Zack Maril Date: Wed, 7 Oct 2026 14:24:48 +0000 Subject: [PATCH 07/25] Finding 7: optimization remarks are dropped the same way Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_018Mnrg9JXj9X1ht6Qkz2ybh --- docs/hunt/issue-asm-warnings-reused-cgu.md | 17 ++++++++++++++--- 1 file changed, 14 insertions(+), 3 deletions(-) diff --git a/docs/hunt/issue-asm-warnings-reused-cgu.md b/docs/hunt/issue-asm-warnings-reused-cgu.md index 354b0cf..a95ec83 100644 --- a/docs/hunt/issue-asm-warnings-reused-cgu.md +++ b/docs/hunt/issue-asm-warnings-reused-cgu.md @@ -1,4 +1,4 @@ -# Warnings from inline assembly disappear when an incremental rebuild reuses the codegen unit +# Warnings from inline assembly (and optimization remarks) disappear when an incremental rebuild reuses the codegen unit @@ -55,8 +55,19 @@ coordinator's `SharedEmitter` while a module is compiled, and a work product (th the `.bc` for ThinLTO) records nothing of them. A reused unit is never compiled, so its warnings are lost, for as long as the unit stays reused. -Inline assembly is the case found here; any other warning LLVM reports while compiling a -module would be lost the same way. +Optimization remarks go the same way. With this `m.rs` instead: + +```rust +pub fn f(n: usize) -> usize { + (0..n).map(|i| i * 3).sum() +} +``` + +and `main.rs` printing `m::f(std::env::args().count())`, the same three builds with +`-C opt-level=2 -C remark=all -C debuginfo=1` print 14 remarks located in `m.rs` the first +time and in the clean build, and none in the rebuild. Remarks are arguably a debugging aid, +but they are what `-C remark` is for, and an incremental build silently drops them for every +reused unit. ### Possible fixes From 92c32883a0db9825b73205ca19bedb6a304f1b8e Mon Sep 17 00:00:00 2001 From: Zack Maril Date: Wed, 7 Oct 2026 14:25:35 +0000 Subject: [PATCH 08/25] Untracked options that take values: none changes reused output Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_018Mnrg9JXj9X1ht6Qkz2ybh --- docs/hunt/issue-untracked-options.md | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/docs/hunt/issue-untracked-options.md b/docs/hunt/issue-untracked-options.md index c67f832..ee0693e 100644 --- a/docs/hunt/issue-untracked-options.md +++ b/docs/hunt/issue-untracked-options.md @@ -41,6 +41,10 @@ otherwise identical), diagnostics and files written, is The other 45 boolean untracked options gave the same output incrementally as clean. (`-Zdump-dep-graph` and `-Zno-parallel-backend` failed to build this crate either way.) +Untracked options that take a value, given one each (`-Ccodegen-units=1`, `=3`, +`-Zmir-include-spans=yes`, `-Zthreads=4`, `-Zterminal-urls=yes`, +`-Zignore-directory-in-diagnostics-source-blocks`, `-Cstrip=symbols`), also gave the same +output incrementally as clean. `-Csave-temps` writing no temporaries for reused codegen units is probably fine for a debugging option. From c03a79432cf17f77c60965226c58a24f0125cdf6 Mon Sep 17 00:00:00 2001 From: Zack Maril Date: Wed, 7 Oct 2026 14:47:06 +0000 Subject: [PATCH 09/25] Flag and thread runs; a threads variant of the fixture Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_018Mnrg9JXj9X1ht6Qkz2ybh --- docs/scale.md | 30 +++++++++++++++++++++ fixtures/sink-threads.patch | 53 +++++++++++++++++++++++++++++++++++++ 2 files changed, 83 insertions(+) create mode 100644 fixtures/sink-threads.patch diff --git a/docs/scale.md b/docs/scale.md index 77bd463..45c28d5 100644 --- a/docs/scale.md +++ b/docs/scale.md @@ -143,6 +143,36 @@ The last 1,455 of those comparisons also checked object code, binaries and diagn Millions of edits means about a week here, or several machines. +## Other flags + +Short fuzz runs (two workers, 40 edits each) with one flag set added to every build, on the +compiler with the three fixes, the reuse check and +[`hunt/debuginfo-checksum-stopgap.patch`](hunt/debuginfo-checksum-stopgap.patch): + +| flags | compared | differences from a clean build | +|---|---:|---| +| `-Copt-level=2` (before the stopgap) | 102 | binaries and objects in a third of the rebuilds: [finding 6](hunt.md) | +| `-Copt-level=2` | 152 | none | +| `-Cinstrument-coverage` (official nightly) | 126 | none besides the known metadata bugs | +| `-Cdebuginfo=line-tables-only` | 71 | none | +| `-Cpanic=abort` | 68 | none | +| `-Zshare-generics=yes -Copt-level=1` | 71 | none | +| `-Copt-level=s` | 67 | none | +| `-Ccodegen-units=1 -Copt-level=3` | 69 | none | +| `-Zdwarf-version=5` | 66 | none | +| `-Csplit-debuginfo=unpacked`, `=packed` | 74, 67 | every rebuild, but two clean builds differ too: objects name their `.dwo` files with the session suffix, so this oracle does not apply | + +The reuse check printed only the known allocation-sharing pattern in all of them. + +**Threads.** With `-Zthreads=8`, two clean builds of `fixtures/sink` already differ +(#162202: the definitions made for `impl Trait` and `async fn` in traits get indices in a +nondeterministic order). [`fixtures/sink-threads.patch`](../fixtures/sink-threads.patch) +turns the two such trait methods into boxed iterators and futures; with it, clean threaded +builds agree. With only one of the two changed, clean builds agreed but incremental rebuilds +differed from clean ones about half the time: the same out-of-order indices, made in one +session and kept by the next (the incremental tables for the made-up associated types ended +two indices earlier). That is #162202 reaching incremental sessions, not a new bug. + ## The survey [`properties.md`](properties.md) has 29 properties plus the crash baseline, ranked by how diff --git a/fixtures/sink-threads.patch b/fixtures/sink-threads.patch new file mode 100644 index 0000000..e32a02d --- /dev/null +++ b/fixtures/sink-threads.patch @@ -0,0 +1,53 @@ +diff -ru a/core/src/asyncs.rs b/core/src/asyncs.rs +--- a/core/src/asyncs.rs 2026-10-07 11:00:50.308004685 +0000 ++++ b/core/src/asyncs.rs 2026-10-07 14:46:19.219199385 +0000 +@@ -35,18 +35,21 @@ + a + b + } + +-#[allow(async_fn_in_trait)] ++/// A trait method returning a boxed future. (An `async fn` here makes `-Zthreads` builds ++/// nondeterministic: rust-lang/rust#162202.) + pub trait Fetch { +- async fn fetch(&self, key: &str) -> Option; ++ fn fetch<'a>(&'a self, key: &'a str) -> core::pin::Pin> + 'a>>; + } + + #[derive(Debug)] + pub struct Static(pub &'static [(&'static str, &'static str)]); + + impl Fetch for Static { +- async fn fetch(&self, key: &str) -> Option { +- YieldOnce::default().await; +- self.0.iter().find(|(k, _)| *k == key).map(|(_, v)| v.to_string()) ++ fn fetch<'a>(&'a self, key: &'a str) -> core::pin::Pin> + 'a>> { ++ Box::pin(async move { ++ YieldOnce::default().await; ++ self.0.iter().find(|(k, _)| *k == key).map(|(_, v)| v.to_string()) ++ }) + } + } + +diff -ru a/core/src/shapes.rs b/core/src/shapes.rs +--- a/core/src/shapes.rs 2026-10-07 11:00:50.316004651 +0000 ++++ b/core/src/shapes.rs 2026-10-07 14:27:50.983376965 +0000 +@@ -178,14 +178,15 @@ + } + } + +-/// Return-position `impl Trait` in a trait. ++/// A trait method returning a boxed iterator. (`impl Trait` here, with the `async fn` in ++/// `asyncs::Fetch`, makes `-Zthreads` builds nondeterministic: rust-lang/rust#162202.) + pub trait Labels { +- fn labels(&self) -> impl Iterator + '_; ++ fn labels(&self) -> Box + '_>; + } + + impl Labels for [Square] { +- fn labels(&self) -> impl Iterator + '_ { +- self.iter().map(|s| format!("square {}", s.side)) ++ fn labels(&self) -> Box + '_> { ++ Box::new(self.iter().map(|s| format!("square {}", s.side))) + } + } + From 904bc67077c9441cbc22233cbb1281f33c38834e Mon Sep 17 00:00:00 2001 From: Zack Maril Date: Wed, 7 Oct 2026 14:54:27 +0000 Subject: [PATCH 10/25] fuzz.py: --check, and a second clean build to tell nondeterminism (P5) from P6 The threads fixture also loses its async fns (rust-lang/rust#162202). Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_018Mnrg9JXj9X1ht6Qkz2ybh --- fixtures/sink-threads.patch | 61 +++++++++++++++++++++++++++++++++++-- rustc/fuzz.py | 24 ++++++++++++--- 2 files changed, 78 insertions(+), 7 deletions(-) diff --git a/fixtures/sink-threads.patch b/fixtures/sink-threads.patch index e32a02d..a10b719 100644 --- a/fixtures/sink-threads.patch +++ b/fixtures/sink-threads.patch @@ -1,8 +1,20 @@ diff -ru a/core/src/asyncs.rs b/core/src/asyncs.rs --- a/core/src/asyncs.rs 2026-10-07 11:00:50.308004685 +0000 -+++ b/core/src/asyncs.rs 2026-10-07 14:46:19.219199385 +0000 -@@ -35,18 +35,21 @@ - a + b ++++ b/core/src/asyncs.rs 2026-10-07 14:53:30.747111230 +0000 +@@ -30,23 +30,30 @@ + } + } + +-pub async fn add_later(a: u32, b: u32) -> u32 { +- YieldOnce::default().await; +- a + b ++// No `async fn` in this variant of the fixture: under `-Zthreads` they make builds ++// nondeterministic (rust-lang/rust#162202). ++pub fn add_later(a: u32, b: u32) -> impl Future { ++ async move { ++ YieldOnce::default().await; ++ a + b ++ } } -#[allow(async_fn_in_trait)] @@ -28,6 +40,23 @@ diff -ru a/core/src/asyncs.rs b/core/src/asyncs.rs } } +@@ -55,10 +62,12 @@ + async move |greeting: &str| format!("{greeting}, {name}") + } + +-pub async fn call_twice(f: impl AsyncFn(&str) -> String) -> String { +- let a = f("hello").await; +- let b = f("bye").await; +- format!("{a}; {b}") ++pub fn call_twice(f: impl AsyncFn(&str) -> String) -> impl Future { ++ async move { ++ let a = f("hello").await; ++ let b = f("bye").await; ++ format!("{a}; {b}") ++ } + } + + pub fn boxed_future(n: u32) -> Pin + Send>> { diff -ru a/core/src/shapes.rs b/core/src/shapes.rs --- a/core/src/shapes.rs 2026-10-07 11:00:50.316004651 +0000 +++ b/core/src/shapes.rs 2026-10-07 14:27:50.983376965 +0000 @@ -51,3 +80,29 @@ diff -ru a/core/src/shapes.rs b/core/src/shapes.rs } } +diff -ru a/mid/src/lib.rs b/mid/src/lib.rs +--- a/mid/src/lib.rs 2026-10-07 11:00:50.316004651 +0000 ++++ b/mid/src/lib.rs 2026-10-07 14:53:30.747111230 +0000 +@@ -96,14 +96,16 @@ + } + } + +-pub async fn lookup(store: &impl Fetch, keys: &[&str]) -> Vec { +- let mut out = Vec::new(); +- for key in keys { +- if let Some(v) = store.fetch(key).await { +- out.push(v); ++pub fn lookup<'a>(store: &'a impl Fetch, keys: &'a [&'a str]) -> impl Future> + 'a { ++ async move { ++ let mut out = Vec::new(); ++ for key in keys { ++ if let Some(v) = store.fetch(key).await { ++ out.push(v); ++ } + } ++ out + } +- out + } + + pub fn labels(squares: &[Square]) -> Vec { diff --git a/rustc/fuzz.py b/rustc/fuzz.py index 8a28ec5..3d802fd 100755 --- a/rustc/fuzz.py +++ b/rustc/fuzz.py @@ -16,6 +16,8 @@ run the binaries' output and exit status reuse the compiler's own check of what it reused (RUSTC_VERIFY_REUSE, docs/hunt/verify-reuse.patch) found nothing stale + P5 when anything differs, a second clean build is made; if the two + clean builds differ, that is reported instead (nondeterminism) ICE neither build crashed the compiler split both builds succeed or both fail @@ -56,6 +58,8 @@ p.add_argument("--rustflags", default="-Zincremental-verify-ich") p.add_argument("--seed", type=int, default=0) p.add_argument("--timeout", type=int, default=180, help="seconds before a build counts as hung") +p.add_argument("--check", action="store_true", + help="cargo check instead of cargo build: metadata only, no code or binaries") p.add_argument("--no-verify-reuse", action="store_true", help="do not set RUSTC_VERIFY_REUSE (needs a compiler with docs/hunt/verify-reuse.patch)") args = p.parse_args() @@ -281,7 +285,7 @@ def build(src, target): if a rustc was still running (a looping build script is the fixture's problem).""" t = time.time() proc = subprocess.Popen( - ["cargo", f"+{args.toolchain}", "build", "--workspace", "--offline", "-j", "4", + ["cargo", f"+{args.toolchain}", "check" if args.check else "build", "--workspace", "--offline", "-j", "4", "--target-dir", str(target), "--message-format=json-render-diagnostics"], cwd=src, env=env(), stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, start_new_session=True) @@ -464,6 +468,19 @@ def report(kind, detail, inc, clean, extra=None): stats["compared"] += 1 a, b = inc["rmetas"], clean["rmetas"] differ = sorted(r for r in set(a) | set(b) if a.get(r) != b.get(r)) + others = {k: v for k, v in artifacts.compare(inc["art"], clean["art"]).items() if k != "rmeta"} + if differ or others: + # Build clean once more: if two clean builds differ, the difference is + # nondeterminism (P5), not incremental reuse. + shutil.rmtree(target) + again = build(src, target) + c = again["rmetas"] + p5 = sorted(r for r in set(b) | set(c) if b.get(r) != c.get(r)) + p5 += [f"{k}: {v[0]}" for k, v in artifacts.compare(clean["art"], again["art"]).items() + if k != "rmeta"] + if again["ok"] and p5: + report("P5", p5[:10], clean, again) + differ, others = [], {} # Known: metadata reused unchanged from the previous session although a source # file changed (its hash and length in the source map are stale). stale = [r for r in differ if r in previous and previous[r] == a.get(r)] @@ -472,9 +489,8 @@ def report(kind, detail, inc, clean, extra=None): elif differ: report("P6", differ, inc, clean) # More oracles: object code in the rlibs, the binary, and the diagnostics. - for kind, detail in artifacts.compare(inc["art"], clean["art"]).items(): - if kind != "rmeta": - report(kind, detail[:10], inc, clean) + for kind, detail in others.items(): + report(kind, detail[:10], inc, clean) ra = run_exe(str(inc["exe"]).replace(str(target), str(inc_target)) if inc["exe"] else None) rb = run_exe(clean["exe"]) if ra != rb: From 549eb2422616cc2fa12688e6767e83ff32a5c33b Mon Sep 17 00:00:00 2001 From: Zack Maril Date: Wed, 7 Oct 2026 15:06:12 +0000 Subject: [PATCH 11/25] shadow-mode.md: what the check reports in the runs Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_018Mnrg9JXj9X1ht6Qkz2ybh --- docs/shadow-mode.md | 19 +++++++++++++++++++ 1 file changed, 19 insertions(+) diff --git a/docs/shadow-mode.md b/docs/shadow-mode.md index 769f1b4..48e2c94 100644 --- a/docs/shadow-mode.md +++ b/docs/shadow-mode.md @@ -109,6 +109,25 @@ not a fix). It costs more than the rest: about 40% on that rebuild. Replayed diagnostics are not checked yet, and diagnostics LLVM emits while compiling a unit are not replayed at all ([finding 7](hunt.md)). +## What it reports in the runs + +Over the long fuzz runs and the history replays (thousands of rebuilds), with the three fixes +applied and finding 6's stopgap, the check has printed two kinds of line. Neither has changed +any output yet. + +- **Constants evaluated in two typing modes.** `eval_to_const_value_raw` in `Codegen` mode + only retries in `PostAnalysis` mode, so a fresh computation of both shares one allocation. + Once one of them is recomputed and the other reused, they refer to two allocations with + the same bytes, and the next sessions keep them apart. Codegen merges equal constants and + metadata does not encode `Codegen`-mode results, so nothing differs. This is most of the + `verify-reuse` findings. +- **A function whose closing brace ends the file.** Codegen extends a scope to another file + when a location is outside the scope's file, and a position equal to the file's end counts + as outside (`adjust_dbg_scope_for_span`). With no newline after the brace, the return's + location gets a `DILexicalBlockFile` naming the same file; once text follows the brace a + fresh codegen does not, while a reused unit keeps it. The file is the same, so the line + table and the object are too. + ## Does it find the known bugs? Each fix reverted in turn on the patched compiler, with the reproduction from From acb362fc5d6300a6440e25f071e55a518bf5984d Mon Sep 17 00:00:00 2001 From: Zack Maril Date: Wed, 7 Oct 2026 15:18:08 +0000 Subject: [PATCH 12/25] scale.md: the threads fixture Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_018Mnrg9JXj9X1ht6Qkz2ybh --- docs/scale.md | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/docs/scale.md b/docs/scale.md index 45c28d5..f1603f2 100644 --- a/docs/scale.md +++ b/docs/scale.md @@ -167,8 +167,11 @@ The reuse check printed only the known allocation-sharing pattern in all of them **Threads.** With `-Zthreads=8`, two clean builds of `fixtures/sink` already differ (#162202: the definitions made for `impl Trait` and `async fn` in traits get indices in a nondeterministic order). [`fixtures/sink-threads.patch`](../fixtures/sink-threads.patch) -turns the two such trait methods into boxed iterators and futures; with it, clean threaded -builds agree. With only one of the two changed, clean builds agreed but incremental rebuilds +turns the two such trait methods into boxed iterators and futures, and the free `async fn`s +into functions returning `impl Future` (once the fuzzer duplicated an `async fn`, clean +threaded builds of the crate differed six ways in six builds, as in #162202's first +example); with it, clean threaded builds agree. The fuzzer now builds clean a second time +whenever anything differs, and reports P5 rather than P6 if the two clean builds disagree. With only one of the two changed, clean builds agreed but incremental rebuilds differed from clean ones about half the time: the same out-of-order indices, made in one session and kept by the next (the incremental tables for the made-up associated types ended two indices earlier). That is #162202 reaching incremental sessions, not a new bug. From b2e8ce5e5fcecb300011b529429276c94e4443d4 Mon Sep 17 00:00:00 2001 From: Zack Maril Date: Wed, 7 Oct 2026 15:19:26 +0000 Subject: [PATCH 13/25] fuzz.py: --p5-builds Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_018Mnrg9JXj9X1ht6Qkz2ybh --- rustc/fuzz.py | 18 ++++++++++++------ 1 file changed, 12 insertions(+), 6 deletions(-) diff --git a/rustc/fuzz.py b/rustc/fuzz.py index 3d802fd..57d958f 100755 --- a/rustc/fuzz.py +++ b/rustc/fuzz.py @@ -58,6 +58,8 @@ p.add_argument("--rustflags", default="-Zincremental-verify-ich") p.add_argument("--seed", type=int, default=0) p.add_argument("--timeout", type=int, default=180, help="seconds before a build counts as hung") +p.add_argument("--p5-builds", type=int, default=1, + help="clean builds made again when anything differs, to tell nondeterminism from P6") p.add_argument("--check", action="store_true", help="cargo check instead of cargo build: metadata only, no code or binaries") p.add_argument("--no-verify-reuse", action="store_true", @@ -472,12 +474,16 @@ def report(kind, detail, inc, clean, extra=None): if differ or others: # Build clean once more: if two clean builds differ, the difference is # nondeterminism (P5), not incremental reuse. - shutil.rmtree(target) - again = build(src, target) - c = again["rmetas"] - p5 = sorted(r for r in set(b) | set(c) if b.get(r) != c.get(r)) - p5 += [f"{k}: {v[0]}" for k, v in artifacts.compare(clean["art"], again["art"]).items() - if k != "rmeta"] + p5 = [] + for _ in range(args.p5_builds): + shutil.rmtree(target) + again = build(src, target) + c = again["rmetas"] + p5 = sorted(r for r in set(b) | set(c) if b.get(r) != c.get(r)) + p5 += [f"{k}: {v[0]}" for k, v in artifacts.compare(clean["art"], again["art"]).items() + if k != "rmeta"] + if p5 or not again["ok"]: + break if again["ok"] and p5: report("P5", p5[:10], clean, again) differ, others = [], {} From 7ae7deb668b831786a42453e60bdf562070cb056 Mon Sep 17 00:00:00 2001 From: Zack Maril Date: Wed, 7 Oct 2026 15:49:27 +0000 Subject: [PATCH 14/25] fuzz-replay.py: keep the last builds' compiler output Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_018Mnrg9JXj9X1ht6Qkz2ybh --- rustc/fuzz-replay.py | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/rustc/fuzz-replay.py b/rustc/fuzz-replay.py index e965b03..f85a58f 100755 --- a/rustc/fuzz-replay.py +++ b/rustc/fuzz-replay.py @@ -4,7 +4,8 @@ rustc/fuzz-replay.py --rustc --fixture fixtures/sink --finding --work [--upto N] -Prints which .rmeta files differ. --upto replays only the first N edits that +Prints which .rmeta files differ, and keeps the compiler's output of the last +incremental build and the clean build as inc.log and clean.log. --upto replays only the first N edits that were kept, to find where the difference appears. """ @@ -45,6 +46,7 @@ def build(t): for f in msg["filenames"]: if f.endswith(".rmeta"): rmetas[str(Path(f).relative_to(t))] = Path(f).read_bytes() + build.log = r.stderr return r.returncode == 0, rmetas @@ -71,8 +73,10 @@ def build(t): print(f"{step['edit']:18} {step['file']:24} {'built' if ok else 'failed'}") ok, inc = build(target) +(work / "inc.log").write_text(build.log) target.rename(inc_target) ok2, clean = build(target) +(work / "clean.log").write_text(build.log) differ = sorted(r for r in set(inc) | set(clean) if inc.get(r) != clean.get(r)) print(json.dumps({"kept_edits": kept, "inc_ok": ok, "clean_ok": ok2, "differ": [Path(d).name for d in differ]})) for d in differ: From 83b42ab9ace98dfdb6fe6ce5f3a179018e5c6a0e Mon Sep 17 00:00:00 2001 From: Zack Maril Date: Wed, 7 Oct 2026 16:34:43 +0000 Subject: [PATCH 15/25] Report untracked reads inside rustc; two more untracked options docs/hunt/report-untracked.patch: under RUSTC_REPORT_UNTRACKED, a read of an [UNTRACKED] option, of source file contents, or of the crate store's crate-wide state inside a task whose result may be reused is reported, unless the task declared it. On one sink build it names finding 5's three options and finding 6, and two new ones: -Zno-leak-check (an incremental rebuild accepts a program a clean build rejects) and -C extra-filename (reused metadata keeps the old value, since 1.90). fuzz.py and replay.py collect the lines in untracked.txt. Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_018Mnrg9JXj9X1ht6Qkz2ybh --- README.md | 1 + docs/hunt.md | 2 +- docs/hunt/debuginfo-checksum-stopgap.patch | 15 +- docs/hunt/issue-untracked-options.md | 63 +- docs/hunt/report-untracked.patch | 2972 ++++++++++++++++++++ docs/hunt/repro.sh | 15 + docs/untracked-reads.md | 84 + rustc/fuzz.py | 18 + rustc/replay.py | 6 + 9 files changed, 3158 insertions(+), 18 deletions(-) create mode 100644 docs/hunt/report-untracked.patch create mode 100644 docs/untracked-reads.md diff --git a/README.md b/README.md index 6de05a5..cbaf276 100644 --- a/README.md +++ b/README.md @@ -37,6 +37,7 @@ fixture, the third from fuzzing edits and replaying ten crates' git histories. - [`docs/motivating.md`](docs/motivating.md): the real rustc bugs behind each property, each reproduced before and after its fix - [`docs/ur-queries.md`](docs/ur-queries.md): the bugs' patterns, and closed bugs' patterns, as Ur queries over rustc's source - [`docs/shadow-mode.md`](docs/shadow-mode.md): checking reuse inside rustc, and what exists today +- [`docs/untracked-reads.md`](docs/untracked-reads.md): reporting reads of untracked state inside rustc - [`docs/plan.md`](docs/plan.md): the plan the work followed, with the properties ## An instrumented compiler diff --git a/docs/hunt.md b/docs/hunt.md index 49d1470..b1c0c7c 100644 --- a/docs/hunt.md +++ b/docs/hunt.md @@ -28,7 +28,7 @@ with `-Zthreads=8`. `rustc/check.sh wide` runs the ordinary checks. | 2 | incremental rebuilds encode a string literal twice where clean builds encode it once | **looks new**; root cause found, regression from #116707 (1.90); fix and regression test written | | 3 | with `-Zthreads=8`, two traits with `-> impl Trait` methods give different metadata from run to run | known: [#162202](https://github.com/rust-lang/rust/issues/162202) | | 4 | incremental rebuilds republish the previous session's metadata when an edit moves no span, so its source map describes old files | **looks new**; found later by the fuzzer and the history replay ([`scale.md`](scale.md)); root cause found, regression from #114669 (1.90); fix and regression test written | -| 5 | `-Zemit-stack-sizes`, `-Zcodegen-source-order` and `-Zbuild-sdylib-interface` are untracked but change output that incremental compilation reuses | **looks new**; found by a query written from a closed bug and an option audit ([`ur-queries.md`](ur-queries.md)); report drafted | +| 5 | five untracked options change results incremental compilation reuses; with `-Zno-leak-check`, a rebuild accepts a program a clean build rejects | **looks new**; the first three found by a query written from a closed bug and an option audit ([`ur-queries.md`](ur-queries.md)), `-Zno-leak-check` and `-C extra-filename` by reporting untracked reads ([`untracked-reads.md`](untracked-reads.md)); report drafted | | 6 | reused object code keeps the previous checksum of an edited source file in its debuginfo, and with `-Zembed-source` the previous file; with optimizations, ThinLTO symbol names then differ from a clean build | **looks new**; found by the fuzzer at `-Copt-level=2`; root cause found, since 1.44 (#69718); report drafted and a regression test (`hunt/tests/incr-debuginfo-embedded-source`, failing: no fix) | | 7 | warnings from inline assembly are not shown again when an incremental rebuild reuses the codegen unit | **looks new**; found while checking reused codegen units ([`shadow-mode.md`](shadow-mode.md)); on 1.60.0 through the nightly; report drafted ([draft](hunt/issue-asm-warnings-reused-cgu.md)) and a regression test (`hunt/tests/incr-asm-warning-reused`, failing: no fix) | diff --git a/docs/hunt/debuginfo-checksum-stopgap.patch b/docs/hunt/debuginfo-checksum-stopgap.patch index 3560af1..6c03133 100644 --- a/docs/hunt/debuginfo-checksum-stopgap.patch +++ b/docs/hunt/debuginfo-checksum-stopgap.patch @@ -1,18 +1,19 @@ -diff --git a/compiler/rustc_codegen_llvm/src/debuginfo/metadata.rs b/compiler/rustc_codegen_llvm/src/debuginfo/metadata.rs -index a3bcce345..ae8647888 100644 ---- a/compiler/rustc_codegen_llvm/src/debuginfo/metadata.rs -+++ b/compiler/rustc_codegen_llvm/src/debuginfo/metadata.rs -@@ -618,6 +618,14 @@ fn alloc_new_file_metadata<'ll>( +--- a/compiler/rustc_codegen_llvm/src/debuginfo/metadata.rs 2026-10-07 15:55:46.835604483 +0000 ++++ b/compiler/rustc_codegen_llvm/src/debuginfo/metadata.rs 2026-10-07 15:55:46.835604483 +0000 +@@ -617,8 +617,15 @@ + rustc_span::SourceFileHashAlgorithm::Sha256 => llvm::ChecksumKind::SHA256, rustc_span::SourceFileHashAlgorithm::Blake3 => llvm::ChecksumKind::None, }; - let hash_value = hex_encode(source_file.src_hash.hash_bytes()); +- rustc_data_structures::untracked::untracked_read("source file contents"); +- let hash_value = hex_encode(source_file.src_hash.hash_bytes()); + // mirth: testing only. A reused codegen unit keeps the checksum of the file as it was + // when the unit was compiled (hunt.md, finding 6); leave it out of incremental sessions + // so that the difference does not hide others. + let (hash_kind, hash_value) = if cx.sess().opts.incremental.is_some() { + (llvm::ChecksumKind::None, String::new()) + } else { -+ (hash_kind, hash_value) ++ rustc_data_structures::untracked::untracked_read("source file contents"); ++ (hash_kind, hex_encode(source_file.src_hash.hash_bytes())) + }; let mut source = None; diff --git a/docs/hunt/issue-untracked-options.md b/docs/hunt/issue-untracked-options.md index ee0693e..c0b58ac 100644 --- a/docs/hunt/issue-untracked-options.md +++ b/docs/hunt/issue-untracked-options.md @@ -1,11 +1,49 @@ -# `-Zemit-stack-sizes`, `-Zcodegen-source-order` and `-Zbuild-sdylib-interface` are untracked but change output that incremental compilation reuses +# Five untracked options change results that incremental compilation reuses; with `-Zno-leak-check`, a rebuild accepts a program a clean build rejects -All three are marked `[UNTRACKED]` in `compiler/rustc_session/src/options.rs`, so they're -left out of the dependency-tracking hash. Adding one between two incremental sessions -leaves the second session's results as the first's, and the option has no effect. A clean -build with the option produces something different. +`-Zno-leak-check`, `-C extra-filename`, `-Zemit-stack-sizes`, `-Zcodegen-source-order` and +`-Zbuild-sdylib-interface` are marked `[UNTRACKED]` in `compiler/rustc_session/src/options.rs`, +so they're left out of the dependency-tracking hash. Changing one between two incremental +sessions leaves the second session's results as the first's. A clean build with the second +session's options produces something different. + +### `-Zno-leak-check`: an incremental rebuild accepts a program a clean build rejects + +The leak check is part of type checking and trait selection +(`rustc_infer/src/infer/relate/higher_ranked.rs`), and whether it runs decides whether some +programs compile. `tests/ui/lub-glb/old-lub-glb-hr-noteq2.rs` is one: it passes with +`-Zno-leak-check` and is rejected without it. + +```sh +cp tests/ui/lub-glb/old-lub-glb-hr-noteq2.rs lib.rs +rustc --crate-type lib -C incremental=incr -Zno-leak-check lib.rs # compiles +rustc --crate-type lib -C incremental=incr lib.rs # compiles: typeck reused +rustc --crate-type lib -C incremental=clean lib.rs # error[E0308]: `match` arms have incompatible types +``` + +On `nightly-2026-10-06`, and on 1.75.0 and 1.90.0 with `RUSTC_BOOTSTRAP=1`, the second +command succeeds although the program is rejected without the flag. (On 1.60.0 the program +does not compile either way.) + +### `-C extra-filename`: reused metadata keeps the old value + +The metadata records the crate's own `-C extra-filename`, and a dependent records each +dependency's, as a hint for finding transitive dependencies' files (`locator.rs`). Since +metadata can be reused (#114669, 1.90), a rebuild with a different `-C extra-filename` +publishes metadata that names the old one: + +```sh +rustc --crate-type lib --emit=metadata -C incremental=incr -C extra-filename=-aaa --out-dir out lib.rs +rustc --crate-type lib --emit=metadata -C incremental=incr -C extra-filename=-bbb --out-dir out lib.rs +grep -a -c -- -aaa out/liblib-bbb.rmeta # 1 on 1.90.0 and the nightly, 0 on 1.89.0 +``` + +The lookup falls back to any matching file and checks the crate hash, so the stale hint +costs at most a wrong first guess; Cargo changes `-C metadata` along with +`-C extra-filename`, which is tracked. It is the same class, found the same way. + +### `-Zemit-stack-sizes`, `-Zcodegen-source-order`, `-Zbuild-sdylib-interface` ### Reproduction @@ -50,14 +88,19 @@ debugging option. ### Suggested fix -Mark `emit_stack_sizes`, `codegen_source_order` and `build_sdylib_interface` `[TRACKED]`. +Mark `no_leak_check`, `extra_filename`, `emit_stack_sizes`, `codegen_source_order` and +`build_sdylib_interface` `[TRACKED]` (`extra_filename` perhaps `[TRACKED_NO_CRATE_HASH]`). An audit like the one above could run in CI over every untracked option, so a new option marked `[UNTRACKED]` that changes reused output is caught when it is added; that is the long tail this issue's discussion worries about. ### How it was found -[mirth](https://github.com/PowderworksCode/mirth) wrote the pattern behind #66955 -(`--remap-path-prefix` untracked) as a query over rustc's source, reads of options, then -joined it with the options marked `[UNTRACKED]`. The audit is the differential check this -issue's third comment suggests. +[mirth](https://github.com/PowderworksCode/mirth) builds rustc with a check that reports, at +run time, every read of an `[UNTRACKED]` option (and of other state the dependency graph +does not track) inside a computation whose result incremental compilation may reuse +([`report-untracked.patch`](report-untracked.patch)). Leaving out options that only produce +debugging output, a build of its test workspace reports exactly these five. The first +three were found earlier by writing #66955's pattern as a query over rustc's source and +auditing every untracked boolean option differentially, which did not exercise the leak +check. diff --git a/docs/hunt/report-untracked.patch b/docs/hunt/report-untracked.patch new file mode 100644 index 0000000..6329762 --- /dev/null +++ b/docs/hunt/report-untracked.patch @@ -0,0 +1,2972 @@ +diff --git a/compiler/rustc_attr_parsing/src/attributes/rustc_internal.rs b/compiler/rustc_attr_parsing/src/attributes/rustc_internal.rs +index b9c03883d..d51e6b650 100644 +--- a/compiler/rustc_attr_parsing/src/attributes/rustc_internal.rs ++++ b/compiler/rustc_attr_parsing/src/attributes/rustc_internal.rs +@@ -784,7 +784,7 @@ fn extend( + cx: &mut AcceptContext<'_, '_>, + args: &ArgParser, + ) -> impl IntoIterator { +- if !cx.cx.sess.opts.unstable_opts.query_dep_graph { ++ if !(*cx.cx.sess.opts.unstable_opts.read_query_dep_graph()) { + cx.emit_err(AttributeRequiresOpt { span: cx.attr_span, opt: "-Z query-dep-graph" }); + } + let list = cx.expect_list(args, cx.attr_span)?; +@@ -877,7 +877,7 @@ impl SingleAttributeParser for RustcIfThisChangedParser { + const STABILITY: AttributeStability = unstable!(rustc_attrs); + + fn convert(cx: &mut AcceptContext<'_, '_>, args: &ArgParser) -> Option { +- if !cx.cx.sess.opts.unstable_opts.query_dep_graph { ++ if !(*cx.cx.sess.opts.unstable_opts.read_query_dep_graph()) { + cx.emit_err(AttributeRequiresOpt { span: cx.attr_span, opt: "-Z query-dep-graph" }); + } + match args { +@@ -942,7 +942,7 @@ fn extend( + cx: &mut AcceptContext<'_, '_>, + args: &ArgParser, + ) -> impl IntoIterator { +- if !cx.cx.sess.opts.unstable_opts.query_dep_graph { ++ if !(*cx.cx.sess.opts.unstable_opts.read_query_dep_graph()) { + cx.emit_err(AttributeRequiresOpt { span: cx.attr_span, opt: "-Z query-dep-graph" }); + } + let item = cx.expect_single_element_list(args, cx.attr_span)?; +diff --git a/compiler/rustc_attr_parsing/src/check_cfg.rs b/compiler/rustc_attr_parsing/src/check_cfg.rs +index fe8a6b91c..11cd97590 100644 +--- a/compiler/rustc_attr_parsing/src/check_cfg.rs ++++ b/compiler/rustc_attr_parsing/src/check_cfg.rs +@@ -20,7 +20,7 @@ fn sort_and_truncate_possibilities( + ) -> (Vec, usize) { + let possibilities_len = possibilities.len(); + +- let n_possibilities = if sess.opts.unstable_opts.check_cfg_all_expected { ++ let n_possibilities = if (*sess.opts.unstable_opts.read_check_cfg_all_expected()) { + possibilities.len() + } else { + match filter_well_known_names { +@@ -396,7 +396,7 @@ pub(crate) fn unexpected_cfg_value( + // basic heuristic, we use the "cheat" unstable feature enable method and the + // non-ui-testing enabled option. + || (matches!(sess.unstable_features, rustc_feature::UnstableFeatures::Cheat) +- && !sess.opts.unstable_opts.ui_testing); ++ && !(*sess.opts.unstable_opts.read_ui_testing())); + + let inst = |escape_quotes| { + to_check_cfg_arg(Ident::new(name, name_span), value.map(|(v, _s)| v), escape_quotes) +diff --git a/compiler/rustc_borrowck/src/nll.rs b/compiler/rustc_borrowck/src/nll.rs +index 62319ce5e..818307d64 100644 +--- a/compiler/rustc_borrowck/src/nll.rs ++++ b/compiler/rustc_borrowck/src/nll.rs +@@ -164,10 +164,10 @@ pub(crate) fn compute_regions<'tcx>( + + // If requested: dump NLL facts, and run legacy polonius analysis. + let polonius_output = polonius_facts.as_ref().and_then(|polonius_facts| { +- if infcx.tcx.sess.opts.unstable_opts.nll_facts { ++ if (*infcx.tcx.sess.opts.unstable_opts.read_nll_facts()) { + let def_id = body.source.def_id(); + let def_path = infcx.tcx.def_path(def_id); +- let dir_path = PathBuf::from(&infcx.tcx.sess.opts.unstable_opts.nll_facts_dir) ++ let dir_path = PathBuf::from(&(*infcx.tcx.sess.opts.unstable_opts.read_nll_facts_dir())) + .join(def_path.to_filename_friendly_no_crate()); + polonius_facts.write_to_dir(dir_path, location_table).unwrap(); + } +@@ -227,7 +227,7 @@ pub(super) fn dump_nll_mir<'tcx>( + // they're always disabled in mir-opt tests to make working with blessed dumps easier. + let options = PrettyPrintMirOptions { + include_extra_comments: matches!( +- infcx.tcx.sess.opts.unstable_opts.mir_include_spans, ++ (*infcx.tcx.sess.opts.unstable_opts.read_mir_include_spans()), + MirIncludeSpans::On | MirIncludeSpans::Nll + ), + }; +diff --git a/compiler/rustc_borrowck/src/polonius/dump.rs b/compiler/rustc_borrowck/src/polonius/dump.rs +index c32b118a9..a4e125446 100644 +--- a/compiler/rustc_borrowck/src/polonius/dump.rs ++++ b/compiler/rustc_borrowck/src/polonius/dump.rs +@@ -90,7 +90,7 @@ pub(crate) fn dump_polonius_mir<'tcx>( + // mir-include-spans` on the CLI still has priority. + let options = PrettyPrintMirOptions { + include_extra_comments: matches!( +- tcx.sess.opts.unstable_opts.mir_include_spans, ++ (*tcx.sess.opts.unstable_opts.read_mir_include_spans()), + MirIncludeSpans::On | MirIncludeSpans::Nll + ), + }; +diff --git a/compiler/rustc_borrowck/src/polonius/legacy/facts.rs b/compiler/rustc_borrowck/src/polonius/legacy/facts.rs +index 1f8177477..3b51986d6 100644 +--- a/compiler/rustc_borrowck/src/polonius/legacy/facts.rs ++++ b/compiler/rustc_borrowck/src/polonius/legacy/facts.rs +@@ -56,7 +56,7 @@ impl PoloniusFacts { + /// Returns `true` if there is a need to gather `PoloniusFacts` given the + /// current `-Z` flags. + fn enabled(tcx: TyCtxt<'_>) -> bool { +- tcx.sess.opts.unstable_opts.nll_facts ++ (*tcx.sess.opts.unstable_opts.read_nll_facts()) + || tcx.sess.opts.unstable_opts.polonius.is_legacy_enabled() + } + +diff --git a/compiler/rustc_codegen_llvm/src/back/llvm_backend.rs b/compiler/rustc_codegen_llvm/src/back/llvm_backend.rs +index 8780af2f6..95628cdcd 100644 +--- a/compiler/rustc_codegen_llvm/src/back/llvm_backend.rs ++++ b/compiler/rustc_codegen_llvm/src/back/llvm_backend.rs +@@ -74,6 +74,39 @@ fn compile_codegen_unit( + ) -> (ModuleCodegen, u64) { + base::compile_codegen_unit(tcx, cgu_name, bitcode_needed) + } ++ fn codegen_unit_again( ++ &self, ++ tcx: TyCtxt<'_>, ++ cgu_name: Symbol, ++ bitcode_needed: bool, ++ ) -> Option> { ++ Some(base::codegen_unit_again(tcx, cgu_name, bitcode_needed)) ++ } ++ fn unoptimized_code(&self, module: &ModuleCodegen) -> Option> { ++ // The module as text, without inline assembly's `srcloc` cookies: they are raw byte ++ // positions, which an edit before the code moves without changing it, and they are ++ // only emitted where a reused module never goes through LLVM again (see `asm.rs`). ++ let text = unsafe { ++ let raw = llvm::LLVMPrintModuleToString(module.module_llvm.llmod()); ++ let text = CStr::from_ptr(raw).to_string_lossy().into_owned(); ++ llvm::LLVMDisposeMessage(raw); ++ text ++ }; ++ let srclocs: Vec<&str> = text ++ .lines() ++ .filter_map(|line| line.split_once(", !srcloc ").map(|(_, node)| node)) ++ .collect(); ++ let mut out = String::with_capacity(text.len()); ++ for line in text.lines() { ++ let line = line.split_once(", !srcloc ").map_or(line, |(code, _)| code); ++ if srclocs.iter().any(|node| line.starts_with(&format!("{node} = "))) { ++ continue; ++ } ++ out.push_str(line); ++ out.push('\n'); ++ } ++ Some(out.into_bytes()) ++ } + } + + impl WriteBackendMethods for LlvmCodegenBackend { +@@ -370,7 +403,7 @@ fn join_codegen( + .expect("Expected LlvmCodegenBackend's OngoingCodegen, found Box") + .join(sess, incr_comp_session, crate_info); + +- if sess.opts.unstable_opts.llvm_time_trace { ++ if (*sess.opts.unstable_opts.read_llvm_time_trace()) { + sess.time("llvm_dump_timing_file", || { + let file_name = outputs.with_extension("llvm_timings.json"); + llvm_util::time_trace_profiler_finish(&file_name); +diff --git a/compiler/rustc_codegen_llvm/src/back/write.rs b/compiler/rustc_codegen_llvm/src/back/write.rs +index a95f25954..746cfd95c 100644 +--- a/compiler/rustc_codegen_llvm/src/back/write.rs ++++ b/compiler/rustc_codegen_llvm/src/back/write.rs +@@ -204,7 +204,7 @@ pub(crate) fn target_machine_factory( + let abi = SmallCStr::new(sess.target.llvm_abiname.desc()); + let trap_unreachable = + sess.opts.unstable_opts.trap_unreachable.unwrap_or(sess.target.trap_unreachable); +- let emit_stack_size_section = sess.opts.unstable_opts.emit_stack_sizes; ++ let emit_stack_size_section = (*sess.opts.unstable_opts.read_emit_stack_sizes()); + + let verbose_asm = sess.opts.unstable_opts.verbose_asm; + let relax_elf_relocations = +diff --git a/compiler/rustc_codegen_llvm/src/base.rs b/compiler/rustc_codegen_llvm/src/base.rs +index 77cb37ede..4bd2556f3 100644 +--- a/compiler/rustc_codegen_llvm/src/base.rs ++++ b/compiler/rustc_codegen_llvm/src/base.rs +@@ -80,145 +80,155 @@ pub(crate) fn compile_codegen_unit( + // the time we needed for codegenning it. + let cost = time_to_codegen.as_nanos() as u64; + +- fn module_codegen( +- tcx: TyCtxt<'_>, +- cgu_name: Symbol, +- needs_bitcode: bool, +- ) -> ModuleCodegen { +- let cgu = tcx.codegen_unit(cgu_name); +- let _prof_timer = +- tcx.prof.generic_activity_with_arg_recorder("codegen_module", |recorder| { +- recorder.record_arg(cgu_name.to_string()); +- recorder.record_arg(cgu.size_estimate().to_string()); +- }); +- // Instantiate monomorphizations without filling out definitions yet... +- let llvm_module = ModuleLlvm::new(tcx, cgu_name.as_str()); +- { +- let mut cx = CodegenCx::new(tcx, cgu, &llvm_module, needs_bitcode); ++ (module, cost) ++} + +- // Declare and store globals shared by all offload kernels +- // +- // These globals are left in the LLVM-IR host module so all kernels can access them. +- // They are necessary for correct offload execution. We do this here to simplify the +- // `offload` intrinsic, avoiding the need for tracking whether it's the first +- // intrinsic call or not. +- let has_host_offload = cx +- .sess() +- .opts +- .unstable_opts +- .offload +- .iter() +- .any(|o| matches!(o, Offload::Host(_) | Offload::Test)); +- if has_host_offload && !cx.sess().target.is_like_gpu { +- cx.offload_globals.replace(Some(OffloadGlobals::declare(&cx))); +- } ++fn module_codegen( ++ tcx: TyCtxt<'_>, ++ cgu_name: Symbol, ++ needs_bitcode: bool, ++) -> ModuleCodegen { ++ let cgu = tcx.codegen_unit(cgu_name); ++ let _prof_timer = ++ tcx.prof.generic_activity_with_arg_recorder("codegen_module", |recorder| { ++ recorder.record_arg(cgu_name.to_string()); ++ recorder.record_arg(cgu.size_estimate().to_string()); ++ }); ++ // Instantiate monomorphizations without filling out definitions yet... ++ let llvm_module = ModuleLlvm::new(tcx, cgu_name.as_str()); ++ { ++ let mut cx = CodegenCx::new(tcx, cgu, &llvm_module, needs_bitcode); + +- let mono_items = cx.codegen_unit.items_in_deterministic_order(cx.tcx); +- for &(mono_item, data) in &mono_items { +- mono_item.predefine::>( +- &mut cx, +- cgu_name.as_str(), +- data.linkage, +- data.visibility, +- ); +- } ++ // Declare and store globals shared by all offload kernels ++ // ++ // These globals are left in the LLVM-IR host module so all kernels can access them. ++ // They are necessary for correct offload execution. We do this here to simplify the ++ // `offload` intrinsic, avoiding the need for tracking whether it's the first ++ // intrinsic call or not. ++ let has_host_offload = cx ++ .sess() ++ .opts ++ .unstable_opts ++ .offload ++ .iter() ++ .any(|o| matches!(o, Offload::Host(_) | Offload::Test)); ++ if has_host_offload && !cx.sess().target.is_like_gpu { ++ cx.offload_globals.replace(Some(OffloadGlobals::declare(&cx))); ++ } + +- // ... and now that we have everything pre-defined, fill out those definitions. +- for &(mono_item, item_data) in &mono_items { +- mono_item.define::>(&mut cx, cgu_name.as_str(), item_data); +- } ++ let mono_items = cx.codegen_unit.items_in_deterministic_order(cx.tcx); ++ for &(mono_item, data) in &mono_items { ++ mono_item.predefine::>( ++ &mut cx, ++ cgu_name.as_str(), ++ data.linkage, ++ data.visibility, ++ ); ++ } + +- // If this codegen unit contains the main function, also create the +- // wrapper here +- if let Some(entry) = +- maybe_create_entry_wrapper::>(&cx, cx.codegen_unit) +- { +- let mut attrs = +- attributes::sanitize_attrs(&cx, tcx, SanitizerFnAttrs::default(), None, None); +- // When pointer authentication is enabled, ensure that the ptrauth-* attributes are +- // also attached to the entry wrapper. +- // +- // FIXME(jchlanda) If it ever becomes necessary to ensure that all compiler +- // generated functions receive the ptrauth-* attributes, `declare_fn` or +- // `declare_raw_fn` could be used to provide those. +- if cx.sess().pointer_authentication() { +- let cfg = cx.sess().pointer_auth_config.as_ref().unwrap(); +- for ptrauth_attr in cfg.fn_attrs() { +- attrs.push(llvm::CreateAttrString(cx.llcx, ptrauth_attr)); +- } +- } +- attributes::apply_to_llfn(entry, llvm::AttributePlace::Function, &attrs); +- } ++ // ... and now that we have everything pre-defined, fill out those definitions. ++ for &(mono_item, item_data) in &mono_items { ++ mono_item.define::>(&mut cx, cgu_name.as_str(), item_data); ++ } + +- // Define Objective-C module info and module flags. Note, the module info will +- // also be added to the `llvm.compiler.used` variable, created later. ++ // If this codegen unit contains the main function, also create the ++ // wrapper here ++ if let Some(entry) = ++ maybe_create_entry_wrapper::>(&cx, cx.codegen_unit) ++ { ++ let mut attrs = ++ attributes::sanitize_attrs(&cx, tcx, SanitizerFnAttrs::default(), None, None); ++ // When pointer authentication is enabled, ensure that the ptrauth-* attributes are ++ // also attached to the entry wrapper. + // +- // These are only necessary when we need the linker to do its Objective-C-specific +- // magic. We could theoretically do it unconditionally, but at a slight cost to linker +- // performance in the common case where it's unnecessary. +- if !cx.objc_classrefs.borrow().is_empty() || !cx.objc_selrefs.borrow().is_empty() { +- if cx.objc_abi_version() == 1 { +- cx.define_objc_module_info(); +- } +- cx.add_objc_module_flags(); +- } +- ++ // FIXME(jchlanda) If it ever becomes necessary to ensure that all compiler ++ // generated functions receive the ptrauth-* attributes, `declare_fn` or ++ // `declare_raw_fn` could be used to provide those. + if cx.sess().pointer_authentication() { + let cfg = cx.sess().pointer_auth_config.as_ref().unwrap(); +- +- let aarch64_elf_pauthabi_version = +- cfg.calculate_pauth_abi_version(&cx.sess().target); +- if aarch64_elf_pauthabi_version != 0 { +- cx.add_ptrauth_pauthabi_version_and_platform_flags( +- aarch64_elf_pauthabi_version, +- ); +- } +- if cfg.elf_got { +- cx.add_ptrauth_elf_got_flag(); +- } +- if cx.sess().pointer_authentication_functions().is_some() { +- cx.add_ptrauth_sign_personality_flag(); ++ for ptrauth_attr in cfg.fn_attrs() { ++ attrs.push(llvm::CreateAttrString(cx.llcx, ptrauth_attr)); + } + } ++ attributes::apply_to_llfn(entry, llvm::AttributePlace::Function, &attrs); ++ } + +- // Finalize code coverage by injecting the coverage map. Note, the coverage map will +- // also be added to the `llvm.compiler.used` variable, created next. +- if cx.sess().instrument_coverage() { +- cx.coverageinfo_finalize(); ++ // Define Objective-C module info and module flags. Note, the module info will ++ // also be added to the `llvm.compiler.used` variable, created later. ++ // ++ // These are only necessary when we need the linker to do its Objective-C-specific ++ // magic. We could theoretically do it unconditionally, but at a slight cost to linker ++ // performance in the common case where it's unnecessary. ++ if !cx.objc_classrefs.borrow().is_empty() || !cx.objc_selrefs.borrow().is_empty() { ++ if cx.objc_abi_version() == 1 { ++ cx.define_objc_module_info(); + } ++ cx.add_objc_module_flags(); ++ } + +- // Create the llvm.used variable. +- if !cx.used_statics.is_empty() { +- cx.create_used_variable_impl(c"llvm.used", &cx.used_statics); +- } ++ if cx.sess().pointer_authentication() { ++ let cfg = cx.sess().pointer_auth_config.as_ref().unwrap(); + +- // Create the llvm.compiler.used variable. +- { +- let compiler_used_statics = cx.compiler_used_statics.borrow(); +- if !compiler_used_statics.is_empty() { +- cx.create_used_variable_impl(c"llvm.compiler.used", &compiler_used_statics); +- } ++ let aarch64_elf_pauthabi_version = ++ cfg.calculate_pauth_abi_version(&cx.sess().target); ++ if aarch64_elf_pauthabi_version != 0 { ++ cx.add_ptrauth_pauthabi_version_and_platform_flags( ++ aarch64_elf_pauthabi_version, ++ ); ++ } ++ if cfg.elf_got { ++ cx.add_ptrauth_elf_got_flag(); + } ++ if cx.sess().pointer_authentication_functions().is_some() { ++ cx.add_ptrauth_sign_personality_flag(); ++ } ++ } + +- // Run replace-all-uses-with for statics that need it. This must +- // happen after the llvm.used variables are created. +- for &(old_g, new_g) in cx.statics_to_rauw().borrow().iter() { +- unsafe { +- llvm::LLVMReplaceAllUsesWith(old_g, new_g); +- llvm::LLVMDeleteGlobal(old_g); +- } ++ // Finalize code coverage by injecting the coverage map. Note, the coverage map will ++ // also be added to the `llvm.compiler.used` variable, created next. ++ if cx.sess().instrument_coverage() { ++ cx.coverageinfo_finalize(); ++ } ++ ++ // Create the llvm.used variable. ++ if !cx.used_statics.is_empty() { ++ cx.create_used_variable_impl(c"llvm.used", &cx.used_statics); ++ } ++ ++ // Create the llvm.compiler.used variable. ++ { ++ let compiler_used_statics = cx.compiler_used_statics.borrow(); ++ if !compiler_used_statics.is_empty() { ++ cx.create_used_variable_impl(c"llvm.compiler.used", &compiler_used_statics); + } ++ } + +- // Finalize debuginfo +- if cx.sess().opts.debuginfo != DebugInfo::None { +- cx.debuginfo_finalize(); ++ // Run replace-all-uses-with for statics that need it. This must ++ // happen after the llvm.used variables are created. ++ for &(old_g, new_g) in cx.statics_to_rauw().borrow().iter() { ++ unsafe { ++ llvm::LLVMReplaceAllUsesWith(old_g, new_g); ++ llvm::LLVMDeleteGlobal(old_g); + } + } + +- ModuleCodegen::new_regular(cgu_name.to_string(), llvm_module) ++ // Finalize debuginfo ++ if cx.sess().opts.debuginfo != DebugInfo::None { ++ cx.debuginfo_finalize(); ++ } + } + +- (module, cost) ++ ModuleCodegen::new_regular(cgu_name.to_string(), llvm_module) ++} ++ ++/// For testing incremental compilation (`RUSTC_VERIFY_REUSE`): the codegen unit generated ++/// again, outside dependency tracking, as a check of one reused from the incremental cache. ++pub(crate) fn codegen_unit_again( ++ tcx: TyCtxt<'_>, ++ cgu_name: Symbol, ++ bitcode_needed: bool, ++) -> ModuleCodegen { ++ tcx.dep_graph.with_ignore(|| module_codegen(tcx, cgu_name, bitcode_needed)) + } + + pub(crate) fn set_link_section(llval: &Value, attrs: &CodegenFnAttrs) { +diff --git a/compiler/rustc_codegen_llvm/src/debuginfo/metadata.rs b/compiler/rustc_codegen_llvm/src/debuginfo/metadata.rs +index a3bcce345..33eec74df 100644 +--- a/compiler/rustc_codegen_llvm/src/debuginfo/metadata.rs ++++ b/compiler/rustc_codegen_llvm/src/debuginfo/metadata.rs +@@ -617,11 +617,13 @@ fn alloc_new_file_metadata<'ll>( + rustc_span::SourceFileHashAlgorithm::Sha256 => llvm::ChecksumKind::SHA256, + rustc_span::SourceFileHashAlgorithm::Blake3 => llvm::ChecksumKind::None, + }; ++ rustc_data_structures::untracked::untracked_read("source file contents"); + let hash_value = hex_encode(source_file.src_hash.hash_bytes()); + + let mut source = None; + let external_src; + if cx.sess().opts.unstable_opts.embed_source { ++ rustc_data_structures::untracked::untracked_read("source file contents"); + source = source_file.src.as_deref().map(String::as_str); + if source.is_none() { + cx.tcx.sess.source_map().ensure_source_file_source_present(source_file); +diff --git a/compiler/rustc_codegen_llvm/src/llvm/ffi.rs b/compiler/rustc_codegen_llvm/src/llvm/ffi.rs +index f576f29a1..bc9ce9580 100644 +--- a/compiler/rustc_codegen_llvm/src/llvm/ffi.rs ++++ b/compiler/rustc_codegen_llvm/src/llvm/ffi.rs +@@ -1649,6 +1649,7 @@ pub(crate) fn LLVMBuildFence<'a>( + pub(crate) fn LLVMGetHostCPUFeatures() -> *mut c_char; + + pub(crate) fn LLVMDisposeMessage(message: *mut c_char); ++ pub(crate) fn LLVMPrintModuleToString(M: &Module) -> *mut c_char; + + pub(crate) fn LLVMIsMultithreaded() -> Bool; + +diff --git a/compiler/rustc_codegen_llvm/src/llvm_util.rs b/compiler/rustc_codegen_llvm/src/llvm_util.rs +index a582e897e..859a2e5c4 100644 +--- a/compiler/rustc_codegen_llvm/src/llvm_util.rs ++++ b/compiler/rustc_codegen_llvm/src/llvm_util.rs +@@ -108,10 +108,10 @@ fn llvm_arg_to_arg_name(full_arg: &str) -> &str { + }; + // Set the llvm "program name" to make usage and invalid argument messages more clear. + add("rustc -Cllvm-args=\"...\" with", true); +- if sess.opts.unstable_opts.time_llvm_passes { ++ if (*sess.opts.unstable_opts.read_time_llvm_passes()) { + add("-time-passes", false); + } +- if sess.opts.unstable_opts.print_llvm_passes { ++ if (*sess.opts.unstable_opts.read_print_llvm_passes()) { + add("-debug-pass=Structure", false); + } + if sess.target.generate_arange_section +@@ -159,7 +159,7 @@ fn llvm_arg_to_arg_name(full_arg: &str) -> &str { + }; + } + +- if sess.opts.unstable_opts.llvm_time_trace { ++ if (*sess.opts.unstable_opts.read_llvm_time_trace()) { + unsafe { llvm::LLVMRustTimeTraceProfilerInitialize() }; + } + +diff --git a/compiler/rustc_codegen_ssa/src/assert_module_sources.rs b/compiler/rustc_codegen_ssa/src/assert_module_sources.rs +index 431783552..8013ec2bd 100644 +--- a/compiler/rustc_codegen_ssa/src/assert_module_sources.rs ++++ b/compiler/rustc_codegen_ssa/src/assert_module_sources.rs +@@ -55,7 +55,7 @@ pub fn assert_module_sources(tcx: TyCtxt<'_>, set_reuse: &dyn Fn(&mut CguReuseTr + let mut ams = AssertModuleSource { + tcx, + available_cgus, +- cgu_reuse_tracker: if tcx.sess.opts.unstable_opts.query_dep_graph { ++ cgu_reuse_tracker: if (*tcx.sess.opts.unstable_opts.read_query_dep_graph()) { + CguReuseTracker::new() + } else { + CguReuseTracker::new_disabled() +@@ -66,7 +66,7 @@ pub fn assert_module_sources(tcx: TyCtxt<'_>, set_reuse: &dyn Fn(&mut CguReuseTr + + set_reuse(&mut ams.cgu_reuse_tracker); + +- if tcx.sess.opts.unstable_opts.print_mono_items ++ if (*tcx.sess.opts.unstable_opts.read_print_mono_items()) + && let Some(data) = &ams.cgu_reuse_tracker.data + { + data.actual_reuse.items().all(|(cgu, reuse)| { +@@ -105,7 +105,7 @@ fn check_attrs(&mut self, attrs: &[rustc_attr_ir::Attribute]) { + | CguFields::PartitionCodegened { cfg, module } + | CguFields::PartitionReused { cfg, module }) = cgu_fields; + +- if !self.tcx.sess.opts.unstable_opts.query_dep_graph { ++ if !(*self.tcx.sess.opts.unstable_opts.read_query_dep_graph()) { + self.tcx.dcx().emit_fatal(diagnostics::MissingQueryDepGraph { span }); + } + +diff --git a/compiler/rustc_codegen_ssa/src/back/archive.rs b/compiler/rustc_codegen_ssa/src/back/archive.rs +index 18b5c001f..46e2ccde7 100644 +--- a/compiler/rustc_codegen_ssa/src/back/archive.rs ++++ b/compiler/rustc_codegen_ssa/src/back/archive.rs +@@ -276,7 +276,7 @@ fn create_mingw_dll_import_lib( + + fn find_binutils_dlltool(sess: &Session) -> OsString { + assert!(sess.target.options.is_like_windows && !sess.target.options.is_like_msvc); +- if let Some(dlltool_path) = &sess.opts.cg.dlltool { ++ if let Some(dlltool_path) = &(*sess.opts.cg.read_dlltool()) { + return dlltool_path.clone().into_os_string(); + } + +diff --git a/compiler/rustc_codegen_ssa/src/back/link.rs b/compiler/rustc_codegen_ssa/src/back/link.rs +index 45ab9b8bc..6aaa60461 100644 +--- a/compiler/rustc_codegen_ssa/src/back/link.rs ++++ b/compiler/rustc_codegen_ssa/src/back/link.rs +@@ -353,7 +353,7 @@ pub fn link_binary( + .unwrap_or_else(|error| { + sess.dcx().emit_fatal(diagnostics::CreateTempDir { error }) + }); +- let path = MaybeTempDir::new(tmpdir, sess.opts.cg.save_temps); ++ let path = MaybeTempDir::new(tmpdir, (*sess.opts.cg.read_save_temps())); + + let crate_name = format!("{}", crate_info.local_crate_name); + let out_filename = output.file_for_writing(outputs, OutputType::Exe, &crate_name); +@@ -442,7 +442,7 @@ pub fn link_binary( + // Remove the temporary object file and metadata if we aren't saving temps. + sess.time("link_binary_remove_temps", || { + // If the user requests that temporaries are saved, don't delete any. +- if sess.opts.cg.save_temps { ++ if (*sess.opts.cg.read_save_temps()) { + return; + } + +@@ -1535,7 +1535,7 @@ fn link_natively( + } + } + +- let strip = sess.opts.cg.strip; ++ let strip = (*sess.opts.cg.read_strip()); + + if sess.target.is_like_darwin { + let stripcmd = "rust-objcopy"; +@@ -1722,7 +1722,7 @@ fn add_sanitizer_libraries( + return; + } + +- if sess.opts.unstable_opts.external_clangrt { ++ if (*sess.opts.unstable_opts.read_external_clangrt()) { + // Linking against in-tree sanitizer runtimes is disabled via + // `-Z external-clangrt` + return; +@@ -1924,11 +1924,11 @@ fn adjust_flavor_to_features( + } + } + +- let features = sess.opts.cg.linker_features; ++ let features = (*sess.opts.cg.read_linker_features()); + + // linker and linker flavor specified via command line have precedence over what the target + // specification specifies +- let linker_flavor = match sess.opts.cg.linker_flavor { ++ let linker_flavor = match (*sess.opts.cg.read_linker_flavor()) { + // The linker flavors that are non-target specific can be directly translated to LinkerFlavor + Some(LinkerFlavorCli::Llbc) => Some(LinkerFlavor::Llbc), + // The linker flavors that corresponds to targets needs logic that keeps the base LinkerFlavor +@@ -1936,7 +1936,7 @@ fn adjust_flavor_to_features( + linker_flavor.map(|flavor| sess.target.linker_flavor.with_cli_hints(flavor)) + } + }; +- if let Some(ret) = infer_from(sess, sess.opts.cg.linker.clone(), linker_flavor, features) { ++ if let Some(ret) = infer_from(sess, (*sess.opts.cg.read_linker()).clone(), linker_flavor, features) { + return ret; + } + +@@ -2317,7 +2317,7 @@ fn self_contained_components( + // Turn the backwards compatible bool values for `self_contained` into fully inferred + // `LinkSelfContainedComponents`. + let self_contained = +- if let Some(self_contained) = sess.opts.cg.link_self_contained.explicitly_set { ++ if let Some(self_contained) = (*sess.opts.cg.read_link_self_contained()).explicitly_set { + // Emit an error if the user requested self-contained mode on the CLI but the target + // explicitly refuses it. + if sess.target.link_self_contained.is_disabled() { +@@ -2401,7 +2401,7 @@ fn add_pre_link_args(cmd: &mut dyn Linker, sess: &Session, flavor: LinkerFlavor) + cmd.verbatim_args(args.iter().map(Deref::deref)); + } + +- cmd.verbatim_args(&sess.opts.unstable_opts.pre_link_args); ++ cmd.verbatim_args(&(*sess.opts.unstable_opts.read_pre_link_args())); + } + + /// Add a link script embedded in the target, if applicable. +@@ -2428,7 +2428,7 @@ fn add_link_script(cmd: &mut dyn Linker, sess: &Session, tmpdir: &Path, crate_ty + /// Add arbitrary "user defined" args defined from command line. + /// FIXME: Determine where exactly these args need to be inserted. + fn add_user_defined_link_args(cmd: &mut dyn Linker, sess: &Session) { +- cmd.verbatim_args(&sess.opts.cg.link_args); ++ cmd.verbatim_args(&(*sess.opts.cg.read_link_args())); + } + + /// Add arbitrary "late link" args defined by the target spec. +@@ -2703,7 +2703,7 @@ fn add_library_search_dirs( + self_contained_components: LinkSelfContainedComponents, + apple_sdk_root: Option<&Path>, + ) { +- if !sess.opts.unstable_opts.link_native_libraries { ++ if !(*sess.opts.unstable_opts.read_link_native_libraries()) { + return; + } + +@@ -2743,7 +2743,7 @@ fn add_rpath_args( + // FIXME (#2397): At some point we want to rpath our guesses as to + // where extern libraries might live, based on the + // add_lib_search_paths +- if sess.opts.cg.rpath { ++ if (*sess.opts.cg.read_rpath()) { + let libs = crate_info + .used_crates + .iter() +@@ -3341,11 +3341,11 @@ fn add_order_independent_options( + ); + + // Pass debuginfo, NatVis debugger visualizers and strip flags down to the linker. +- cmd.debuginfo(sess.opts.cg.strip, &natvis_visualizers); ++ cmd.debuginfo((*sess.opts.cg.read_strip()), &natvis_visualizers); + + // We want to prevent the compiler from accidentally leaking in any system libraries, + // so by default we tell linkers not to link to any default libraries. +- if !sess.opts.cg.default_linker_libraries && sess.target.no_default_libraries { ++ if !(*sess.opts.cg.read_default_linker_libraries()) && sess.target.no_default_libraries { + cmd.no_default_libraries(); + } + +@@ -3409,7 +3409,7 @@ fn add_native_libs_from_crate( + link_dynamic: bool, + link_output_kind: LinkOutputKind, + ) { +- if !sess.opts.unstable_opts.link_native_libraries { ++ if !(*sess.opts.unstable_opts.read_link_native_libraries()) { + // If `-Zlink-native-libraries=false` is set, then the assumption is that an + // external build system already has the native dependencies defined, and it + // will provide them to the linker itself. +@@ -4126,11 +4126,11 @@ fn add_lld_args( + // the CLI, and what the target spec enables (as it can't disable components): + // - if the self-contained linker is enabled on the CLI or by the target spec, + // - and if the self-contained linker is not disabled on the CLI. +- let self_contained_cli = sess.opts.cg.link_self_contained.is_linker_enabled(); ++ let self_contained_cli = (*sess.opts.cg.read_link_self_contained()).is_linker_enabled(); + let self_contained_target = self_contained_components.is_linker_enabled(); + + let self_contained_linker = self_contained_cli || self_contained_target; +- if self_contained_linker && !sess.opts.cg.link_self_contained.is_linker_disabled() { ++ if self_contained_linker && !(*sess.opts.cg.read_link_self_contained()).is_linker_disabled() { + let mut linker_path_exists = false; + for path in sess.get_tools_search_paths(false) { + let linker_path = path.join("gcc-ld"); +diff --git a/compiler/rustc_codegen_ssa/src/back/linker.rs b/compiler/rustc_codegen_ssa/src/back/linker.rs +index ffb9130ed..9e37a7cb7 100644 +--- a/compiler/rustc_codegen_ssa/src/back/linker.rs ++++ b/compiler/rustc_codegen_ssa/src/back/linker.rs +@@ -73,7 +73,7 @@ pub(crate) fn get_linker<'a>( + | LinkerFlavor::WasmLld(Cc::No) + | LinkerFlavor::Msvc(Lld::Yes) => Command::lld(linker, flavor.lld_flavor()), + LinkerFlavor::Msvc(Lld::No) +- if sess.opts.cg.linker.is_none() && sess.target.linker.is_none() => ++ if (*sess.opts.cg.read_linker()).is_none() && sess.target.linker.is_none() => + { + Command::new(msvc_tool.as_ref().map_or(linker, |t| t.path())) + } +@@ -444,7 +444,7 @@ fn build_dylib(&mut self, crate_type: CrateType, out_filename: &Path) { + // purely to support bootstrap right now, we should get a more + // principled solution at some point to force the compiler to pass + // the right `-Wl,-install_name` with an `@rpath` in it. +- if self.sess.opts.cg.rpath || self.sess.opts.unstable_opts.osx_rpath_install_name { ++ if (*self.sess.opts.cg.read_rpath()) || self.sess.opts.unstable_opts.osx_rpath_install_name { + let mut rpath = OsString::from("@rpath/"); + rpath.push(out_filename.file_name().unwrap()); + self.link_arg("-install_name").link_arg(rpath); +diff --git a/compiler/rustc_codegen_ssa/src/back/write.rs b/compiler/rustc_codegen_ssa/src/back/write.rs +index 04be7956e..3d6cdf857 100644 +--- a/compiler/rustc_codegen_ssa/src/back/write.rs ++++ b/compiler/rustc_codegen_ssa/src/back/write.rs +@@ -126,7 +126,7 @@ macro_rules! if_regular { + let sess = tcx.sess; + let opt_level_and_size = if_regular!(Some(sess.opts.optimize), None); + +- let save_temps = sess.opts.cg.save_temps; ++ let save_temps = (*sess.opts.cg.read_save_temps()); + + let should_emit_obj = sess.opts.output_types.contains_key(&OutputType::Exe) + || match kind { +@@ -536,7 +536,7 @@ pub fn produce_final_output_artifacts( + } else { + copy_gracefully(&path, &output); + } +- if !sess.opts.cg.save_temps && !keep_numbered { ++ if !(*sess.opts.cg.read_save_temps()) && !keep_numbered { + // The user just wants `foo.x`, not `foo.#module-name#.x`. + ensure_removed(sess.dcx(), &path); + } +@@ -599,7 +599,7 @@ pub fn produce_final_output_artifacts( + // We may create additional files if requested by the user (through + // `-C save-temps` or `--emit=` flags). + +- if !sess.opts.cg.save_temps { ++ if !(*sess.opts.cg.read_save_temps()) { + // Remove the temporary .#module-name#.rcgu.o objects. If the user didn't + // explicitly request bitcode (with --emit=bc), and the bitcode is not + // needed for building an rlib, then we must remove .#module-name#.bc as +@@ -1242,7 +1242,7 @@ fn start_executing_work( + let opt_level = tcx.backend_optimization_level(()); + let tm_factory = backend.target_machine_factory(tcx.sess, opt_level); + +- let remark_dir = if let Some(ref dir) = sess.opts.unstable_opts.remark_dir { ++ let remark_dir = if let Some(ref dir) = (*sess.opts.unstable_opts.read_remark_dir()) { + let result = fs::create_dir_all(dir).and_then(|_| dir.canonicalize()); + match result { + Ok(dir) => Some(dir), +@@ -1258,12 +1258,12 @@ fn start_executing_work( + crate_types: tcx.crate_types().to_vec(), + lto: sess.lto(), + use_linker_plugin_lto: sess.opts.cg.linker_plugin_lto.enabled(), +- dylib_lto: sess.opts.unstable_opts.dylib_lto, ++ dylib_lto: (*sess.opts.unstable_opts.read_dylib_lto()), + prefer_dynamic: sess.opts.cg.prefer_dynamic, + fewer_names: sess.fewer_names(), +- save_temps: sess.opts.cg.save_temps, +- time_trace: sess.opts.unstable_opts.llvm_time_trace, +- remark: sess.opts.cg.remark.clone(), ++ save_temps: (*sess.opts.cg.read_save_temps()), ++ time_trace: (*sess.opts.unstable_opts.read_llvm_time_trace()), ++ remark: (*sess.opts.cg.read_remark()).clone(), + remark_dir, + old_incr_comp_session_dir: tcx + .incr_comp_session +diff --git a/compiler/rustc_codegen_ssa/src/base.rs b/compiler/rustc_codegen_ssa/src/base.rs +index f9dfd04fa..3d3e17231 100644 +--- a/compiler/rustc_codegen_ssa/src/base.rs ++++ b/compiler/rustc_codegen_ssa/src/base.rs +@@ -1,4 +1,5 @@ + use std::collections::BTreeSet; ++use std::path::PathBuf; + use std::sync::Arc; + use std::time::{Duration, Instant}; + use std::{cmp, iter}; +@@ -808,7 +809,7 @@ pub fn codegen_crate< + }); + + let mut total_codegen_time = Duration::new(0, 0); +- let start_rss = tcx.sess.opts.unstable_opts.time_passes.then(|| get_resident_set_size()); ++ let start_rss = (*tcx.sess.opts.unstable_opts.read_time_passes()).then(|| get_resident_set_size()); + + // The non-parallel compiler can only translate codegen units to LLVM IR + // on a single thread, leading to a staircase effect where the N LLVM +@@ -847,6 +848,7 @@ pub fn codegen_crate< + FxHashMap::default() + }; + ++ let verify_reuse = std::env::var_os("RUSTC_VERIFY_REUSE").is_some(); + for (i, cgu) in codegen_units.iter().enumerate() { + ongoing_codegen.wait_for_signal_to_codegen_item(); + ongoing_codegen.check_for_errors(tcx.sess); +@@ -868,9 +870,15 @@ pub fn codegen_crate< + // compilation hang on post-monomorphization errors. + tcx.dcx().abort_if_errors(); + ++ if verify_reuse { ++ record_codegen_unit(&backend, tcx, cgu.name(), &module); ++ } + submit_codegened_module_to_llvm(&ongoing_codegen.coordinator, module, cost); + } + CguReuse::PreLto => { ++ if verify_reuse { ++ verify_reused_codegen_unit(&backend, tcx, cgu.name(), bitcode_needed); ++ } + submit_pre_lto_module_to_llvm( + tcx, + &ongoing_codegen.coordinator, +@@ -885,6 +893,9 @@ pub fn codegen_crate< + tcx.dcx().abort_if_errors(); + } + CguReuse::PostLto => { ++ if verify_reuse { ++ verify_reused_codegen_unit(&backend, tcx, cgu.name(), bitcode_needed); ++ } + submit_post_lto_module_to_llvm( + &ongoing_codegen.coordinator, + CachedModuleCodegen { +@@ -900,7 +911,7 @@ pub fn codegen_crate< + + // Since the main thread is sometimes blocked during codegen, we keep track + // -Ztime-passes output manually. +- if tcx.sess.opts.unstable_opts.time_passes { ++ if (*tcx.sess.opts.unstable_opts.read_time_passes()) { + let end_rss = get_resident_set_size(); + + print_time_passes_entry( +@@ -908,7 +919,7 @@ pub fn codegen_crate< + total_codegen_time, + start_rss.unwrap(), + end_rss, +- tcx.sess.opts.unstable_opts.time_passes_format, ++ (*tcx.sess.opts.unstable_opts.read_time_passes_format()), + ); + } + +@@ -1253,6 +1264,61 @@ pub(crate) fn provide(providers: &mut Providers) { + }; + } + ++/// For testing incremental compilation (`RUSTC_VERIFY_REUSE`): where a codegen unit's ++/// unoptimized code is kept, from the session that generated it, to check the unit when a ++/// later session reuses it. In the crate's incremental directory, outside any one session's. ++fn recorded_codegen_unit(tcx: TyCtxt<'_>, cgu_name: Symbol) -> Option { ++ let session: &std::path::Path = &tcx.incr_comp_session?.new_session_directory; ++ Some(session.parent()?.join("verify-reuse").join(format!("{cgu_name}.ll"))) ++} ++ ++fn record_codegen_unit( ++ backend: &B, ++ tcx: TyCtxt<'_>, ++ cgu_name: Symbol, ++ module: &ModuleCodegen, ++) { ++ let (Some(path), Some(code)) = ++ (recorded_codegen_unit(tcx, cgu_name), backend.unoptimized_code(module)) ++ else { ++ return; ++ }; ++ if let Some(dir) = path.parent() { ++ let _ = std::fs::create_dir_all(dir); ++ } ++ let _ = std::fs::write(path, code); ++} ++ ++/// Generates a reused codegen unit again and compares its unoptimized code with that of the ++/// session that generated the reused one. A difference prints a line starting ++/// `rustc-verify-reuse:` and keeps the fresh code next to the recorded one. ++fn verify_reused_codegen_unit( ++ backend: &B, ++ tcx: TyCtxt<'_>, ++ cgu_name: Symbol, ++ bitcode_needed: bool, ++) { ++ let Some(path) = recorded_codegen_unit(tcx, cgu_name) else { return }; ++ let Ok(recorded) = std::fs::read(&path) else { return }; ++ let Some(module) = backend.codegen_unit_again(tcx, cgu_name, bitcode_needed) else { return }; ++ let Some(fresh) = backend.unoptimized_code(&module) else { return }; ++ if std::env::var_os("RUSTC_VERIFY_REUSE").is_some_and(|v| v == "verbose") { ++ eprintln!("rustc-verify-reuse-checked: codegen unit `{cgu_name}`"); ++ } ++ if fresh != recorded { ++ let kept = path.with_extension("fresh.ll"); ++ let _ = std::fs::write(&kept, &fresh); ++ eprintln!( ++ "rustc-verify-reuse: codegen unit `{cgu_name}` of `{}` reused from the incremental cache differs from a fresh codegen (unoptimized code, {} and {} bytes); kept as {} next to {}", ++ tcx.crate_name(LOCAL_CRATE), ++ recorded.len(), ++ fresh.len(), ++ kept.display(), ++ path.display(), ++ ); ++ } ++} ++ + pub fn determine_cgu_reuse<'tcx>(tcx: TyCtxt<'tcx>, cgu: &CodegenUnit<'tcx>) -> CguReuse { + if !tcx.dep_graph.is_fully_enabled() + || tcx.sess.opts.unstable_opts.disable_incr_comp_backend_caching +diff --git a/compiler/rustc_codegen_ssa/src/traits/backend.rs b/compiler/rustc_codegen_ssa/src/traits/backend.rs +index e06ac36fd..1fea1d7c8 100644 +--- a/compiler/rustc_codegen_ssa/src/traits/backend.rs ++++ b/compiler/rustc_codegen_ssa/src/traits/backend.rs +@@ -167,4 +167,23 @@ fn compile_codegen_unit( + cgu_name: Symbol, + bitcode_needed: bool, + ) -> (ModuleCodegen, u64); ++ ++ /// For testing incremental compilation (`RUSTC_VERIFY_REUSE`): the codegen unit generated ++ /// again, outside dependency tracking, as a check of one reused from the incremental cache. ++ /// `None` if the backend cannot. ++ fn codegen_unit_again( ++ &self, ++ _tcx: TyCtxt<'_>, ++ _cgu_name: Symbol, ++ _bitcode_needed: bool, ++ ) -> Option> { ++ None ++ } ++ ++ /// For testing incremental compilation (`RUSTC_VERIFY_REUSE`): a generated module's code ++ /// before optimization, as bytes that are equal when the code is. `None` if the backend ++ /// cannot say. ++ fn unoptimized_code(&self, _module: &ModuleCodegen) -> Option> { ++ None ++ } + } +diff --git a/compiler/rustc_data_structures/src/lib.rs b/compiler/rustc_data_structures/src/lib.rs +index aae378939..dc71d2ae3 100644 +--- a/compiler/rustc_data_structures/src/lib.rs ++++ b/compiler/rustc_data_structures/src/lib.rs +@@ -84,6 +84,7 @@ + pub mod unhash; + pub mod union_find; + pub mod unord; ++pub mod untracked; + pub mod vec_cache; + + mod atomic_ref; +diff --git a/compiler/rustc_data_structures/src/untracked.rs b/compiler/rustc_data_structures/src/untracked.rs +new file mode 100644 +index 000000000..1ee27a720 +--- /dev/null ++++ b/compiler/rustc_data_structures/src/untracked.rs +@@ -0,0 +1,23 @@ ++//! For testing incremental compilation: reports of reads of state that the dependency graph ++//! does not track, such as the contents of source files. ++//! ++//! Code that reads such state calls [`untracked_read`] with what it reads. When enabled ++//! (`RUSTC_REPORT_UNTRACKED`, by `rustc_middle`), a read inside a computation whose result ++//! incremental compilation may reuse, and which has not declared that state as an input it ++//! tracks another way, is reported: the result may be reused after the state changed. ++ ++use std::panic::Location; ++use std::sync::OnceLock; ++ ++/// The function called on each read, set once when reporting is enabled. ++pub static UNTRACKED_READ_HOOK: OnceLock)> = ++ OnceLock::new(); ++ ++/// Notes a read of `what`, state the dependency graph does not track. ++#[inline] ++#[track_caller] ++pub fn untracked_read(what: &'static str) { ++ if let Some(hook) = UNTRACKED_READ_HOOK.get() { ++ hook(what, Location::caller()); ++ } ++} +diff --git a/compiler/rustc_driver_impl/src/lib.rs b/compiler/rustc_driver_impl/src/lib.rs +index 1ba7ce953..3bd2fdb0f 100644 +--- a/compiler/rustc_driver_impl/src/lib.rs ++++ b/compiler/rustc_driver_impl/src/lib.rs +@@ -162,8 +162,8 @@ fn config(&mut self, config: &mut interface::Config) { + // If a --print=... option has been given, we don't print the "total" + // time because it will mess up the --print output. See #64339. + // +- self.time_passes = (config.opts.prints.is_empty() && config.opts.unstable_opts.time_passes) +- .then_some(config.opts.unstable_opts.time_passes_format); ++ self.time_passes = (config.opts.prints.is_empty() && (*config.opts.unstable_opts.read_time_passes())) ++ .then_some((*config.opts.unstable_opts.read_time_passes_format())); + config.opts.trimmed_def_paths = true; + } + } +@@ -263,7 +263,7 @@ pub fn compiler_entrypoint(at_args: &[String], callbacks: &mut (dyn Callbacks + + sess.dcx().fatal("no input filename given"); // this is fatal + } + +- if !sess.opts.unstable_opts.ls.is_empty() { ++ if !(*sess.opts.unstable_opts.read_ls()).is_empty() { + list_metadata(sess, &*codegen_backend.metadata_loader()); + return; + } +@@ -296,7 +296,7 @@ pub fn compiler_entrypoint(at_args: &[String], callbacks: &mut (dyn Callbacks + + return; + } + +- if sess.opts.unstable_opts.parse_crate_root_only { ++ if (*sess.opts.unstable_opts.read_parse_crate_root_only()) { + return; + } + +@@ -318,7 +318,7 @@ pub fn compiler_entrypoint(at_args: &[String], callbacks: &mut (dyn Callbacks + + return None; + } + +- if sess.opts.unstable_opts.no_analysis { ++ if (*sess.opts.unstable_opts.read_no_analysis()) { + return None; + } + +@@ -336,7 +336,7 @@ pub fn compiler_entrypoint(at_args: &[String], callbacks: &mut (dyn Callbacks + + + let linker = Linker::codegen_and_build_linker(tcx, codegen_backend); + +- if let Some(metrics_dir) = &sess.opts.unstable_opts.metrics_dir { ++ if let Some(metrics_dir) = &(*sess.opts.unstable_opts.read_metrics_dir()) { + dump_feature_usage_metrics(tcx, metrics_dir); + } + +@@ -602,7 +602,7 @@ fn list_metadata(sess: &Session, metadata_loader: &dyn MetadataLoader) { + path, + metadata_loader, + &mut v, +- &sess.opts.unstable_opts.ls, ++ &(*sess.opts.unstable_opts.read_ls()), + sess.cfg_version, + ) { + if path.extension().is_some_and(|extension| extension == "rs") { +diff --git a/compiler/rustc_expand/src/base.rs b/compiler/rustc_expand/src/base.rs +index 01e8cef49..b72b27d7d 100644 +--- a/compiler/rustc_expand/src/base.rs ++++ b/compiler/rustc_expand/src/base.rs +@@ -896,7 +896,7 @@ pub fn new( + None => (None, helper_attrs), + }; + let diagnostic_opaque = builtin_name.is_some() +- || (!sess.opts.unstable_opts.macro_backtrace && find_attr!(attrs, Opaque)); ++ || (!(*sess.opts.unstable_opts.read_macro_backtrace()) && find_attr!(attrs, Opaque)); + + let stability = find_attr!(attrs, Stability { stability, .. } => *stability); + +diff --git a/compiler/rustc_expand/src/expand.rs b/compiler/rustc_expand/src/expand.rs +index 5a2442fbc..3ec74c962 100644 +--- a/compiler/rustc_expand/src/expand.rs ++++ b/compiler/rustc_expand/src/expand.rs +@@ -719,7 +719,7 @@ fn expand_invoc( + return ExpandResult::Ready(invoc.fragment_kind.dummy(invoc.span(), guar)); + } + +- let macro_stats = self.cx.sess.opts.unstable_opts.macro_stats; ++ let macro_stats = (*self.cx.sess.opts.unstable_opts.read_macro_stats()); + + let (fragment_kind, span) = (invoc.fragment_kind, invoc.span()); + ExpandResult::Ready(match invoc.kind { +diff --git a/compiler/rustc_hir_typeck/src/upvar.rs b/compiler/rustc_hir_typeck/src/upvar.rs +index 98155d9a8..e729e0537 100644 +--- a/compiler/rustc_hir_typeck/src/upvar.rs ++++ b/compiler/rustc_hir_typeck/src/upvar.rs +@@ -629,7 +629,7 @@ fn analyze_closure( + + self.typeck_results.borrow_mut().closure_fake_reads.insert(closure_def_id, fake_reads); + +- if self.tcx.sess.opts.unstable_opts.profile_closures { ++ if (*self.tcx.sess.opts.unstable_opts.read_profile_closures()) { + self.typeck_results.borrow_mut().closure_size_eval.insert( + closure_def_id, + ClosureSizeProfileData { +diff --git a/compiler/rustc_incremental/src/assert_dep_graph.rs b/compiler/rustc_incremental/src/assert_dep_graph.rs +index 8147e3c3a..12e13a0b4 100644 +--- a/compiler/rustc_incremental/src/assert_dep_graph.rs ++++ b/compiler/rustc_incremental/src/assert_dep_graph.rs +@@ -59,13 +59,13 @@ pub(crate) fn assert_dep_graph(tcx: TyCtxt<'_>) { + // checks below, rather than locking and cloning it separately for each. + let retained_dep_graph = tcx.dep_graph.retained_dep_graph(); + +- if tcx.sess.opts.unstable_opts.dump_dep_graph { ++ if (*tcx.sess.opts.unstable_opts.read_dump_dep_graph()) { + if let Some(graph) = &retained_dep_graph { + dump_graph(graph); + } + } + +- if !tcx.sess.opts.unstable_opts.query_dep_graph { ++ if !(*tcx.sess.opts.unstable_opts.read_query_dep_graph()) { + return; + } + +@@ -87,7 +87,7 @@ pub(crate) fn assert_dep_graph(tcx: TyCtxt<'_>) { + + if !if_this_changed.is_empty() || !then_this_would_need.is_empty() { + assert!( +- tcx.sess.opts.unstable_opts.query_dep_graph, ++ (*tcx.sess.opts.unstable_opts.read_query_dep_graph()), + "cannot use the `#[{}]` or `#[{}]` annotations \ + without supplying `-Z query-dep-graph`", + sym::rustc_if_this_changed, +diff --git a/compiler/rustc_incremental/src/persist/clean.rs b/compiler/rustc_incremental/src/persist/clean.rs +index b6402181f..ed1ca5be1 100644 +--- a/compiler/rustc_incremental/src/persist/clean.rs ++++ b/compiler/rustc_incremental/src/persist/clean.rs +@@ -125,7 +125,7 @@ struct Assertion { + } + + pub(crate) fn check_clean_annotations(tcx: TyCtxt<'_>) { +- if !tcx.sess.opts.unstable_opts.query_dep_graph { ++ if !(*tcx.sess.opts.unstable_opts.read_query_dep_graph()) { + return; + } + +diff --git a/compiler/rustc_incremental/src/persist/file_format.rs b/compiler/rustc_incremental/src/persist/file_format.rs +index 853a5c9ba..c6d38410f 100644 +--- a/compiler/rustc_incremental/src/persist/file_format.rs ++++ b/compiler/rustc_incremental/src/persist/file_format.rs +@@ -175,7 +175,7 @@ pub(crate) fn open_incremental_file( + fn report_format_mismatch(sess: &Session, file: &Path, message: &str) { + debug!("read_file: {}", message); + +- if sess.opts.unstable_opts.incremental_info { ++ if (*sess.opts.unstable_opts.read_incremental_info()) { + eprintln!( + "[incremental] ignoring cache artifact `{}`: {}", + file.file_name().unwrap().to_string_lossy(), +diff --git a/compiler/rustc_incremental/src/persist/load.rs b/compiler/rustc_incremental/src/persist/load.rs +index 76e0bf92c..0caeb2789 100644 +--- a/compiler/rustc_incremental/src/persist/load.rs ++++ b/compiler/rustc_incremental/src/persist/load.rs +@@ -64,7 +64,7 @@ fn load_dep_graph(sess: &Session, incr_comp_session: &IncrCompSession) -> LoadRe + for swp in work_products { + let all_files_exist = swp.work_product.saved_files.items().all(|(_, path)| { + let exists = in_old_incr_comp_dir_sess(incr_comp_session, path).unwrap().exists(); +- if !exists && sess.opts.unstable_opts.incremental_info { ++ if !exists && (*sess.opts.unstable_opts.read_incremental_info()) { + eprintln!("incremental: could not find file for work product: {path}",); + } + exists +@@ -93,7 +93,7 @@ fn load_dep_graph(sess: &Session, incr_comp_session: &IncrCompSession) -> LoadRe + let prev_commandline_args_hash = Hash64::decode(&mut decoder); + + if prev_commandline_args_hash != expected_hash { +- if sess.opts.unstable_opts.incremental_info { ++ if (*sess.opts.unstable_opts.read_incremental_info()) { + eprintln!( + "[incremental] completely ignoring cache because of \ + differing commandline arguments" +@@ -150,7 +150,7 @@ pub fn load_query_result_cache( + /// the outcome of trying to load previous-session state. + fn maybe_assert_incr_state(sess: &Session, load_result: &LoadResult) { + // Return immediately if there's nothing to assert. +- let Some(assertion) = sess.opts.unstable_opts.assert_incr_state else { return }; ++ let Some(assertion) = (*sess.opts.unstable_opts.read_assert_incr_state()) else { return }; + + // Match exhaustively to make sure we don't miss any cases. + let loaded = match load_result { +diff --git a/compiler/rustc_incremental/src/persist/save.rs b/compiler/rustc_incremental/src/persist/save.rs +index 46f47d6c8..69ea0af7a 100644 +--- a/compiler/rustc_incremental/src/persist/save.rs ++++ b/compiler/rustc_incremental/src/persist/save.rs +@@ -41,6 +41,7 @@ pub(crate) fn save_dep_graph(tcx: TyCtxt<'_>) { + + sess.time("assert_dep_graph", || assert_dep_graph(tcx)); + sess.time("check_clean", || clean::check_clean_annotations(tcx)); ++ sess.time("verify_reused_values", || tcx.verify_reused_values()); + + par_join( + move || { +diff --git a/compiler/rustc_infer/src/infer/relate/higher_ranked.rs b/compiler/rustc_infer/src/infer/relate/higher_ranked.rs +index 324725a07..baa0f2b04 100644 +--- a/compiler/rustc_infer/src/infer/relate/higher_ranked.rs ++++ b/compiler/rustc_infer/src/infer/relate/higher_ranked.rs +@@ -92,7 +92,7 @@ pub fn leak_check( + // subtyping errors that it would have caught will now be + // caught later on, during region checking. However, we + // continue to use it for a transition period. +- if self.tcx.sess.opts.unstable_opts.no_leak_check || self.skip_leak_check { ++ if (*self.tcx.sess.opts.unstable_opts.read_no_leak_check()) || self.skip_leak_check { + return Ok(()); + } + +diff --git a/compiler/rustc_interface/src/interface.rs b/compiler/rustc_interface/src/interface.rs +index 7a6a49f37..3df272fa9 100644 +--- a/compiler/rustc_interface/src/interface.rs ++++ b/compiler/rustc_interface/src/interface.rs +@@ -398,7 +398,7 @@ pub fn run_compiler(config: Config, f: impl FnOnce(&Compiler) -> R + Se + &early_dcx, + &config.opts.target_triple, + config.opts.sysroot.path(), +- config.opts.unstable_opts.unstable_options, ++ (*config.opts.unstable_opts.read_unstable_options()), + ); + let file_loader = config.file_loader.unwrap_or_else(|| Box::new(RealFileLoader)); + let path_mapping = config.opts.file_path_mapping(); +@@ -416,7 +416,7 @@ pub fn run_compiler(config: Config, f: impl FnOnce(&Compiler) -> R + Se + // impl `Send`. Creating a new one is fine. + let early_dcx = EarlyDiagCtxt::new(config.opts.error_format); + +- let temps_dir = config.opts.unstable_opts.temps_dir.as_deref().map(PathBuf::from); ++ let temps_dir = (*config.opts.unstable_opts.read_temps_dir()).as_deref().map(PathBuf::from); + + let early_sess = + rustc_session::build_early_session(config.opts, target, config.ice_file); +diff --git a/compiler/rustc_interface/src/passes.rs b/compiler/rustc_interface/src/passes.rs +index 8975f01de..9245d5d45 100644 +--- a/compiler/rustc_interface/src/passes.rs ++++ b/compiler/rustc_interface/src/passes.rs +@@ -204,10 +204,10 @@ fn configure_and_expand( + crate_name, + features, + recursion_limit, +- trace_mac: sess.opts.unstable_opts.trace_macros, ++ trace_mac: (*sess.opts.unstable_opts.read_trace_macros()), + should_test: sess.is_test_crate(), +- span_debug: sess.opts.unstable_opts.span_debug, +- proc_macro_backtrace: sess.opts.unstable_opts.proc_macro_backtrace, ++ span_debug: (*sess.opts.unstable_opts.read_span_debug()), ++ proc_macro_backtrace: (*sess.opts.unstable_opts.read_proc_macro_backtrace()), + }; + + let lint_store = LintStoreExpandImpl(lint_store); +@@ -243,7 +243,7 @@ fn configure_and_expand( + } + } + +- if ecx.sess.opts.unstable_opts.macro_stats { ++ if (*ecx.sess.opts.unstable_opts.read_macro_stats()) { + print_macro_stats(&ecx); + } + +@@ -413,7 +413,7 @@ fn early_lint_checks(tcx: TyCtxt<'_>, (): ()) { + let krate = &*krate.borrow(); + let mut lint_buffer = resolver.lint_buffer.steal(); + +- if sess.opts.unstable_opts.input_stats { ++ if (*sess.opts.unstable_opts.read_input_stats()) { + input_stats::print_ast_stats(tcx, krate); + } + +@@ -933,6 +933,7 @@ pub fn create_and_enter_global_ctxt FnOnce(TyCtxt<'tcx>) -> T>( + f: F, + ) -> (T, Option) { + let sess = &compiler.sess; ++ rustc_middle::dep_graph::enable_untracked_read_reports(); + + let pre_configured_attrs = rustc_expand::config::pre_configure_attrs(sess, &krate.attrs); + +@@ -1091,7 +1092,7 @@ pub fn emit_delayed_lints(tcx: TyCtxt<'_>) { + /// Runs all analyses that we guarantee to run, even if errors were reported in earlier analyses. + /// This function never fails. + fn run_required_analyses(tcx: TyCtxt<'_>) { +- if tcx.sess.opts.unstable_opts.input_stats { ++ if (*tcx.sess.opts.unstable_opts.read_input_stats()) { + rustc_passes::input_stats::print_hir_stats(tcx); + } + // When using rustdoc's "jump to def" feature, it enters this code and `check_crate` +@@ -1272,7 +1273,7 @@ fn analysis(tcx: TyCtxt<'_>, (): ()) { + // that requires the optimized/ctfe MIR, coroutine bodies, or evaluating consts. + // Nevertheless, wait after type checking is finished, as optimizing code that does not + // type-check is very prone to ICEs. +- if tcx.sess.opts.unstable_opts.validate_mir { ++ if (*tcx.sess.opts.unstable_opts.read_validate_mir()) { + sess.time("ensuring_final_MIR_is_computable", || { + tcx.par_hir_body_owners(|def_id| { + if !tcx.is_trivial_const(def_id) { +@@ -1348,7 +1349,7 @@ pub(crate) fn start_codegen<'tcx>( + + // This must run after monomorphization so that all generic types + // have been instantiated. +- if tcx.sess.opts.unstable_opts.print_type_sizes { ++ if (*tcx.sess.opts.unstable_opts.read_print_type_sizes()) { + tcx.sess.code_stats.print_type_sizes(); + } + +@@ -1447,7 +1448,7 @@ pub fn collect_crate_types( + } + + // Shadow `sdylib` crate type in interface build. +- if session.opts.unstable_opts.build_sdylib_interface { ++ if (*session.opts.unstable_opts.read_build_sdylib_interface()) { + return vec![CrateType::Rlib]; + } + +diff --git a/compiler/rustc_interface/src/queries.rs b/compiler/rustc_interface/src/queries.rs +index 759297bc6..60516087d 100644 +--- a/compiler/rustc_interface/src/queries.rs ++++ b/compiler/rustc_interface/src/queries.rs +@@ -67,7 +67,7 @@ pub fn link( + } + }); + +- if sess.codegen_units().as_usize() == 1 && sess.opts.unstable_opts.time_llvm_passes { ++ if sess.codegen_units().as_usize() == 1 && (*sess.opts.unstable_opts.read_time_llvm_passes()) { + codegen_backend.print_pass_timings() + } + +diff --git a/compiler/rustc_interface/src/tests.rs b/compiler/rustc_interface/src/tests.rs +index 34ac92fae..f0a43672b 100644 +--- a/compiler/rustc_interface/src/tests.rs ++++ b/compiler/rustc_interface/src/tests.rs +@@ -47,7 +47,7 @@ fn sess_and_cfg(args: &[&'static str], f: F) + &early_dcx, + &sessopts.target_triple, + sessopts.sysroot.path(), +- sessopts.unstable_opts.unstable_options, ++ (*sessopts.unstable_opts.read_unstable_options()), + ); + let hash_kind = sessopts.unstable_opts.src_hash_algorithm(&target); + let checksum_hash_kind = sessopts.unstable_opts.checksum_hash_algorithm(); +@@ -59,7 +59,7 @@ fn sess_and_cfg(args: &[&'static str], f: F) + }); + + rustc_span::create_session_globals_then(DEFAULT_EDITION, &[], sm_inputs, || { +- let temps_dir = sessopts.unstable_opts.temps_dir.as_deref().map(PathBuf::from); ++ let temps_dir = (*sessopts.unstable_opts.read_temps_dir()).as_deref().map(PathBuf::from); + let io = CompilerIO { + input: Input::Str { name: FileName::Custom(String::new()), input: String::new() }, + output_dir: None, +diff --git a/compiler/rustc_interface/src/util.rs b/compiler/rustc_interface/src/util.rs +index d58b7bfdd..6e8f6570c 100644 +--- a/compiler/rustc_interface/src/util.rs ++++ b/compiler/rustc_interface/src/util.rs +@@ -675,7 +675,7 @@ pub fn build_output_filenames(attrs: &[ast::Attribute], sess: &Session) -> Outpu + sess.io.temps_dir.clone(), + invocation_temp, + sess.opts.unstable_opts.split_dwarf_out_dir.clone(), +- sess.opts.cg.extra_filename.clone(), ++ (*sess.opts.cg.read_extra_filename()).clone(), + sess.opts.output_types.clone(), + ) + } +@@ -687,7 +687,7 @@ pub fn build_output_filenames(attrs: &[ast::Attribute], sess: &Session) -> Outpu + sess.dcx().emit_warn(diagnostics::MultipleOutputTypesAdaption); + None + } else { +- if !sess.opts.cg.extra_filename.is_empty() { ++ if !(*sess.opts.cg.read_extra_filename()).is_empty() { + sess.dcx().emit_warn(diagnostics::IgnoringExtraFilename); + } + Some(out_file.clone()) +@@ -706,7 +706,7 @@ pub fn build_output_filenames(attrs: &[ast::Attribute], sess: &Session) -> Outpu + sess.io.temps_dir.clone(), + invocation_temp, + sess.opts.unstable_opts.split_dwarf_out_dir.clone(), +- sess.opts.cg.extra_filename.clone(), ++ (*sess.opts.cg.read_extra_filename()).clone(), + sess.opts.output_types.clone(), + ) + } +diff --git a/compiler/rustc_metadata/src/creader.rs b/compiler/rustc_metadata/src/creader.rs +index 10693448e..70fa9ba41 100644 +--- a/compiler/rustc_metadata/src/creader.rs ++++ b/compiler/rustc_metadata/src/creader.rs +@@ -299,23 +299,33 @@ pub(crate) fn crate_dependencies_in_postorder(&self, cnum: CrateNum) -> IndexSet + deps + } + ++ #[track_caller] + pub(crate) fn injected_panic_runtime(&self) -> Option { ++ rustc_data_structures::untracked::untracked_read("crate store"); + self.injected_panic_runtime + } + ++ #[track_caller] + pub(crate) fn allocator_kind(&self) -> Option { ++ rustc_data_structures::untracked::untracked_read("crate store"); + self.allocator_kind + } + ++ #[track_caller] + pub(crate) fn alloc_error_handler_kind(&self) -> Option { ++ rustc_data_structures::untracked::untracked_read("crate store"); + self.alloc_error_handler_kind + } + ++ #[track_caller] + pub(crate) fn has_global_allocator(&self) -> bool { ++ rustc_data_structures::untracked::untracked_read("crate store"); + self.has_global_allocator + } + ++ #[track_caller] + pub(crate) fn has_alloc_error_handler(&self) -> bool { ++ rustc_data_structures::untracked::untracked_read("crate store"); + self.has_alloc_error_handler + } + +@@ -348,7 +358,7 @@ fn report_target_modifiers_extended( + dep_mods: &TargetModifiers, + data: &CrateMetadata, + ) { +- let allowed_flag_mismatches = &tcx.sess.opts.cg.unsafe_allow_abi_mismatch; ++ let allowed_flag_mismatches = &(*tcx.sess.opts.cg.read_unsafe_allow_abi_mismatch()); + let local_crate = tcx.crate_name(LOCAL_CRATE); + let tmod_extender = |tmod: &TargetModifier| (tmod.extend(), tmod.clone()); + let report_diff = |prefix: &String, +@@ -454,7 +464,7 @@ pub fn report_session_incompatibilities(&self, tcx: TyCtxt<'_>, krate: &Crate) { + } + + pub fn report_incompatible_target_modifiers(&self, tcx: TyCtxt<'_>) { +- for flag_name in &tcx.sess.opts.cg.unsafe_allow_abi_mismatch { ++ for flag_name in &(*tcx.sess.opts.cg.read_unsafe_allow_abi_mismatch()) { + if !OptionsTargetModifiers::is_target_modifier(flag_name) { + tcx.dcx().emit_err(diagnostics::UnknownTargetModifierUnsafeAllowed { + flag_name: flag_name.clone(), +diff --git a/compiler/rustc_metadata/src/fs.rs b/compiler/rustc_metadata/src/fs.rs +index ed177708f..fb8b0e57c 100644 +--- a/compiler/rustc_metadata/src/fs.rs ++++ b/compiler/rustc_metadata/src/fs.rs +@@ -46,7 +46,7 @@ pub fn encode_and_write_metadata(tcx: TyCtxt<'_>) -> Result, LocalCrate: LocalCrate) -> Vec + // and enforce pointer authentication constraints. + if tcx.sess.target.cfg_abi == CfgAbi::Pauthtest { + if let NativeLibKind::Static { .. } = lib.kind { +- if !tcx.sess.opts.unstable_opts.ui_testing { ++ if !(*tcx.sess.opts.unstable_opts.read_ui_testing()) { + let diag = if lib.foreign_module.is_none() { + diagnostics::StaticLinkingNotSupported::UserRequested { + lib_name: lib.name, +diff --git a/compiler/rustc_metadata/src/rmeta/encoder.rs b/compiler/rustc_metadata/src/rmeta/encoder.rs +index 03493d0e1..6c65d6650 100644 +--- a/compiler/rustc_metadata/src/rmeta/encoder.rs ++++ b/compiler/rustc_metadata/src/rmeta/encoder.rs +@@ -15,6 +15,7 @@ + use rustc_data_structures::owned_slice::slice_owned; + use rustc_data_structures::stable_hash::{StableHash, StableHasher}; + use rustc_data_structures::sync::{par_for_each_in, par_join}; ++use rustc_data_structures::svh::Svh; + use rustc_data_structures::temp_dir::MaybeTempDir; + use rustc_data_structures::thousands::usize_with_underscores; + use rustc_hir as hir; +@@ -52,6 +53,9 @@ + pub(super) struct EncodeContext<'a, 'tcx> { + opaque: FileEncoder<'a>, + metadata_hasher: Arc>, ++ /// Set for a shadow encoding (see `verify_reused_metadata`), which keeps the crate hash it ++ /// computes here instead of setting the session's, which the reused metadata already set. ++ shadow_hash: Option>, + tcx: TyCtxt<'tcx>, + feat: &'tcx rustc_feature::Features, + tables: TableBuilders, +@@ -585,6 +589,8 @@ fn encode_source_map( + // + // At this point we also erase the actual on-disk path and only keep + // the remapped version -- as is necessary for reproducible builds. ++ // Its length, line table and content hash are read from the session. ++ rustc_data_structures::untracked::untracked_read("source file contents"); + let mut adapted_source_file = (**source_file).clone(); + + match source_file.name { +@@ -817,7 +823,11 @@ macro_rules! stat { + } else { + tcx.crate_hash(LOCAL_CRATE) + }; +- tcx.untracked().local_crate_hash.set(hash).expect("local_crate_hash set twice"); ++ if let Some(shadow_hash) = &mut self.shadow_hash { ++ *shadow_hash = Some(hash); ++ } else { ++ tcx.untracked().local_crate_hash.set(hash).expect("local_crate_hash set twice"); ++ } + + let unhashed = stat!("final", || { + // Indexed by dependency `CrateNum`, matching the numbering `encode_crate_deps` uses. +@@ -831,7 +841,7 @@ macro_rules! stat { + } + + self.lazy(CrateRootUnhashed { +- extra_filename: self.tcx.sess.opts.cg.extra_filename.clone(), ++ extra_filename: (*self.tcx.sess.opts.cg.read_extra_filename()).clone(), + dep_extra_filenames, + }) + }); +@@ -841,7 +851,7 @@ macro_rules! stat { + let computed_total_bytes: usize = stats.iter().map(|(_, size)| size).sum(); + assert_eq!(total_bytes, computed_total_bytes); + +- if tcx.sess.opts.unstable_opts.meta_stats { ++ if (*tcx.sess.opts.unstable_opts.read_meta_stats()) { + use std::fmt::Write; + + self.opaque.flush(); +@@ -2561,6 +2571,10 @@ pub fn encode_metadata(tcx: TyCtxt<'_>, path: &Path, ref_path: Option<&Path>) { + let hash = blob.expect("file already created").get_crate_hash(); + tcx.untracked().local_crate_hash.set(hash).expect("local_crate_hash set twice"); + ++ if std::env::var_os("RUSTC_VERIFY_REUSE").is_some() { ++ verify_reused_metadata(tcx, path); ++ } ++ + // Generate the metadata stub manually, as that is a small file compared to full metadata. + if let Some(ref_path) = ref_path { + let _prof_timer = tcx.prof.verbose_generic_activity("generate_crate_metadata_stub"); +@@ -2598,6 +2612,9 @@ pub fn encode_metadata(tcx: TyCtxt<'_>, path: &Path, ref_path: Option<&Path>) { + dep_node, + tcx, + || { ++ // Make the metadata depend on the source files it describes. ++ let _ = tcx.local_source_files_fingerprint(()); ++ rustc_middle::dep_graph::declare_untracked_input("source file contents"); + with_encode_metadata_header(tcx, path, |ecx| { + // Encode all the entries and extra information in the crate, + // culminating in the `CrateRoot` which points to all of it. +@@ -2638,6 +2655,15 @@ fn with_encode_metadata_header( + tcx: TyCtxt<'_>, + path: &Path, + f: impl FnOnce(&mut EncodeContext<'_, '_>) -> (usize, usize), ++) { ++ with_encode_metadata_header_shadow(tcx, path, false, f) ++} ++ ++fn with_encode_metadata_header_shadow( ++ tcx: TyCtxt<'_>, ++ path: &Path, ++ shadow: bool, ++ f: impl FnOnce(&mut EncodeContext<'_, '_>) -> (usize, usize), + ) { + // By default the crate hash (SVH) is computed from the bytes of the encoded metadata, + // Under `-Z metadata-crate-hash=no` the SVH comes from the legacy `crate_hash` query instead and +@@ -2694,6 +2720,7 @@ fn with_encode_metadata_header( + let mut ecx = EncodeContext { + opaque: encoder, + metadata_hasher: Arc::clone(&metadata_hasher), ++ shadow_hash: shadow.then_some(None), + tcx, + feat: tcx.features(), + tables: Default::default(), +@@ -2733,12 +2760,15 @@ fn with_encode_metadata_header( + tcx.dcx().emit_fatal(FailWriteFile { path: ecx.opaque.path(), err }); + } + +- let hash = tcx +- .untracked() +- .local_crate_hash +- .get() +- .copied() +- .expect("local_crate_hash set during encoding"); ++ let hash = match ecx.shadow_hash { ++ Some(shadow_hash) => shadow_hash.expect("crate hash computed during the shadow encoding"), ++ None => tcx ++ .untracked() ++ .local_crate_hash ++ .get() ++ .copied() ++ .expect("local_crate_hash set during encoding"), ++ }; + if let Err(err) = encode_crate_hash(file, hash) { + tcx.dcx().emit_fatal(FailWriteFile { path: ecx.opaque.path(), err }); + } +@@ -2748,6 +2778,50 @@ fn with_encode_metadata_header( + } + } + ++/// Shadow verification, for testing incremental compilation: metadata was just reused from the ++/// incremental cache because its dep-node is green. Encode it again from this session's ++/// queries, outside dependency tracking, and check that the bytes are the same. A difference ++/// means the reuse was wrong: something the metadata depends on is not tracked. ++/// ++/// On a difference this prints a line starting `rustc-verify-reuse:` and keeps the fresh ++/// encoding beside the reused file, as `.fresh`; otherwise it removes it. ++fn verify_reused_metadata(tcx: TyCtxt<'_>, path: &Path) { ++ let fresh = path.with_extension("rmeta.fresh"); ++ // `encode_metadata` already runs with dependency tracking ignored. ++ with_encode_metadata_header_shadow(tcx, &fresh, true, |ecx| { ++ let (root, unhashed) = ecx.encode_crate_root(); ++ ecx.opaque.flush(); ++ (root.position.get(), unhashed.position.get()) ++ }); ++ let reused = std::fs::read(path).unwrap_or_default(); ++ let encoded = std::fs::read(&fresh).unwrap_or_default(); ++ if std::env::var_os("RUSTC_VERIFY_REUSE").is_some_and(|v| v == "verbose") { ++ eprintln!("rustc-verify-reuse-checked: metadata of `{}`", tcx.crate_name(LOCAL_CRATE)); ++ } ++ if reused == encoded { ++ let _ = std::fs::remove_file(&fresh); ++ } else { ++ // Keep it next to the output, outside the temporary directory. ++ let fresh = match tcx.output_filenames(()).path(rustc_session::config::OutputType::Metadata) { ++ rustc_session::config::OutFileName::Real(out) ++ if std::fs::rename(&fresh, out.with_extension("rmeta.fresh")).is_ok() => ++ { ++ out.with_extension("rmeta.fresh") ++ } ++ _ => fresh, ++ }; ++ let at = reused.iter().zip(&encoded).position(|(a, b)| a != b).unwrap_or(reused.len().min(encoded.len())); ++ eprintln!( ++ "rustc-verify-reuse: metadata of `{}` reused from the incremental cache differs from a fresh encoding ({} and {} bytes, first difference at byte {}); kept as {}", ++ tcx.crate_name(LOCAL_CRATE), ++ reused.len(), ++ encoded.len(), ++ at, ++ fresh.display(), ++ ); ++ } ++} ++ + fn encode_root_position(mut file: &File, pos: usize) -> Result<(), std::io::Error> { + file.seek(SeekFrom::Start(ROOT_POS_OFFSET as u64))?; + file.write_all(&pos.to_le_bytes())?; +@@ -2768,6 +2842,18 @@ fn encode_crate_hash(mut file: &File, hash: Svh) -> Result<(), std::io::Error> { + + pub(crate) fn provide(providers: &mut Providers) { + *providers = Providers { ++ local_source_files_fingerprint: |tcx, ()| { ++ let mut hasher = StableHasher::new(); ++ for file in tcx.sess.source_map().files().iter() { ++ if file.is_imported() { ++ continue; ++ } ++ std::hash::Hash::hash(&file.name, &mut hasher); ++ std::hash::Hash::hash(&file.unnormalized_source_len, &mut hasher); ++ std::hash::Hash::hash(&file.src_hash, &mut hasher); ++ } ++ Svh::new(hasher.finish()) ++ }, + doc_link_resolutions: |tcx, def_id| { + tcx.resolutions(()) + .doc_link_resolutions +diff --git a/compiler/rustc_middle/src/dep_graph/graph.rs b/compiler/rustc_middle/src/dep_graph/graph.rs +index dcb5775f2..711ef1b99 100644 +--- a/compiler/rustc_middle/src/dep_graph/graph.rs ++++ b/compiler/rustc_middle/src/dep_graph/graph.rs +@@ -390,6 +390,7 @@ pub fn with_task<'tcx, OP, R>( + let task_deps = Lock::new(TaskDeps::new( + #[cfg(debug_assertions)] + Some(dep_node), ++ Some(dep_node.kind), + )); + (with_deps(TaskDepsRef::Allow(&task_deps), op), Some(task_deps.into_inner())) + }; +@@ -428,6 +429,7 @@ fn with_anon_task_inner<'tcx, OP, R>( + let task_deps = Lock::new(TaskDeps::new( + #[cfg(debug_assertions)] + None, ++ Some(dep_kind), + )); + let result = with_deps(TaskDepsRef::Allow(&task_deps), op); + let task_deps = task_deps.into_inner(); +@@ -1311,15 +1313,22 @@ pub struct TaskDeps { + node: Option, + + reads: TaskReads, ++ ++ /// For reports of untracked reads (`rustc_data_structures::untracked`): the kind of the ++ /// task, and the untracked state it has declared as an input it tracks another way. ++ pub(super) kind: Option, ++ pub(super) declared: Vec<&'static str>, + } + + impl TaskDeps { + #[inline] +- fn new(#[cfg(debug_assertions)] node: Option) -> Self { ++ fn new(#[cfg(debug_assertions)] node: Option, kind: Option) -> Self { + TaskDeps { + #[cfg(debug_assertions)] + node, + reads: TaskReads::new(), ++ kind, ++ declared: Vec::new(), + } + } + +diff --git a/compiler/rustc_middle/src/dep_graph/mod.rs b/compiler/rustc_middle/src/dep_graph/mod.rs +index e63a28100..f531e5e71 100644 +--- a/compiler/rustc_middle/src/dep_graph/mod.rs ++++ b/compiler/rustc_middle/src/dep_graph/mod.rs +@@ -77,6 +77,56 @@ fn read_deps(op: OP) + }) + } + ++/// Declares that the current task tracks `what` another way (through a query that fingerprints ++/// it, say), where `what` is state that [`rustc_data_structures::untracked::untracked_read`] ++/// reports reading. Reads of it in this task are then not reported. ++pub fn declare_untracked_input(what: &'static str) { ++ read_deps(|task_deps| { ++ if let TaskDepsRef::Allow(deps) = task_deps { ++ deps.lock().declared.push(what); ++ } ++ }) ++} ++ ++/// For testing incremental compilation: under `RUSTC_REPORT_UNTRACKED`, report each read of ++/// untracked state inside a task whose result may be reused, unless the task declared it. ++/// With `RUSTC_REPORT_UNTRACKED=all`, declared reads are reported too, marked as declared. ++pub fn enable_untracked_read_reports() { ++ if std::env::var_os("RUSTC_REPORT_UNTRACKED").is_some() { ++ let _ = rustc_data_structures::untracked::UNTRACKED_READ_HOOK.set(report_untracked_read); ++ } ++} ++ ++fn report_untracked_read(what: &'static str, at: &'static panic::Location<'static>) { ++ static SEEN: std::sync::Mutex> = std::sync::Mutex::new(Vec::new()); ++ static ALL: std::sync::LazyLock = std::sync::LazyLock::new(|| { ++ std::env::var_os("RUSTC_REPORT_UNTRACKED").is_some_and(|v| v == "all") ++ }); ++ let mut found = None; ++ read_deps(|task_deps| { ++ if let TaskDepsRef::Allow(deps) = task_deps { ++ let deps = deps.lock(); ++ let declared = deps.declared.contains(&what); ++ if !declared || *ALL { ++ found = deps.kind.map(|kind| (kind, declared)); ++ } ++ } ++ }); ++ let Some((kind, declared)) = found else { return }; ++ let line = if declared { ++ format!("rustc-untracked-read-declared: {what}, read at {at}, while computing `{kind:?}`") ++ } else { ++ format!( ++ "rustc-untracked-read: {what}, read at {at}, while computing `{kind:?}`, whose result incremental compilation may reuse" ++ ) ++ }; ++ let mut seen = SEEN.lock().unwrap(); ++ if !seen.contains(&line) { ++ eprintln!("{line}"); ++ seen.push(line); ++ } ++} ++ + impl<'tcx> TyCtxt<'tcx> { + #[inline] + pub fn dep_kind_vtable(self, dk: DepKind) -> &'tcx DepKindVTable<'tcx> { +diff --git a/compiler/rustc_middle/src/dep_graph/serialized.rs b/compiler/rustc_middle/src/dep_graph/serialized.rs +index 1c476fc91..4a76764f4 100644 +--- a/compiler/rustc_middle/src/dep_graph/serialized.rs ++++ b/compiler/rustc_middle/src/dep_graph/serialized.rs +@@ -911,7 +911,7 @@ pub(crate) fn new( + .unstable_opts + .query_dep_graph + .then(|| Lock::new(RetainedDepGraph::new(prev_index_space_len))); +- let status = EncoderState::new(encoder, sess.opts.unstable_opts.incremental_info, previous); ++ let status = EncoderState::new(encoder, (*sess.opts.unstable_opts.read_incremental_info()), previous); + GraphEncoder { status, retained_graph, profiler: sess.prof.clone() } + } + +diff --git a/compiler/rustc_middle/src/hooks.rs b/compiler/rustc_middle/src/hooks.rs +index 7bccc34db..71f2341b1 100644 +--- a/compiler/rustc_middle/src/hooks.rs ++++ b/compiler/rustc_middle/src/hooks.rs +@@ -106,6 +106,10 @@ fn clone(&self) -> Self { *self } + + hook verify_query_key_hashes() -> (); + ++ /// Under `RUSTC_VERIFY_REUSE`, computes reused query values again and reports any that ++ /// differ. For testing incremental compilation. ++ hook verify_reused_values() -> (); ++ + /// Ensure the given scalar is valid for the given type. + /// This checks non-recursive runtime validity. + hook validate_scalar_in_layout(scalar: crate::ty::ScalarInt, ty: Ty<'tcx>) -> bool; +diff --git a/compiler/rustc_middle/src/lint.rs b/compiler/rustc_middle/src/lint.rs +index 8822ffee6..2cccc2a36 100644 +--- a/compiler/rustc_middle/src/lint.rs ++++ b/compiler/rustc_middle/src/lint.rs +@@ -408,7 +408,7 @@ fn emit_lint_base_impl<'a>( + + let has_future_breakage = future_incompatible.map_or( + // Default allow lints trigger too often for testing. +- sess.opts.unstable_opts.future_incompat_test && lint.default_level != Level::Allow, ++ (*sess.opts.unstable_opts.read_future_incompat_test()) && lint.default_level != Level::Allow, + |incompat| incompat.report_in_deps, + ); + +diff --git a/compiler/rustc_middle/src/mir/generic_graph.rs b/compiler/rustc_middle/src/mir/generic_graph.rs +index 3fd73712b..d7012b88f 100644 +--- a/compiler/rustc_middle/src/mir/generic_graph.rs ++++ b/compiler/rustc_middle/src/mir/generic_graph.rs +@@ -7,7 +7,7 @@ pub(crate) fn mir_fn_to_generic_graph<'tcx>(tcx: TyCtxt<'tcx>, body: &Body<'_>) + let def_id = body.source.def_id(); + let def_name = graphviz_safe_def_name(def_id); + let graph_name = format!("Mir_{def_name}"); +- let dark_mode = tcx.sess.opts.unstable_opts.graphviz_dark_mode; ++ let dark_mode = (*tcx.sess.opts.unstable_opts.read_graphviz_dark_mode()); + + // Nodes + let nodes: Vec = body +diff --git a/compiler/rustc_middle/src/mir/generic_graphviz.rs b/compiler/rustc_middle/src/mir/generic_graphviz.rs +index bce7beb52..684df020d 100644 +--- a/compiler/rustc_middle/src/mir/generic_graphviz.rs ++++ b/compiler/rustc_middle/src/mir/generic_graphviz.rs +@@ -58,11 +58,11 @@ pub fn write_graphviz<'tcx, W>(&self, tcx: TyCtxt<'tcx>, w: &mut W) -> io::Resul + writeln!(w, "{} {}{} {{", kind, cluster, self.graphviz_name)?; + + // Global graph properties +- let font = format!(r#"fontname="{}""#, tcx.sess.opts.unstable_opts.graphviz_font); ++ let font = format!(r#"fontname="{}""#, (*tcx.sess.opts.unstable_opts.read_graphviz_font())); + let mut graph_attrs = vec![&font[..]]; + let mut content_attrs = vec![&font[..]]; + +- let dark_mode = tcx.sess.opts.unstable_opts.graphviz_dark_mode; ++ let dark_mode = (*tcx.sess.opts.unstable_opts.read_graphviz_dark_mode()); + if dark_mode { + graph_attrs.push(r#"bgcolor="black""#); + graph_attrs.push(r#"fontcolor="white""#); +diff --git a/compiler/rustc_middle/src/mir/graphviz.rs b/compiler/rustc_middle/src/mir/graphviz.rs +index 1cd34c35b..05d476384 100644 +--- a/compiler/rustc_middle/src/mir/graphviz.rs ++++ b/compiler/rustc_middle/src/mir/graphviz.rs +@@ -51,11 +51,11 @@ pub fn write_mir_fn_graphviz<'tcx, W>( + W: Write, + { + // Global graph properties +- let font = format!(r#"fontname="{}""#, tcx.sess.opts.unstable_opts.graphviz_font); ++ let font = format!(r#"fontname="{}""#, (*tcx.sess.opts.unstable_opts.read_graphviz_font())); + let mut graph_attrs = vec![&font[..]]; + let mut content_attrs = vec![&font[..]]; + +- let dark_mode = tcx.sess.opts.unstable_opts.graphviz_dark_mode; ++ let dark_mode = (*tcx.sess.opts.unstable_opts.read_graphviz_dark_mode()); + if dark_mode { + graph_attrs.push(r#"bgcolor="black""#); + graph_attrs.push(r#"fontcolor="white""#); +diff --git a/compiler/rustc_middle/src/mir/interpret/mod.rs b/compiler/rustc_middle/src/mir/interpret/mod.rs +index 10aa9532b..6ff5cb4fa 100644 +--- a/compiler/rustc_middle/src/mir/interpret/mod.rs ++++ b/compiler/rustc_middle/src/mir/interpret/mod.rs +@@ -99,6 +99,10 @@ enum AllocDiscriminant { + VTable, + Static, + Type, ++ /// Memory that was deduplicated when it was created (string literals, for instance), and ++ /// must be deduplicated again when decoded: otherwise the same allocation decoded from the ++ /// incremental cache and created afresh would get two `AllocId`s. ++ DedupAlloc, + } + + pub fn specialized_encode_alloc_id<'tcx, E: TyEncoder<'tcx>>( +@@ -109,7 +113,13 @@ pub fn specialized_encode_alloc_id<'tcx, E: TyEncoder<'tcx>>( + match tcx.global_alloc(alloc_id) { + GlobalAlloc::Memory(alloc) => { + trace!("encoding {:?} with {:#?}", alloc_id, alloc); +- AllocDiscriminant::Alloc.encode(encoder); ++ let deduplicated = tcx.alloc_map.dedup.lock().get(&(GlobalAlloc::Memory(alloc), CTFE_ALLOC_SALT)) ++ == Some(&alloc_id); ++ if deduplicated { ++ AllocDiscriminant::DedupAlloc.encode(encoder); ++ } else { ++ AllocDiscriminant::Alloc.encode(encoder); ++ } + alloc.encode(encoder); + } + GlobalAlloc::Function { instance } => { +@@ -216,6 +226,12 @@ pub fn decode_alloc_id<'tcx, D>(&self, decoder: &mut D) -> AllocId + trace!("decoded alloc {:?}", alloc); + decoder.interner().reserve_and_set_memory_alloc(alloc) + } ++ AllocDiscriminant::DedupAlloc => { ++ trace!("creating deduplicated memory alloc ID"); ++ let alloc = as Decodable<_>>::decode(decoder); ++ trace!("decoded alloc {:?}", alloc); ++ decoder.interner().reserve_and_set_memory_dedup(alloc, CTFE_ALLOC_SALT) ++ } + AllocDiscriminant::Fn => { + trace!("creating fn alloc ID"); + let instance = ty::Instance::decode(decoder); +diff --git a/compiler/rustc_middle/src/mir/pretty.rs b/compiler/rustc_middle/src/mir/pretty.rs +index ae30093b1..5ceeb0c88 100644 +--- a/compiler/rustc_middle/src/mir/pretty.rs ++++ b/compiler/rustc_middle/src/mir/pretty.rs +@@ -58,7 +58,7 @@ pub struct PrettyPrintMirOptions { + impl PrettyPrintMirOptions { + /// Create the default set of MIR pretty-printing options from the CLI flags. + pub fn from_cli(tcx: TyCtxt<'_>) -> Self { +- Self { include_extra_comments: tcx.sess.opts.unstable_opts.mir_include_spans.is_enabled() } ++ Self { include_extra_comments: (*tcx.sess.opts.unstable_opts.read_mir_include_spans()).is_enabled() } + } + } + +@@ -80,7 +80,7 @@ impl<'a, 'tcx> MirDumper<'a, 'tcx> { + // - `writer.extra_data`: a no-op + // - `writer.options`: default options derived from CLI flags + pub fn new(tcx: TyCtxt<'tcx>, pass_name: &'static str, body: &Body<'tcx>) -> Option { +- let dump_enabled = if let Some(ref filters) = tcx.sess.opts.unstable_opts.dump_mir { ++ let dump_enabled = if let Some(ref filters) = (*tcx.sess.opts.unstable_opts.read_dump_mir()) { + // see notes on #41697 below + let node_path = ty::print::with_no_trimmed_paths!( + ty::print::with_forced_impl_filename_line!(tcx.def_path_str(body.source.def_id())) +@@ -166,7 +166,7 @@ pub fn dump_mir(&self, body: &Body<'tcx>) { + self.dump_mir_to_writer(body, &mut file)?; + }; + +- if self.tcx().sess.opts.unstable_opts.dump_mir_graphviz { ++ if (*self.tcx().sess.opts.unstable_opts.read_dump_mir_graphviz()) { + let _ = try { + let mut file = self.create_dump_file("dot", body)?; + write_mir_fn_graphviz(self.tcx(), body, false, &mut file)?; +@@ -207,7 +207,7 @@ fn dump_path(&self, extension: &str, body: &Body<'tcx>) -> PathBuf { + None => String::new(), + }; + +- let pass_num = if tcx.sess.opts.unstable_opts.dump_mir_exclude_pass_number { ++ let pass_num = if (*tcx.sess.opts.unstable_opts.read_dump_mir_exclude_pass_number()) { + String::new() + } else if self.show_pass_num { + let (dialect_index, phase_index) = body.phase.index(); +@@ -273,7 +273,7 @@ fn dump_path(&self, extension: &str, body: &Body<'tcx>) -> PathBuf { + }; + + let mut file_path = PathBuf::new(); +- file_path.push(Path::new(&tcx.sess.opts.unstable_opts.dump_mir_dir)); ++ file_path.push(Path::new(&(*tcx.sess.opts.unstable_opts.read_dump_mir_dir()))); + + let pass_name = self.pass_name; + let disambiguator = self.disambiguator; +@@ -1179,7 +1179,7 @@ fn fmt(&self, fmt: &mut Formatter<'_>) -> fmt::Result { + + // When printing regions, add trailing space if necessary. + let print_region = ty::tls::with(|tcx| { +- tcx.sess.verbose_internals() || tcx.sess.opts.unstable_opts.identify_regions ++ tcx.sess.verbose_internals() || (*tcx.sess.opts.unstable_opts.read_identify_regions()) + }); + let region = if print_region { + let mut region = region.to_string(); +@@ -1252,7 +1252,7 @@ fn fmt(&self, fmt: &mut Formatter<'_>) -> fmt::Result { + + AggregateKind::Closure(def_id, args) + | AggregateKind::CoroutineClosure(def_id, args) => ty::tls::with(|tcx| { +- let name = if tcx.sess.opts.unstable_opts.span_free_formats { ++ let name = if (*tcx.sess.opts.unstable_opts.read_span_free_formats()) { + let args = tcx.lift(args); + format!("{{closure@{}}}", tcx.def_path_str_with_args(def_id, args),) + } else { +@@ -1724,7 +1724,7 @@ fn fmt(&self, w: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + // We are done. + return write!(w, " {{}}"); + } +- if tcx.sess.opts.unstable_opts.dump_mir_exclude_alloc_bytes { ++ if (*tcx.sess.opts.unstable_opts.read_dump_mir_exclude_alloc_bytes()) { + return write!(w, " {{ .. }}"); + } + // Write allocation bytes. +diff --git a/compiler/rustc_middle/src/mono.rs b/compiler/rustc_middle/src/mono.rs +index cf567fa1c..82b2c8ef3 100644 +--- a/compiler/rustc_middle/src/mono.rs ++++ b/compiler/rustc_middle/src/mono.rs +@@ -568,7 +568,7 @@ fn item_sort_key<'tcx>(tcx: TyCtxt<'tcx>, item: MonoItem<'tcx>) -> ItemSortKey<' + } + + let mut items: Vec<_> = self.items().iter().map(|(&i, &data)| (i, data)).collect(); +- if !tcx.sess.opts.unstable_opts.codegen_source_order { ++ if !(*tcx.sess.opts.unstable_opts.read_codegen_source_order()) { + // In this case, we do not need to keep the items in any specific order, as the input + // is already deterministic. + // +diff --git a/compiler/rustc_middle/src/queries.rs b/compiler/rustc_middle/src/queries.rs +index d9d618519..b9c94ce6b 100644 +--- a/compiler/rustc_middle/src/queries.rs ++++ b/compiler/rustc_middle/src/queries.rs +@@ -190,6 +190,15 @@ fn describe_as_module(def_id: impl Into, tcx: TyCtxt<'_>) -> String + desc { "get the value of an environment variable" } + } + ++ /// A fingerprint of every local source file's name, length and contents. Metadata ++ /// encodes all three in its source map, so reusing metadata from the incremental ++ /// cache has to depend on them; nothing else that metadata reads does. ++ query local_source_files_fingerprint(_: ()) -> Svh { ++ // The source map is global state ++ eval_always ++ desc { "fingerprinting the local source files" } ++ } ++ + query resolutions(_: ()) -> &'tcx ResolverGlobalCtxt { + desc { "getting the resolver outputs" } + } +diff --git a/compiler/rustc_middle/src/query/on_disk_cache.rs b/compiler/rustc_middle/src/query/on_disk_cache.rs +index 956c59012..957ddcb20 100644 +--- a/compiler/rustc_middle/src/query/on_disk_cache.rs ++++ b/compiler/rustc_middle/src/query/on_disk_cache.rs +@@ -756,6 +756,51 @@ pub struct CacheEncoder<'tcx> { + side_effects_index: Vec<(SerializedDepNodeIndex, AbsoluteBytePos)>, + } + ++/// For testing incremental compilation (`RUSTC_VERIFY_REUSE`): `value` as the cache would ++/// encode it, alone, in a fresh encoder, so that two values can be compared by the bytes a later ++/// session would read, and the allocations it refers to in the order it first refers to them. ++/// `path` is a scratch file. ++pub fn encode_alone<'tcx, V: Encodable>>( ++ tcx: TyCtxt<'tcx>, ++ path: &std::path::Path, ++ value: &V, ++) -> (Vec, Vec) { ++ let file_to_file_index = tcx ++ .sess ++ .source_map() ++ .files() ++ .iter() ++ .enumerate() ++ .map(|(index, file)| (&raw const **file, SourceFileIndex(index as u32))) ++ .collect(); ++ let Ok(file) = FileEncoder::new(path) else { return Default::default() }; ++ let mut encoder = CacheEncoder { ++ tcx, ++ encoder: file, ++ type_shorthands: Default::default(), ++ predicate_shorthands: Default::default(), ++ interpret_allocs: Default::default(), ++ caching_source_map_view: CachingSourceMapView::new(tcx.sess.source_map()), ++ file_to_file_index, ++ hygiene_context: Default::default(), ++ symbol_index_table: Default::default(), ++ source_span_cache: Default::default(), ++ query_values_index: Default::default(), ++ side_effects_index: Default::default(), ++ }; ++ value.encode(&mut encoder); ++ // The allocations it refers to, as `serialize` encodes them, so that their contents are ++ // compared too. ++ let mut n = 0; ++ while n < encoder.interpret_allocs.len() { ++ let id = encoder.interpret_allocs[n]; ++ interpret::specialized_encode_alloc_id(&mut encoder, tcx, id); ++ n += 1; ++ } ++ let _ = encoder.encoder.finish(); ++ (std::fs::read(path).unwrap_or_default(), encoder.interpret_allocs.into_iter().collect()) ++} ++ + impl<'tcx> fmt::Debug for CacheEncoder<'tcx> { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + // Add more details here if/when necessary. +diff --git a/compiler/rustc_middle/src/ty/context.rs b/compiler/rustc_middle/src/ty/context.rs +index aa7fe0872..e56a7d587 100644 +--- a/compiler/rustc_middle/src/ty/context.rs ++++ b/compiler/rustc_middle/src/ty/context.rs +@@ -1155,7 +1155,7 @@ pub fn needs_hir_hash(self) -> bool { + || self.sess.opts.incremental.is_some() + || self.needs_metadata() + || self.sess.instrument_coverage() +- || self.sess.opts.unstable_opts.metrics_dir.is_some() ++ || (*self.sess.opts.unstable_opts.read_metrics_dir()).is_some() + } + + /// Whether the combined per-owner HIR hash (`OwnerInfo::opt_hash`, which folds `parenting`, +@@ -2805,7 +2805,7 @@ pub fn is_const_trait_impl(self, def_id: DefId) -> bool { + } + + pub fn is_sdylib_interface_build(self) -> bool { +- self.sess.opts.unstable_opts.build_sdylib_interface ++ (*self.sess.opts.unstable_opts.read_build_sdylib_interface()) + } + + pub fn intrinsic(self, def_id: impl IntoQueryKey) -> Option { +diff --git a/compiler/rustc_middle/src/ty/error.rs b/compiler/rustc_middle/src/ty/error.rs +index fb4e30b16..c6a883286 100644 +--- a/compiler/rustc_middle/src/ty/error.rs ++++ b/compiler/rustc_middle/src/ty/error.rs +@@ -286,7 +286,7 @@ pub fn short_string_namespace( + let regular = FmtPrinter::print_string(self, namespace, |p| t.print(p)) + .expect("could not write to `String`"); + +- if !self.sess.opts.unstable_opts.write_long_types_to_disk || self.sess.opts.verbose { ++ if !(*self.sess.opts.unstable_opts.read_write_long_types_to_disk()) || self.sess.opts.verbose { + return regular; + } + +diff --git a/compiler/rustc_middle/src/ty/generics.rs b/compiler/rustc_middle/src/ty/generics.rs +index 353ac74bd..5c52883cf 100644 +--- a/compiler/rustc_middle/src/ty/generics.rs ++++ b/compiler/rustc_middle/src/ty/generics.rs +@@ -1,7 +1,7 @@ + use std::ops::ControlFlow; + + use rustc_ast as ast; +-use rustc_data_structures::fx::FxHashMap; ++use rustc_data_structures::fx::FxIndexMap; + use rustc_hir::def_id::DefId; + use rustc_macros::{StableHash, TyDecodable, TyEncodable}; + use rustc_span::{Span, Symbol, bug, kw}; +@@ -124,7 +124,7 @@ pub struct Generics { + + /// Reverse map to the `index` field of each `GenericParamDef`. + #[stable_hash(ignore)] +- pub param_def_id_to_index: FxHashMap, ++ pub param_def_id_to_index: FxIndexMap, + + pub has_self: bool, + pub has_late_bound_regions: Option, +@@ -132,8 +132,6 @@ pub struct Generics { + + impl std::fmt::Debug for Generics { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> Result<(), std::fmt::Error> { +- // ironically, we get this warning because of what we're trying to fix. +- #[expect(rustc::potential_query_instability)] + let mut stabilized_hashmap = self.param_def_id_to_index.iter().collect::>(); + stabilized_hashmap.sort_by_key(|(_, v)| **v); + f.debug_struct("Generics") +diff --git a/compiler/rustc_middle/src/ty/print/pretty.rs b/compiler/rustc_middle/src/ty/print/pretty.rs +index cafe93c59..5da0ec74c 100644 +--- a/compiler/rustc_middle/src/ty/print/pretty.rs ++++ b/compiler/rustc_middle/src/ty/print/pretty.rs +@@ -453,7 +453,7 @@ fn try_print_trimmed_def_path(&mut self, def_id: DefId) -> Result Result<(), PrintError> { + if self.should_truncate() { + write!(self, "@...") +- } else if self.tcx().sess.opts.unstable_opts.span_free_formats { ++ } else if (*self.tcx().sess.opts.unstable_opts.read_span_free_formats()) { + write!(self, "@")?; + self.print_def_path(did, args) + } else if let Some(did) = did.as_local() { +@@ -2585,7 +2585,7 @@ fn should_print_optional_region(&self, region: ty::Region<'tcx>) -> bool { + return false; + } + +- let identify_regions = self.tcx.sess.opts.unstable_opts.identify_regions; ++ let identify_regions = (*self.tcx.sess.opts.unstable_opts.read_identify_regions()); + + match region.kind() { + ty::ReEarlyParam(ref data) => data.is_named(), +@@ -2648,7 +2648,7 @@ pub fn pretty_print_region(&mut self, region: ty::Region<'tcx>) -> Result<(), fm + return Ok(()); + } + +- let identify_regions = self.tcx.sess.opts.unstable_opts.identify_regions; ++ let identify_regions = (*self.tcx.sess.opts.unstable_opts.read_identify_regions()); + + // These printouts are concise. They do not contain all the information + // the user might want to diagnose an error, but there is basically no way +diff --git a/compiler/rustc_mir_build/src/check_unsafety.rs b/compiler/rustc_mir_build/src/check_unsafety.rs +index e83c3560a..134b4f41d 100644 +--- a/compiler/rustc_mir_build/src/check_unsafety.rs ++++ b/compiler/rustc_mir_build/src/check_unsafety.rs +@@ -182,7 +182,7 @@ fn visit_inner_body(&mut self, def: LocalDefId) { + if let Ok((inner_thir, expr)) = self.tcx.thir_body(def) { + // Run all other queries that depend on THIR. + self.tcx.ensure_done().mir_built(def); +- let inner_thir = if self.tcx.sess.opts.unstable_opts.no_steal_thir { ++ let inner_thir = if (*self.tcx.sess.opts.unstable_opts.read_no_steal_thir()) { + &inner_thir.borrow() + } else { + // We don't have other use for the THIR. Steal it to reduce memory usage. +@@ -1061,7 +1061,7 @@ pub(crate) fn check_unsafety(tcx: TyCtxt<'_>, def: LocalDefId) { + let Ok((thir, expr)) = tcx.thir_body(def) else { return }; + // Runs all other queries that depend on THIR. + tcx.ensure_done().mir_built(def); +- let thir = if tcx.sess.opts.unstable_opts.no_steal_thir { ++ let thir = if (*tcx.sess.opts.unstable_opts.read_no_steal_thir()) { + &thir.borrow() + } else { + // We don't have other use for the THIR. Steal it to reduce memory usage. +diff --git a/compiler/rustc_mir_dataflow/src/framework/graphviz.rs b/compiler/rustc_mir_dataflow/src/framework/graphviz.rs +index b1e4bbe98..dc7407ce0 100644 +--- a/compiler/rustc_mir_dataflow/src/framework/graphviz.rs ++++ b/compiler/rustc_mir_dataflow/src/framework/graphviz.rs +@@ -64,8 +64,8 @@ pub(super) fn write_graphviz_results<'tcx, A>( + + let graphviz = Formatter::new(body, results, style); + let mut render_opts = +- vec![dot::RenderOption::Fontname(tcx.sess.opts.unstable_opts.graphviz_font.clone())]; +- if tcx.sess.opts.unstable_opts.graphviz_dark_mode { ++ vec![dot::RenderOption::Fontname((*tcx.sess.opts.unstable_opts.read_graphviz_font()).clone())]; ++ if (*tcx.sess.opts.unstable_opts.read_graphviz_dark_mode()) { + render_opts.push(dot::RenderOption::DarkTheme); + } + with_no_trimmed_paths!(dot::render_opts(&graphviz, &mut buf, &render_opts))?; +diff --git a/compiler/rustc_mir_dataflow/src/framework/mod.rs b/compiler/rustc_mir_dataflow/src/framework/mod.rs +index 8bb5e7458..a795ee8cb 100644 +--- a/compiler/rustc_mir_dataflow/src/framework/mod.rs ++++ b/compiler/rustc_mir_dataflow/src/framework/mod.rs +@@ -383,7 +383,7 @@ struct BasicBlockRank {} + + let results = Results { analysis: self, entry_states }; + +- if tcx.sess.opts.unstable_opts.dump_mir_dataflow { ++ if (*tcx.sess.opts.unstable_opts.read_dump_mir_dataflow()) { + let res = write_graphviz_results(tcx, body, &results, pass_name); + if let Err(e) = res { + error!("Failed to write graphviz dataflow results: {}", e); +diff --git a/compiler/rustc_mir_transform/src/coroutine/mod.rs b/compiler/rustc_mir_transform/src/coroutine/mod.rs +index 1e63dca78..e42069318 100644 +--- a/compiler/rustc_mir_transform/src/coroutine/mod.rs ++++ b/compiler/rustc_mir_transform/src/coroutine/mod.rs +@@ -1106,7 +1106,7 @@ fn run_pass(&self, tcx: TyCtxt<'tcx>, body: &mut Body<'tcx>) { + let liveness_info = + locals_live_across_suspend_points(tcx, body, &always_live_locals, movable); + +- if tcx.sess.opts.unstable_opts.validate_mir { ++ if (*tcx.sess.opts.unstable_opts.read_validate_mir()) { + let mut vis = EnsureCoroutineFieldAssignmentsNeverAlias { + assigned_local: None, + saved_locals: &liveness_info.saved_locals, +diff --git a/compiler/rustc_mir_transform/src/pass_manager.rs b/compiler/rustc_mir_transform/src/pass_manager.rs +index f3f65a070..8ecc26910 100644 +--- a/compiler/rustc_mir_transform/src/pass_manager.rs ++++ b/compiler/rustc_mir_transform/src/pass_manager.rs +@@ -321,8 +321,8 @@ fn run_passes_inner<'tcx>( + let prof_arg = tcx.sess.prof.enabled().then(|| format!("{:?}", body.source.def_id())); + + if !body.should_skip() { +- let validate = validate_each & tcx.sess.opts.unstable_opts.validate_mir; +- let lint = tcx.sess.opts.unstable_opts.lint_mir; ++ let validate = validate_each & (*tcx.sess.opts.unstable_opts.read_validate_mir()); ++ let lint = (*tcx.sess.opts.unstable_opts.read_lint_mir()); + + let ctx = PassCtx::for_body(tcx, body.source.def_id()); + +@@ -394,9 +394,9 @@ fn run_passes_inner<'tcx>( + dump_mir_for_phase_change(tcx, body); + + let validate = +- (validate_each & tcx.sess.opts.unstable_opts.validate_mir & !body.should_skip()) ++ (validate_each & (*tcx.sess.opts.unstable_opts.read_validate_mir()) & !body.should_skip()) + || new_phase == MirPhase::Runtime(RuntimePhase::Optimized); +- let lint = tcx.sess.opts.unstable_opts.lint_mir & !body.should_skip(); ++ let lint = (*tcx.sess.opts.unstable_opts.read_lint_mir()) & !body.should_skip(); + if validate { + validate_body(tcx, body, format!("after phase change to {}", new_phase.name())); + } +diff --git a/compiler/rustc_mir_transform/src/validate.rs b/compiler/rustc_mir_transform/src/validate.rs +index 28a508bde..a456c9d50 100644 +--- a/compiler/rustc_mir_transform/src/validate.rs ++++ b/compiler/rustc_mir_transform/src/validate.rs +@@ -433,7 +433,7 @@ fn visit_terminator(&mut self, terminator: &Terminator<'tcx>, location: Location + + // Call arguments are moved by reference, so they must be plain locals + // or the contents of a box; other moved places violate MIR invariants. +- if self.tcx.sess.opts.unstable_opts.validate_mir ++ if (*self.tcx.sess.opts.unstable_opts.read_validate_mir()) + && self.body.phase < MirPhase::Runtime(RuntimePhase::Initial) + { + let is_plain_local = place.projection.is_empty(); +@@ -648,7 +648,7 @@ fn predicate_must_hold_modulo_regions( + impl<'a, 'tcx> Visitor<'tcx> for TypeChecker<'a, 'tcx> { + fn visit_operand(&mut self, operand: &Operand<'tcx>, location: Location) { + // This check is somewhat expensive, so only run it when -Zvalidate-mir is passed. +- if self.tcx.sess.opts.unstable_opts.validate_mir ++ if (*self.tcx.sess.opts.unstable_opts.read_validate_mir()) + && self.body.phase < MirPhase::Runtime(RuntimePhase::Initial) + { + // `Operand::Copy` is only supposed to be used with `Copy` types. +diff --git a/compiler/rustc_monomorphize/src/collector.rs b/compiler/rustc_monomorphize/src/collector.rs +index bfe3ba91b..7b204f959 100644 +--- a/compiler/rustc_monomorphize/src/collector.rs ++++ b/compiler/rustc_monomorphize/src/collector.rs +@@ -1223,7 +1223,7 @@ fn create_fn_mono_item<'tcx>( + source: Span, + ) -> Spanned> { + let def_id = instance.def_id(); +- if tcx.sess.opts.unstable_opts.profile_closures ++ if (*tcx.sess.opts.unstable_opts.read_profile_closures()) + && def_id.is_local() + && tcx.is_closure_like(def_id) + { +diff --git a/compiler/rustc_monomorphize/src/partitioning.rs b/compiler/rustc_monomorphize/src/partitioning.rs +index 16df895b0..75376ac08 100644 +--- a/compiler/rustc_monomorphize/src/partitioning.rs ++++ b/compiler/rustc_monomorphize/src/partitioning.rs +@@ -1221,7 +1221,7 @@ fn collect_and_partition_mono_items(tcx: TyCtxt<'_>, (): ()) -> MonoItemPartitio + tcx.dcx().emit_fatal(CouldntDumpMonoStats { error: err.to_string() }); + } + +- if tcx.sess.opts.unstable_opts.print_mono_items { ++ if (*tcx.sess.opts.unstable_opts.read_print_mono_items()) { + let mut item_to_cgus: UnordMap<_, Vec<_>> = Default::default(); + + for cgu in codegen_units { +@@ -1287,7 +1287,7 @@ fn dump_mono_items_stats<'tcx>( + Path::new(".") + }; + +- let format = tcx.sess.opts.unstable_opts.dump_mono_stats_format; ++ let format = (*tcx.sess.opts.unstable_opts.read_dump_mono_stats_format()); + let ext = format.extension(); + let filename = format!("{crate_name}.mono_items.{ext}"); + let output_path = output_directory.join(&filename); +diff --git a/compiler/rustc_query_impl/src/execution.rs b/compiler/rustc_query_impl/src/execution.rs +index ca60614ba..e3bf3285f 100644 +--- a/compiler/rustc_query_impl/src/execution.rs ++++ b/compiler/rustc_query_impl/src/execution.rs +@@ -543,7 +543,7 @@ fn load_from_disk_or_invoke_provider_green<'tcx, C: QueryCache>( + }; + let (value, verify) = match try_value { + Some(value) => { +- if std::intrinsics::unlikely(tcx.sess.opts.unstable_opts.query_dep_graph) { ++ if std::intrinsics::unlikely((*tcx.sess.opts.unstable_opts.read_query_dep_graph())) { + dep_graph_data.mark_debug_loaded_from_disk(*dep_node) + } + +diff --git a/compiler/rustc_query_impl/src/incremental.rs b/compiler/rustc_query_impl/src/incremental.rs +index a030fe71e..0cfa01b39 100644 +--- a/compiler/rustc_query_impl/src/incremental.rs ++++ b/compiler/rustc_query_impl/src/incremental.rs +@@ -1,17 +1,20 @@ + use rustc_data_structures::fingerprint::{Fingerprint, PackedFingerprint}; ++use rustc_data_structures::fx::FxHashMap; + use rustc_data_structures::unord::UnordMap; + #[expect(unused_imports, reason = "used by doc comments")] + use rustc_middle::dep_graph::DepKindVTable; + use rustc_middle::dep_graph::{ + DepGraphData, DepNode, DepNodeIndex, DepNodeKey, SerializedDepNodeIndex, + }; ++use rustc_middle::mir::interpret::AllocId; + use rustc_middle::query::erase::{Erasable, Erased}; +-use rustc_middle::query::on_disk_cache::{CacheDecoder, CacheEncoder}; +-use rustc_middle::query::{QueryCache, QueryState, QueryVTable, erase}; ++use rustc_middle::query::on_disk_cache::{self, CacheDecoder, CacheEncoder}; ++use rustc_middle::query::{QueryCache, QueryKey, QueryState, QueryVTable, erase}; + use rustc_middle::ty::TyCtxt; + use rustc_middle::verify_ich::incremental_verify_ich; + use rustc_serialize::{Decodable, Encodable}; + use rustc_span::bug; ++use rustc_span::def_id::LocalDefId; + + use crate::query_vtables::for_each_query_vtable; + +@@ -25,6 +28,319 @@ pub(crate) fn encode_query_values<'tcx>(tcx: TyCtxt<'tcx>, encoder: &mut CacheEn + }); + } + ++/// Shadow verification, for testing incremental compilation, under `RUSTC_VERIFY_REUSE`: at ++/// the end of the session, before the dependency graph and the cache are saved, compute every ++/// green value of a cached query again with its provider, outside dependency tracking, and ++/// compare: by stable hash; by `Debug` text, which also sees fields the stable hash ignores; ++/// and by the bytes the cache would encode, which also see the order of hash maps. A green ++/// value was reused from the previous session; a difference means it was stale (its ++/// computation read something that is not tracked), or that it did not survive the round trip ++/// through the cache. ++/// ++/// It runs here because no query is executing, so recomputing cannot cycle back into the ++/// query being checked, and the previous session's cache is still readable. A difference ++/// prints a line starting `rustc-verify-reuse:`. `RUSTC_VERIFY_REUSE_SKIP` lists more queries, comma-separated, not to recompute. ++pub(crate) fn verify_reused_values<'tcx>(tcx: TyCtxt<'tcx>) { ++ if std::env::var_os("RUSTC_VERIFY_REUSE").is_none() { ++ return; ++ } ++ // Load every green value that was not used this session, so it is checked too. ++ tcx.dep_graph.exec_cache_promotions(tcx); ++ // Printing values must not use trimmed paths, which assume a diagnostic is emitted. ++ let mut allocs = Allocs::default(); ++ rustc_middle::ty::print::with_no_trimmed_paths!(for_each_query_vtable!( ++ CACHE_ON_DISK, ++ tcx, ++ |query| verify_reused_values_inner(tcx, query, &mut allocs) ++ )); ++} ++ ++/// `Debug` text in a form two equal values print the same: the contents of `OnceLock`s are ++/// removed, since they are caches filled on demand and a value that has been used shows more ++/// than a fresh one; the elements of `UnordMap`s and `UnordSet`s are sorted, since their ++/// order is unspecified; and the numbers of `AllocId`s (`alloc12`) are removed, since a ++/// decoded allocation and a computed one get different ones. (The encoding comparison sees ++/// the allocations' contents.) ++fn normalized(text: &str) -> String { ++ normalized_inner(&without_alloc_ids(text)) ++} ++ ++fn without_alloc_ids(text: &str) -> String { ++ let mut out = String::with_capacity(text.len()); ++ let mut rest = text; ++ while let Some(at) = rest.find("alloc") { ++ let after = &rest[at + "alloc".len()..]; ++ let digits = after.len() - after.trim_start_matches(|c: char| c.is_ascii_digit()).len(); ++ let word = rest[..at].chars().next_back().is_some_and(|c| c.is_alphanumeric() || c == '_'); ++ out.push_str(&rest[..at + "alloc".len()]); ++ if digits > 0 && !word { ++ out.push('_'); ++ } else { ++ out.push_str(&after[..digits]); ++ } ++ rest = &after[digits..]; ++ } ++ out.push_str(rest); ++ out ++} ++ ++fn normalized_inner(text: &str) -> String { ++ // The index of the bracket closing the one just before `text[from..]`. ++ fn closing(text: &str, from: usize) -> usize { ++ let mut depth = 1; ++ for (i, c) in text[from..].char_indices() { ++ match c { ++ '(' | '[' | '{' => depth += 1, ++ ')' | ']' | '}' => depth -= 1, ++ _ => {} ++ } ++ if depth == 0 { ++ return from + i; ++ } ++ } ++ text.len() ++ } ++ // `text` split at top-level `, `. ++ fn elements(text: &str) -> Vec<&str> { ++ let (mut out, mut depth, mut start) = (Vec::new(), 0, 0); ++ let bytes = text.as_bytes(); ++ for (i, c) in text.char_indices() { ++ match c { ++ '(' | '[' | '{' => depth += 1, ++ ')' | ']' | '}' => depth -= 1, ++ ',' if depth == 0 && bytes.get(i + 1) == Some(&b' ') => { ++ out.push(&text[start..i]); ++ start = i + 2; ++ } ++ _ => {} ++ } ++ } ++ if start < text.len() { ++ out.push(&text[start..]); ++ } ++ out ++ } ++ let mut out = String::with_capacity(text.len()); ++ let mut rest = text; ++ loop { ++ let cache = rest.find("OnceLock(").map(|at| (at, "OnceLock(", false)); ++ let unord = ["UnordMap { inner: {", "UnordSet { inner: {"] ++ .into_iter() ++ .filter_map(|m| rest.find(m).map(|at| (at, m, true))) ++ .min(); ++ let Some((at, marker, sort)) = [cache, unord].into_iter().flatten().min() else { ++ break; ++ }; ++ let open = at + marker.len(); ++ let end = closing(rest, open); ++ out.push_str(&rest[..open]); ++ if sort { ++ let mut items: Vec = ++ elements(&rest[open..end]).into_iter().map(normalized_inner).collect(); ++ items.sort(); ++ out.push_str(&items.join(", ")); ++ } else { ++ out.push('_'); ++ } ++ rest = &rest[end..]; ++ } ++ out.push_str(rest); ++ out ++} ++ ++/// Which allocations values share. For every value of the queries in `SHARING`, each ++/// allocation a fresh computation refers to is mapped to the one the value in use refers to in ++/// the same place (a value computed this session maps its allocations to themselves). Two ++/// different allocations for one fresh one mean the reused value does not share an allocation ++/// that a clean session would share: the round trip through the cache lost a deduplication. ++#[derive(Default)] ++struct Allocs { ++ /// Fresh allocation → the allocation in use, and where it was first seen. ++ seen: FxHashMap, ++} ++ ++impl Allocs { ++ fn check(&mut self, fresh: AllocId, used: AllocId, here: impl Fn() -> String) { ++ match self.seen.get(&fresh) { ++ None => { ++ self.seen.insert(fresh, (used, here())); ++ } ++ Some((other_used, other)) if *other_used != used => eprintln!( ++ "rustc-verify-reuse: allocation shared differently: {} uses {used:?} where a fresh computation uses {fresh:?}, but {other} uses {other_used:?} for it", ++ here(), ++ ), ++ Some(_) => {} ++ } ++ } ++} ++ ++/// Queries whose values refer to allocations that may be shared with other values. ++const SHARING: &[&str] = &[ ++ "eval_static_initializer", ++ "eval_to_allocation_raw", ++ "eval_to_const_value_raw", ++ "mir_for_ctfe", ++ "optimized_mir", ++ "promoted_mir", ++ "trivial_const", ++]; ++ ++/// Queries whose providers read the MIR or THIR of the definition they are given, which is ++/// stolen once it has been used: they are recomputed only if none of it has been stolen. ++const READING_BODIES: &[&str] = &[ ++ "check_liveness", ++ "check_match", ++ "check_tail_calls", ++ "check_unsafety", ++ "has_ffi_unwind_calls", ++ "mir_const_qualif", ++ "mir_coroutine_witnesses", ++ "mir_for_ctfe", ++ "mir_inliner_callees", ++ "optimized_mir", ++ "promoted_mir", ++ "thir_abstract_const", ++ "trivial_const", ++]; ++ ++/// Queries never recomputed: `mir_borrowck` reads the MIR of nested bodies too, and ++/// `coroutine_by_move_body_def_id` makes a definition. ++const NOT_RECOMPUTED: &[&str] = &["coroutine_by_move_body_def_id", "mir_borrowck"]; ++ ++/// Whether any MIR or THIR of `def` computed this session has been stolen. ++// Only decides whether to check a value, outside dependency tracking. ++#[allow(rustc::untracked_query_information)] ++fn body_stolen<'tcx>(tcx: TyCtxt<'tcx>, def: LocalDefId) -> bool { ++ use rustc_middle::queries::{ ++ mir_built, mir_drops_elaborated_and_const_checked as mir_elaborated, mir_promoted, ++ thir_body, ++ }; ++ let vtables = &tcx.query_system.query_vtables; ++ let built = vtables.mir_built.cache.lookup(&def).is_some_and(|(body, _)| { ++ erase::restore_val::>(body).is_stolen() ++ }); ++ let elaborated = vtables.mir_drops_elaborated_and_const_checked.cache.lookup(&def).is_some_and( ++ |(body, _)| erase::restore_val::>(body).is_stolen(), ++ ); ++ let promoted = vtables.mir_promoted.cache.lookup(&def).is_some_and(|(value, _)| { ++ let (body, promoted) = erase::restore_val::>(value); ++ body.is_stolen() || promoted.is_stolen() ++ }); ++ let thir = vtables.thir_body.cache.lookup(&def).is_some_and(|(value, _)| { ++ erase::restore_val::>(value).is_ok_and(|(thir, _)| thir.is_stolen()) ++ }); ++ built || elaborated || promoted || thir ++} ++ ++fn verify_reused_values_inner<'tcx, C, V>( ++ tcx: TyCtxt<'tcx>, ++ query: &'tcx QueryVTable<'tcx, C>, ++ allocs: &mut Allocs, ++) where ++ C: QueryCache>, ++ V: Erasable + Encodable>, ++{ ++ if NOT_RECOMPUTED.contains(&query.name) { ++ return; ++ } ++ let reads_body = READING_BODIES.contains(&query.name); ++ if let Some(skip) = std::env::var_os("RUSTC_VERIFY_REUSE_SKIP") ++ && skip.to_string_lossy().split(',').any(|name| name == query.name) ++ { ++ return; ++ } ++ let mut entries = Vec::new(); ++ query.cache.for_each(&mut |key, value, _| entries.push((*key, *value))); ++ let hash = |value: &C::Value| { ++ query.hash_value_fn.map_or(Fingerprint::ZERO, |f| { ++ tcx.with_stable_hashing_context(|mut hcx| f(&mut hcx, value)) ++ }) ++ }; ++ let scratch = std::env::temp_dir().join(format!("rustc-verify-reuse-{}", std::process::id())); ++ let mut checked = 0; ++ for (key, value) in entries { ++ if !query.will_cache_on_disk_for_key(key) { ++ continue; ++ } ++ let sharing = SHARING.contains(&query.name); ++ let dep_node = DepNode::construct(tcx, query.dep_kind, &key); ++ if !tcx.dep_graph.is_green(&dep_node) { ++ if sharing { ++ let (_, ids) = ++ on_disk_cache::encode_alone(tcx, &scratch, &erase::restore_val::(value)); ++ let here = || format!("query `{}` for {:?}, computed this session", query.name, key); ++ for id in ids { ++ allocs.check(id, id, here); ++ } ++ } ++ continue; ++ } ++ // A feedable query's value for a definition the compiler made up is set by whatever ++ // made it, and the provider may not accept the key: the associated type for an ++ // `impl Trait` in a trait (a synthetic HIR node), an elided lifetime that lowering ++ // adds as a parameter, or the type of a const argument, set when it is lowered. ++ if query.feedable ++ && key.key_as_def_id().and_then(|id| id.as_local()).is_some_and(|id| { ++ matches!( ++ tcx.def_kind(id), ++ rustc_hir::def::DefKind::LifetimeParam | rustc_hir::def::DefKind::AnonConst ++ ) || matches!(tcx.hir_node_by_def_id(id), rustc_hir::Node::Synthetic) ++ }) ++ { ++ continue; ++ } ++ if reads_body ++ && key ++ .key_as_def_id() ++ .and_then(|id| id.as_local()) ++ .is_none_or(|id| body_stolen(tcx, id) || body_stolen(tcx, tcx.typeck_root_def_id_local(id))) ++ { ++ continue; ++ } ++ checked += 1; ++ let fresh = tcx.dep_graph.with_ignore(|| (query.invoke_provider_fn)(tcx, key)); ++ let text = (query.format_value)(&value); ++ let loaded = normalized(&text); ++ let computed = normalized(&(query.format_value)(&fresh)); ++ let same_hash = hash(&value) == hash(&fresh); ++ let same_text = loaded == computed; ++ // `UnordMap`s and `UnordSet`s encode in their iteration order, which differs between a ++ // decoded value and a computed one and which nothing may observe, so the encoding of a ++ // value containing one is not compared. ++ let (loaded_bytes, loaded_ids) = ++ on_disk_cache::encode_alone(tcx, &scratch, &erase::restore_val::(value)); ++ let (fresh_bytes, fresh_ids) = ++ on_disk_cache::encode_alone(tcx, &scratch, &erase::restore_val::(fresh)); ++ let same_bytes = ++ text.contains("UnordMap {") || text.contains("UnordSet {") || loaded_bytes == fresh_bytes; ++ if sharing && loaded_ids.len() == fresh_ids.len() { ++ let here = || format!("query `{}` for {:?}, green", query.name, key); ++ for (fresh_id, loaded_id) in fresh_ids.into_iter().zip(loaded_ids) { ++ allocs.check(fresh_id, loaded_id, here); ++ } ++ } ++ if !(same_hash && same_text && same_bytes) { ++ let short = |s: &str| s.chars().take(2000).collect::(); ++ let what = [(same_hash, "stable hash"), (same_text, "Debug text"), (same_bytes, "encoding")] ++ .into_iter() ++ .filter_map(|(same, what)| (!same).then_some(what)) ++ .collect::>() ++ .join(", "); ++ eprintln!( ++ "rustc-verify-reuse: query `{}` for {:?}, green, differs from a fresh computation ({what})\n reused: {}\n fresh: {}", ++ query.name, ++ key, ++ short(&loaded), ++ short(&computed), ++ ); ++ } ++ } ++ let _ = std::fs::remove_file(&scratch); ++ if checked > 0 && std::env::var_os("RUSTC_VERIFY_REUSE").is_some_and(|v| v == "verbose") { ++ eprintln!("rustc-verify-reuse-checked: {} {checked}", query.name); ++ } ++} ++ + fn encode_query_values_inner<'tcx, C, V>( + tcx: TyCtxt<'tcx>, + query: &'tcx QueryVTable<'tcx, C>, +@@ -44,7 +360,7 @@ fn encode_query_values_inner<'tcx, C, V>( + } + + pub(crate) fn verify_query_key_hashes<'tcx>(tcx: TyCtxt<'tcx>) { +- if tcx.sess.opts.unstable_opts.incremental_verify_ich || cfg!(debug_assertions) { ++ if (*tcx.sess.opts.unstable_opts.read_incremental_verify_ich()) || cfg!(debug_assertions) { + tcx.sess.time("verify_query_key_hashes", || { + for_each_query_vtable!(ALL, tcx, |query| { + verify_query_key_hashes_inner(query, tcx); +@@ -99,7 +415,7 @@ pub(crate) fn should_verify_loaded_value( + ) -> bool { + let hash = Fingerprint::from(key_fingerprint).to_smaller_hash().as_u64(); + hash % 32 == dep_graph_data.session_count() % 32 +- || tcx.sess.opts.unstable_opts.incremental_verify_ich ++ || (*tcx.sess.opts.unstable_opts.read_incremental_verify_ich()) + } + + /// Inner implementation of [`DepKindVTable::promote_from_disk_fn`] for queries. +diff --git a/compiler/rustc_query_impl/src/lib.rs b/compiler/rustc_query_impl/src/lib.rs +index eed772cae..5eecf938d 100644 +--- a/compiler/rustc_query_impl/src/lib.rs ++++ b/compiler/rustc_query_impl/src/lib.rs +@@ -52,4 +52,5 @@ pub fn provide(providers: &mut rustc_middle::util::Providers) { + self_profile::alloc_self_profile_query_strings; + providers.hooks.verify_query_key_hashes = incremental::verify_query_key_hashes; + providers.hooks.encode_query_values = incremental::encode_query_values; ++ providers.hooks.verify_reused_values = incremental::verify_reused_values; + } +diff --git a/compiler/rustc_session/src/config.rs b/compiler/rustc_session/src/config.rs +index 64cfdfd5f..7c3993f87 100644 +--- a/compiler/rustc_session/src/config.rs ++++ b/compiler/rustc_session/src/config.rs +@@ -1528,8 +1528,8 @@ impl Options { + /// Returns `true` if there is a reason to build the dep graph. + pub fn build_dep_graph(&self) -> bool { + self.incremental.is_some() +- || self.unstable_opts.dump_dep_graph +- || self.unstable_opts.query_dep_graph ++ || (*self.unstable_opts.read_dump_dep_graph()) ++ || (*self.unstable_opts.read_query_dep_graph()) + } + + pub fn file_path_mapping(&self) -> FilePathMapping { +@@ -1542,8 +1542,8 @@ pub fn file_path_mapping(&self) -> FilePathMapping { + + /// Returns `true` if there will be an output file generated. + pub fn will_create_output_file(&self) -> bool { +- !self.unstable_opts.parse_crate_root_only && // The file is just being parsed +- self.unstable_opts.ls.is_empty() // The file is just being queried ++ !(*self.unstable_opts.read_parse_crate_root_only()) && // The file is just being parsed ++ (*self.unstable_opts.read_ls()).is_empty() // The file is just being queried + } + + #[inline] +diff --git a/compiler/rustc_session/src/diagnostics.rs b/compiler/rustc_session/src/diagnostics.rs +index e673435da..8fad9bb43 100644 +--- a/compiler/rustc_session/src/diagnostics.rs ++++ b/compiler/rustc_session/src/diagnostics.rs +@@ -129,7 +129,7 @@ pub fn add_feature_diagnostics_for_issue( + // We're unlikely to stabilize something out of `rustc_attrs` + // without at least renaming it, so pointing out how old + // the compiler is will do little good. +- } else if sess.opts.unstable_opts.ui_testing { ++ } else if (*sess.opts.unstable_opts.read_ui_testing()) { + err.subdiagnostic(SuggestUpgradeCompiler::ui_testing()); + } else if let Some(suggestion) = SuggestUpgradeCompiler::new() { + err.subdiagnostic(suggestion); +@@ -168,7 +168,7 @@ pub fn feature_err_unstable_feature_bound( + // We're unlikely to stabilize something out of `rustc_attrs` + // without at least renaming it, so pointing out how old + // the compiler is will do little good. +- } else if sess.opts.unstable_opts.ui_testing { ++ } else if (*sess.opts.unstable_opts.read_ui_testing()) { + err.subdiagnostic(SuggestUpgradeCompiler::ui_testing()); + } else if let Some(suggestion) = SuggestUpgradeCompiler::new() { + err.subdiagnostic(suggestion); +diff --git a/compiler/rustc_session/src/lib.rs b/compiler/rustc_session/src/lib.rs +index e3580d36e..b807713e9 100644 +--- a/compiler/rustc_session/src/lib.rs ++++ b/compiler/rustc_session/src/lib.rs +@@ -6,6 +6,7 @@ + #![feature(iter_intersperse)] + #![feature(macro_derive)] + #![feature(macro_metavar_expr)] ++#![feature(macro_metavar_expr_concat)] + #![feature(option_into_flat_iter)] + #![feature(rustc_attrs)] + // To generate CodegenOptionsTargetModifiers and UnstableOptionsTargetModifiers enums +diff --git a/compiler/rustc_session/src/options.rs b/compiler/rustc_session/src/options.rs +index 9a3e74c28..264c4c123 100644 +--- a/compiler/rustc_session/src/options.rs ++++ b/compiler/rustc_session/src/options.rs +@@ -543,6 +543,26 @@ pub struct $struct_name { + )* + } + ++ impl $struct_name { ++ $( ++ /// The option's value. A read of an option the dependency graph does not ++ /// track is reported, for testing incremental compilation ++ /// (`rustc_data_structures::untracked`). ++ #[inline] ++ #[track_caller] ++ #[allow(dead_code)] ++ pub fn ${concat(read_, $opt)}(&self) -> &$t { ++ if stringify!($dep_tracking_marker) == "UNTRACKED" { ++ rustc_data_structures::untracked::untracked_read(concat!( ++ "the untracked option ", ++ stringify!($opt) ++ )); ++ } ++ &self.$opt ++ } ++ )* ++ } ++ + #[derive(PartialEq, Eq, PartialOrd, Ord, Debug, Copy, Clone, Encodable, BlobDecodable)] + pub enum $tmod_enum { + $( +diff --git a/compiler/rustc_session/src/output.rs b/compiler/rustc_session/src/output.rs +index 4824abae6..ba42fc461 100644 +--- a/compiler/rustc_session/src/output.rs ++++ b/compiler/rustc_session/src/output.rs +@@ -98,7 +98,7 @@ pub fn filename_for_input( + crate_name: Symbol, + outputs: &OutputFilenames, + ) -> OutFileName { +- let libname = format!("{}{}", crate_name, sess.opts.cg.extra_filename); ++ let libname = format!("{}{}", crate_name, (*sess.opts.cg.read_extra_filename())); + + match crate_type { + CrateType::Rlib => { +diff --git a/compiler/rustc_session/src/session.rs b/compiler/rustc_session/src/session.rs +index f068aaa58..447a2d9a6 100644 +--- a/compiler/rustc_session/src/session.rs ++++ b/compiler/rustc_session/src/session.rs +@@ -352,11 +352,11 @@ pub fn early_lto(&self) -> LtoCli { + } + + pub fn print_llvm_stats(&self) -> bool { +- self.opts.unstable_opts.print_codegen_stats ++ (*self.opts.unstable_opts.read_print_codegen_stats()) + } + + pub fn print_llvm_stats_json(&self) -> Option<&String> { +- self.opts.unstable_opts.print_codegen_stats_json.as_ref() ++ (*self.opts.unstable_opts.read_print_codegen_stats_json()).as_ref() + } + + pub fn relocation_model(&self) -> RelocModel { +@@ -673,10 +673,10 @@ pub fn create_feature_err<'a>(&'a self, err: impl Diagnostic<'a>, feature: Symbo + /// Record the fact that we called `trimmed_def_paths`, and do some + /// checking about whether its cost was justified. + pub fn record_trimmed_def_paths(&self) { +- if self.opts.unstable_opts.print_type_sizes +- || self.opts.unstable_opts.query_dep_graph +- || self.opts.unstable_opts.dump_mir.is_some() +- || self.opts.unstable_opts.unpretty.is_some() ++ if (*self.opts.unstable_opts.read_print_type_sizes()) ++ || (*self.opts.unstable_opts.read_query_dep_graph()) ++ || (*self.opts.unstable_opts.read_dump_mir()).is_some() ++ || (*self.opts.unstable_opts.read_unpretty()).is_some() + || self.prof.is_args_recording_enabled() + || self.opts.output_types.contains_key(&OutputType::Mir) + || std::env::var_os("RUSTC_LOG").is_some() +@@ -897,7 +897,7 @@ pub fn diagnostic_width(&self) -> usize { + let default_column_width = 140; + if let Some(width) = self.opts.diagnostic_width { + width +- } else if self.opts.unstable_opts.ui_testing { ++ } else if (*self.opts.unstable_opts.read_ui_testing()) { + default_column_width + } else { + termize::dimensions().map_or(default_column_width, |(w, _)| w) +@@ -1070,7 +1070,7 @@ pub fn fewer_names(&self) -> bool { + } + + pub fn unstable_options(&self) -> bool { +- self.opts.unstable_opts.unstable_options ++ (*self.opts.unstable_opts.read_unstable_options()) + } + + pub fn is_nightly_build(&self) -> bool { +@@ -1280,9 +1280,9 @@ pub fn pointer_authentication_init_fini(&self) -> Option<&PointerAuthSchema> { + // JUSTIFICATION: part of session construction + #[allow(rustc::bad_opt_access)] + fn default_emitter(sopts: &config::Options, source_map: Arc) -> Box { +- let macro_backtrace = sopts.unstable_opts.macro_backtrace; +- let track_diagnostics = sopts.unstable_opts.track_diagnostics; +- let terminal_url = match sopts.unstable_opts.terminal_urls { ++ let macro_backtrace = (*sopts.unstable_opts.read_macro_backtrace()); ++ let track_diagnostics = (*sopts.unstable_opts.read_track_diagnostics()); ++ let terminal_url = match (*sopts.unstable_opts.read_terminal_urls()) { + TerminalUrl::Auto => { + match (std::env::var("COLORTERM").as_deref(), std::env::var("TERM").as_deref()) { + (Ok("truecolor"), Ok("xterm-256color")) +@@ -1310,9 +1310,9 @@ fn default_emitter(sopts: &config::Options, source_map: Arc) -> Box Box::new( +@@ -1323,9 +1323,9 @@ fn default_emitter(sopts: &config::Options, source_map: Arc) -> Box Some(Arc::new(profiler)), +@@ -1438,7 +1438,7 @@ pub fn build_session( + + let prof = SelfProfilerRef::new( + self_profiler, +- sopts.unstable_opts.time_passes.then(|| sopts.unstable_opts.time_passes_format), ++ (*sopts.unstable_opts.read_time_passes()).then(|| (*sopts.unstable_opts.read_time_passes_format())), + ); + + let ctfe_backtrace = Lock::new(match env::var("RUSTC_CTFE_BACKTRACE") { +@@ -1759,7 +1759,7 @@ fn validate_commandline_args_with_session_available(sess: &Session) { + } + + if !sess.target.options.supported_split_debuginfo.contains(&sess.split_debuginfo()) +- && !sess.opts.unstable_opts.unstable_options ++ && !(*sess.opts.unstable_opts.read_unstable_options()) + { + sess.dcx().emit_err(diagnostics::SplitDebugInfoUnstablePlatform { + debuginfo: sess.split_debuginfo(), +@@ -1794,7 +1794,7 @@ fn validate_commandline_args_with_session_available(sess: &Session) { + sess.dcx().emit_err(diagnostics::InstrumentationNotSupported { us: "XRay".to_string() }); + } + +- if let Some(flavor) = sess.opts.cg.linker_flavor ++ if let Some(flavor) = (*sess.opts.cg.read_linker_flavor()) + && let Some(compatible_list) = sess.target.linker_flavor.check_compatibility(flavor) + { + let flavor = flavor.desc(); +diff --git a/compiler/rustc_trait_selection/src/error_reporting/infer/mod.rs b/compiler/rustc_trait_selection/src/error_reporting/infer/mod.rs +index b2b57e018..9d88ae348 100644 +--- a/compiler/rustc_trait_selection/src/error_reporting/infer/mod.rs ++++ b/compiler/rustc_trait_selection/src/error_reporting/infer/mod.rs +@@ -2385,7 +2385,7 @@ fn expected_found_str_term( + let exp_s = exp.content(); + let fnd_s = fnd.content(); + if !self.tcx.sess.opts.verbose +- && self.tcx.sess.opts.unstable_opts.write_long_types_to_disk ++ && (*self.tcx.sess.opts.unstable_opts.read_write_long_types_to_disk()) + { + // We aren't explicitly asking for `--verbose` output, and we are storing long + // types to disk, so we try to shorten the output. +diff --git a/compiler/rustc_ty_utils/src/layout.rs b/compiler/rustc_ty_utils/src/layout.rs +index 8d8e50139..bb4978b23 100644 +--- a/compiler/rustc_ty_utils/src/layout.rs ++++ b/compiler/rustc_ty_utils/src/layout.rs +@@ -101,7 +101,7 @@ fn layout_of<'tcx>( + + // If we are running with `-Zprint-type-sizes`, maybe record layouts + // for dumping later. +- if cx.tcx().sess.opts.unstable_opts.print_type_sizes { ++ if (*cx.tcx().sess.opts.unstable_opts.read_print_type_sizes()) { + record_layout_for_printing(&cx, layout); + } + diff --git a/docs/hunt/repro.sh b/docs/hunt/repro.sh index 3b3e424..6730b61 100755 --- a/docs/hunt/repro.sh +++ b/docs/hunt/repro.sh @@ -65,3 +65,18 @@ echo "// a comment at the end" >> "$d/w/m.rs" n1=$("$rustc" --edition 2024 $wf -Cincremental="$d/i9" -o "$d/w/a" "$d/w/main.rs" 2>&1 | grep -c "from the assembler") n2=$("$rustc" --edition 2024 $wf -Cincremental="$d/i10" -o "$d/w/b" "$d/w/main.rs" 2>&1 | grep -c "from the assembler") if [ "$n1" = "$n2" ]; then echo same; else echo "DIFFER (the rebuild shows no warning)"; fi + +echo -n "no-leak-check, a session with -Zno-leak-check then one without, vs a clean build: " +mkdir -p "$d/nl" +cat > "$d/nl/lib.rs" <<'RS' +fn foo(x: for<'a, 'b> fn(&'a u8, &'b u8) -> &'a u8, y: for<'a> fn(&'a u8, &'a u8) -> &'a u8) { + let z = match 22 { + 0 => y, + _ => x, + }; +} +RS +"$rustc" --crate-type lib -Cincremental="$d/i11" -Zno-leak-check --out-dir "$d/nl" "$d/nl/lib.rs" 2> /dev/null +r1=$("$rustc" --crate-type lib -Cincremental="$d/i11" --out-dir "$d/nl" "$d/nl/lib.rs" > /dev/null 2>&1; echo $?) +r2=$("$rustc" --crate-type lib -Cincremental="$d/i12" --out-dir "$d/nl" "$d/nl/lib.rs" > /dev/null 2>&1; echo $?) +if [ "$r1" = "$r2" ]; then echo same; else echo "DIFFER (the rebuild exits $r1, a clean build $r2)"; fi diff --git a/docs/untracked-reads.md b/docs/untracked-reads.md new file mode 100644 index 0000000..d9a5baa --- /dev/null +++ b/docs/untracked-reads.md @@ -0,0 +1,84 @@ +# Reporting untracked reads inside rustc + +Every incremental bug found here so far comes down to one mistake: while computing +something incremental compilation may reuse, rustc read state that the dependency graph +does not track. When that state changes and nothing tracked does, the old result is reused. + +| bug | what was read | while computing | +|---|---|---| +| [stale metadata](hunt/issue-stale-metadata-reuse.md) | source file hashes and lengths | metadata | +| [untracked options](hunt/issue-untracked-options.md) | `-Zno-leak-check`, `-C extra-filename`, `-Zemit-stack-sizes`, ... | type checking, metadata, codegen units | +| [stale debuginfo checksum](hunt/issue-stale-debuginfo-source.md) | `SourceFile::src_hash`, source text | codegen units | + +The [reuse check](shadow-mode.md) catches such a bug once an edit has made a reused result +stale. [`hunt/report-untracked.patch`](hunt/report-untracked.patch) catches the read itself, on +any build. + +## The patch + +Applied after the three fixes and [`verify-reuse.patch`](hunt/verify-reuse.patch); on under +`RUSTC_REPORT_UNTRACKED`. + +- **The hook.** `rustc_data_structures::untracked::untracked_read(what)` is called where + untracked state is read. When reporting is on, it looks at the current task: if the read + happens inside a task whose result can be reused (an ordinary query, metadata, a codegen + unit; not `eval_always`, not ignored), it prints + + ```text + rustc-untracked-read: , read at , while computing ``, whose result incremental compilation may reuse + ``` + + once per distinct line. Each task records its kind for this (`TaskDeps::kind`). +- **Declarations.** A task that tracks the state another way says so with + `rustc_middle::dep_graph::declare_untracked_input(what)`, and its reads of it are not + reported. The fix for the stale metadata declares "source file contents" right after + reading the query that fingerprints the source files; without that fix, the same reads + would be reported. `RUSTC_REPORT_UNTRACKED=all` reports declared reads too, marked + `rustc-untracked-read-declared`. +- **Where it is called:** + - every read of an `[UNTRACKED]` option: `options!` generates a `read_