diff --git a/.github/workflows/test.yml b/.github/workflows/test.yml index 5e18718..5849ef8 100644 --- a/.github/workflows/test.yml +++ b/.github/workflows/test.yml @@ -138,7 +138,9 @@ jobs: set -euo pipefail INSTALL_ROOT="$(mktemp -d)" cargo install --path cli --root "$INSTALL_ROOT" - "$INSTALL_ROOT/bin/code2graph" --help + for CLI_BIN in "$INSTALL_ROOT"/bin/*; do + "$CLI_BIN" --help + done bindings: if: ${{ !inputs.skip_bindings }} diff --git a/README.md b/README.md index 1e2239e..014e0c6 100644 --- a/README.md +++ b/README.md @@ -111,9 +111,11 @@ c2g callers helper c2g impact helper --depth 3 ``` -By default the CLI rejects an incomplete index. `--allow-partial` explicitly permits -publishing and querying a partial source set; inspect the reported omissions before -relying on its results. +By default the CLI rejects an incomplete index. `--allow-partial` explicitly permits publishing and querying a partial source set. Inspect the reported omissions before relying on its results. Each reported omission list is capped at 256 entries. Project metadata retains the full count in `omittedFiles` and full reason totals in `omissionReasons`. Index results use `omitted_files` and `omission_reasons`. Status inventory retains the full count in `inventory.omitted_files`. The `omissionsTruncated` project flag and `omissions_truncated` index flag identify capped lists. + +Without `--root`, the selected project is the working directory. A working directory +that is a home directory or the filesystem root is refused: walking one costs minutes +and describes no project. Name the project (`--root `) to proceed. Driving the CLI from a coding agent: [`docs/agent-integration.md`](docs/agent-integration.md) carries a copy-pasteable rule block for `CLAUDE.md` / `AGENTS.md` and explains why a mechanical trigger is the only kind an agent reliably follows. diff --git a/cli/src/config.rs b/cli/src/config.rs index 302b636..c81b979 100644 --- a/cli/src/config.rs +++ b/cli/src/config.rs @@ -95,6 +95,13 @@ pub const DEFAULT_MAX_TOTAL_BYTES: usize = 256 * 1_024 * 1_024; pub const DEFAULT_MAX_DEPTH: u32 = 32; /// Default number of rows rendered by a command. pub const DEFAULT_LIMIT: usize = 50; +/// Default maximum number of individual omission entries reported in any one +/// list. An over-broad root (a home directory, a parent of many repositories) +/// omits tens of thousands of files, and a JSON envelope carrying one entry per +/// omitted file grows to megabytes. The entry lists are diagnostics: each list +/// is capped to this many entries while the totals (`omittedFiles`, +/// `inventory.omitted_files`) and the rendered reason counts stay complete. +pub const DEFAULT_MAX_OMISSIONS: usize = 256; /// Default reverse-reachability depth for `impact`. pub const DEFAULT_IMPACT_DEPTH: u32 = 2; diff --git a/cli/src/execution/lifecycle.rs b/cli/src/execution/lifecycle.rs index 190af01..5322578 100644 --- a/cli/src/execution/lifecycle.rs +++ b/cli/src/execution/lifecycle.rs @@ -1878,13 +1878,15 @@ fn graph_from_snapshot( }) } -fn project_output( +pub(super) fn project_output( selection: &crate::ProjectSelection, snapshot: &LoadedSnapshot, tier: crate::ResolverTier, freshness: Freshness, cache: CacheDisposition, ) -> ProjectOutput { + let omission_reasons = crate::result::cache_omission_reasons(&snapshot.omissions); + let (omissions, omissions_truncated) = crate::result::capped_omissions(&snapshot.omissions); ProjectOutput { root: selection.canonical_root.to_string_lossy().into_owned(), snapshot: snapshot.candidate_id.to_string(), @@ -1893,7 +1895,9 @@ fn project_output( cache, completeness: snapshot.completeness.into(), omitted_files: snapshot.omissions.len(), - omissions: snapshot.omissions.iter().map(Into::into).collect(), + omissions, + omission_reasons, + omissions_truncated, // Only the paths that actually refreshed against a store can observe a // recovery; they fill this in from the store afterwards. cache_recovery: None, @@ -2675,11 +2679,11 @@ mod tests { // One project root disappears; one cache is left on an older schema. fs::remove_dir_all(temp.path().join("deleted")).expect("remove project"); - let stale_key = crate::cache::CacheLocation::for_project( - Some(cache.as_path()), - &temp.path().join("stale"), - ) - .expect("stale location"); + let stale_root = + fs::canonicalize(temp.path().join(".").join("stale")).expect("canonical stale project"); + let stale_key = + crate::cache::CacheLocation::for_project(Some(cache.as_path()), &stale_root) + .expect("stale location"); let connection = rusqlite::Connection::open(&stale_key.database_path).expect("open stale cache"); connection diff --git a/cli/src/execution/omission_output_tests.rs b/cli/src/execution/omission_output_tests.rs new file mode 100644 index 0000000..512aabc --- /dev/null +++ b/cli/src/execution/omission_output_tests.rs @@ -0,0 +1,175 @@ +// SPDX-License-Identifier: Apache-2.0 + +use crate::cache::{ + CacheCompleteness, CacheOmission, CandidateId, CompatibilityFingerprint, CompatibilityRecord, + LanguageFeatureFingerprint, LoadedSnapshot, PackageFingerprint, ProjectInputDigest, +}; +use crate::config::{DEFAULT_MAX_OMISSIONS, ResourceLimits}; +use crate::result::{CacheReasonCountOutput, IndexOutput, PlanDecisionCountsOutput}; +use crate::{ + CacheDisposition, Freshness, OutputEnvelope, OutputStatus, ProjectSelection, ResolverTier, + SelectionProvenance, StatusOutput, +}; + +use super::super::lifecycle::{CommandOutput, project_output}; +use super::render_human; + +fn mixed_snapshot() -> LoadedSnapshot { + let omissions = (0..265) + .rev() + .map(|index| CacheOmission { + path: format!("src/file{index:04}.rs"), + reason: match index { + 0..250 => "file-count-limit", + 250..260 => "file-too-large", + _ => "read-error:other", + } + .into(), + detail: "resource limit".into(), + }) + .collect::>(); + let language = LanguageFeatureFingerprint::current(); + let package = PackageFingerprint::from_normalized(["test"]); + let compatibility = CompatibilityFingerprint::new(language, package); + let digest = ProjectInputDigest::from_inputs([] as [(&str, &str, [u8; 32]); 0]); + LoadedSnapshot { + candidate_id: CandidateId::new( + compatibility, + digest, + CacheCompleteness::Partial, + &omissions, + ), + compatibility: CompatibilityRecord { + id: compatibility, + language_fingerprint: language, + package_fingerprint: package, + created_at_ns: 1, + }, + input_digest: digest, + completeness: CacheCompleteness::Partial, + omissions, + created_at_ns: 2, + inventory_file_count: 3, + inventory_total_bytes: 42, + files: Vec::new(), + tier_graphs: Vec::new(), + } +} + +#[test] +fn index_and_cached_status_count_reasons_outside_capped_entries() { + let snapshot = mixed_snapshot(); + let expected = vec![ + CacheReasonCountOutput { + reason: "file-count-limit".into(), + count: 250, + }, + CacheReasonCountOutput { + reason: "file-too-large".into(), + count: 10, + }, + CacheReasonCountOutput { + reason: "read-error:other".into(), + count: 5, + }, + ]; + let index = IndexOutput::from_loaded_snapshot( + &snapshot, + ResolverTier::Scope, + 0, + 0, + 0, + 1, + PlanDecisionCountsOutput::default(), + ); + let selection = ProjectSelection { + canonical_root: "/project".into(), + canonical_source: None, + provenance: SelectionProvenance::RootArgument, + }; + let project = project_output( + &selection, + &snapshot, + ResolverTier::Scope, + Freshness::Frozen, + CacheDisposition::Hit, + ); + assert_eq!(index.omission_reasons, expected); + assert_eq!(project.omission_reasons, expected); + assert_eq!(index.omissions.len(), DEFAULT_MAX_OMISSIONS); + assert_eq!(index.omissions, project.omissions); + assert_eq!(index.omitted_files, 265); + assert_eq!(project.omitted_files, 265); + assert!(index.omissions_truncated); + assert!(project.omissions_truncated); + assert!( + index + .omissions + .iter() + .all(|entry| entry.reason != "read-error:other") + ); + assert_eq!(index.omissions[0].path, "src/file0000.rs"); + assert_eq!(index.omissions[255].path, "src/file0255.rs"); + + let status = StatusOutput::from_loaded_snapshot(project, &snapshot, &ResourceLimits::default()); + assert_eq!(status.cached_omissions, index.omissions); + assert_eq!(status.project.omission_reasons, expected); + assert_eq!(status.inventory.omitted_files, 265); + let expected_json = serde_json::json!([ + {"reason": "file-count-limit", "count": 250}, + {"reason": "file-too-large", "count": 10}, + {"reason": "read-error:other", "count": 5}, + ]); + let mut index_json = serde_json::to_value(&index).expect("index JSON"); + let mut project_json = serde_json::to_value(&status.project).expect("project JSON"); + assert_eq!(index_json["omission_reasons"], expected_json); + assert_eq!(project_json["omissionReasons"], expected_json); + index_json + .as_object_mut() + .expect("index object") + .remove("omission_reasons"); + project_json + .as_object_mut() + .expect("project object") + .remove("omissionReasons"); + assert!( + serde_json::from_value::(index_json) + .expect("older index contract") + .omission_reasons + .is_empty() + ); + assert!( + serde_json::from_value::(project_json) + .expect("older project contract") + .omission_reasons + .is_empty() + ); + + for output in [ + CommandOutput::Index(OutputEnvelope::new(OutputStatus::Partial, index)), + CommandOutput::Status(OutputEnvelope::new(OutputStatus::Partial, status)), + ] { + let rendered = render_human(&output); + assert!(rendered.contains("warning: omission entries truncated; listing 256 of 265\n")); + let reasons = rendered + .lines() + .filter(|line| line.starts_with("omission reason=")) + .collect::>(); + assert_eq!( + reasons, + [ + "omission reason=file-count-limit count=250", + "omission reason=file-too-large count=10", + "omission reason=read-error:other count=5", + ] + ); + assert_eq!( + rendered + .lines() + .filter(|line| line.starts_with("omitted src/")) + .count(), + DEFAULT_MAX_OMISSIONS + ); + assert!(!rendered.contains("omitted src/file0260.rs")); + } +} diff --git a/cli/src/execution/output.rs b/cli/src/execution/output.rs index c7727f7..a15d036 100644 --- a/cli/src/execution/output.rs +++ b/cli/src/execution/output.rs @@ -70,6 +70,13 @@ fn query_warning(project: Option<&ProjectOutput>) -> String { "warning: partial snapshot; {} source files omitted\n", project.omitted_files )); + if project.omissions_truncated { + output.push_str(&format!( + "warning: omission entries truncated; listing {} of {}\n", + project.omissions.len(), + project.omitted_files + )); + } for omission in sorted_omissions(&project.omissions) { output.push_str(&format!( "warning: omitted {} reason={} detail={}\n", @@ -107,15 +114,21 @@ fn render_index(envelope: &crate::OutputEnvelope) -> String } output.push_str(&format!( "omitted files={}\n", - envelope.results.omissions.len() + envelope.results.omitted_files )); - let omissions = sorted_omissions(&envelope.results.omissions); - let mut counts = std::collections::BTreeMap::<&str, usize>::new(); - for omission in &omissions { - *counts.entry(&omission.reason).or_default() += 1; + if envelope.results.omissions_truncated { + output.push_str(&format!( + "warning: omission entries truncated; listing {} of {}\n", + envelope.results.omissions.len(), + envelope.results.omitted_files + )); } - for (reason, count) in counts { - output.push_str(&format!("omission reason={} count={}\n", reason, count)); + let omissions = sorted_omissions(&envelope.results.omissions); + for reason in &envelope.results.omission_reasons { + output.push_str(&format!( + "omission reason={} count={}\n", + reason.reason, reason.count + )); } for omission in omissions { output.push_str(&format!( @@ -153,13 +166,19 @@ fn render_status(status: &crate::StatusOutput) -> String { .timeout_millis .map_or_else(|| "none".into(), |value| value.to_string()), ); - let mut counts = std::collections::BTreeMap::<&str, usize>::new(); - let omissions = sorted_omissions(&status.project.omissions); - for omission in &omissions { - *counts.entry(&omission.reason).or_default() += 1; + if status.project.omissions_truncated { + output.push_str(&format!( + "warning: omission entries truncated; listing {} of {}\n", + status.project.omissions.len(), + status.project.omitted_files + )); } - for (reason, count) in counts { - output.push_str(&format!("omission reason={} count={}\n", reason, count)); + let omissions = sorted_omissions(&status.project.omissions); + for reason in &status.project.omission_reasons { + output.push_str(&format!( + "omission reason={} count={}\n", + reason.reason, reason.count + )); } for omission in omissions { output.push_str(&format!( @@ -547,6 +566,17 @@ mod tests { detail: "limit=12".into(), }, ], + omission_reasons: vec![ + crate::result::CacheReasonCountOutput { + reason: "file-too-large".into(), + count: 1, + }, + crate::result::CacheReasonCountOutput { + reason: "read-error:other".into(), + count: 1, + }, + ], + omissions_truncated: false, cache_recovery: None, } } @@ -595,6 +625,10 @@ mod tests { inventory_file_count: 3, inventory_total_bytes: 42, omissions: project(Freshness::Fresh, CacheCompletenessOutput::Partial).omissions, + omitted_files: 2, + omission_reasons: project(Freshness::Fresh, CacheCompletenessOutput::Partial) + .omission_reasons, + omissions_truncated: false, changed: 2, deleted: 1, ignored_omissions: 0, @@ -617,6 +651,7 @@ mod tests { let mut project = project(Freshness::Fresh, CacheCompletenessOutput::Complete); project.omitted_files = 0; project.omissions = Vec::new(); + project.omission_reasons = Vec::new(); project.cache_recovery = Some(detail.into()); let mut envelope = OutputEnvelope::new( @@ -629,6 +664,9 @@ mod tests { inventory_file_count: 1, inventory_total_bytes: 42, omissions: Vec::new(), + omitted_files: 0, + omission_reasons: Vec::new(), + omissions_truncated: false, changed: 1, deleted: 0, ignored_omissions: 0, @@ -703,3 +741,7 @@ mod tests { assert_eq!(render_human(&CommandOutput::Impact(impact)), expected); } } + +#[cfg(test)] +#[path = "omission_output_tests.rs"] +mod omission_output_tests; diff --git a/cli/src/lib.rs b/cli/src/lib.rs index fa681ea..d37d344 100644 --- a/cli/src/lib.rs +++ b/cli/src/lib.rs @@ -54,14 +54,14 @@ pub use refresh::{ pub use request::{CacheOp, CliRequest, CommandRequest, Selector, SourcePosition}; pub use result::{ CacheClearScope, CacheCompletenessOutput, CacheDetail, CacheDisposition, CacheOmissionOutput, - CacheProjectOutput, CacheProjectState, CacheReport, CacheSnapshotOutput, ConfidenceOutput, - ErrorEnvelope, Freshness, ImpactOutput, IndexOutput, InventoryCompletenessOutput, - InventoryOmissionReasonOutput, InventoryReasonCountOutput, InventorySummaryOutput, - ModuleDependencyOutput, ModuleDependencyTargetOutput, OUTPUT_SCHEMA_VERSION, OccurrenceOutput, - OutputEnvelope, OutputStatus, PlanDecisionCountsOutput, ProjectOutput, ProvenanceOutput, - RefRoleOutput, ReferenceOutput, RelationOutput, SelectorOutput, StableIoErrorOutput, - StatusOutput, SymbolKindOutput, SymbolOutput, TypeRefContextOutput, success_exit_code, - success_status, + CacheProjectOutput, CacheProjectState, CacheReasonCountOutput, CacheReport, + CacheSnapshotOutput, ConfidenceOutput, ErrorEnvelope, Freshness, ImpactOutput, IndexOutput, + InventoryCompletenessOutput, InventoryOmissionReasonOutput, InventoryReasonCountOutput, + InventorySummaryOutput, ModuleDependencyOutput, ModuleDependencyTargetOutput, + OUTPUT_SCHEMA_VERSION, OccurrenceOutput, OutputEnvelope, OutputStatus, + PlanDecisionCountsOutput, ProjectOutput, ProvenanceOutput, RefRoleOutput, ReferenceOutput, + RelationOutput, SelectorOutput, StableIoErrorOutput, StatusOutput, SymbolKindOutput, + SymbolOutput, TypeRefContextOutput, success_exit_code, success_status, }; pub use selector::{ SelectorContext, SelectorOptions, SelectorPurpose, SelectorRequest, SelectorResolution, diff --git a/cli/src/project/select.rs b/cli/src/project/select.rs index 5fc885c..fe03bbe 100644 --- a/cli/src/project/select.rs +++ b/cli/src/project/select.rs @@ -33,6 +33,14 @@ pub struct ProjectSelection { /// An explicit index file may be outside `cwd`; marker discovery climbs from its /// parent through the filesystem root. pub fn select_project(request: &CliRequest, cwd: &Path) -> Result { + select_project_with_home(request, cwd, home_directory) +} + +fn select_project_with_home( + request: &CliRequest, + cwd: &Path, + home_directory: impl FnOnce() -> Option, +) -> Result { let cwd = validated_cwd(cwd)?; if let Some(root) = &request.global.root { @@ -68,7 +76,33 @@ pub fn select_project(request: &CliRequest, cwd: &Path) -> Result) -> bool { + path.parent().is_none() || home.is_some_and(|home| path == home) +} + +fn home_directory() -> Option { + directories::BaseDirs::new().map(|dirs| dirs.home_dir().to_path_buf()) +} + +fn select_implicit_directory(cwd: &ValidatedCwd, home: Option<&Path>) -> Result { + let home = home.and_then(|path| fs::canonicalize(path).ok()); + if is_forbidden_default_root(&cwd.canonical, home.as_deref()) { + return Err(CliError::ProjectPath { + path: cwd.canonical.clone(), + reason: "refusing the current directory as an implicit project root \ + (home or filesystem root); pass --root " + .into(), + }); + } + select_directory(&cwd.canonical, cwd, SelectionProvenance::CurrentDirectory) } struct ValidatedCwd { @@ -222,7 +256,9 @@ mod tests { #[cfg(target_os = "macos")] use super::is_trusted_system_ancestor; - use super::{SelectionProvenance, select_project}; + use super::{ + SelectionProvenance, is_forbidden_default_root, select_project, select_project_with_home, + }; use crate::config::GlobalOptions; use crate::error::CliError; use crate::request::{CliRequest, CommandRequest}; @@ -355,6 +391,81 @@ mod tests { ); } + #[test] + fn implicit_cwd_root_refuses_home_and_filesystem_root_only() { + let home = Path::new("/home/example"); + assert!(is_forbidden_default_root(Path::new("/"), Some(home))); + assert!(is_forbidden_default_root(Path::new("/"), None)); + assert!(is_forbidden_default_root(home, Some(home))); + assert!(!is_forbidden_default_root( + Path::new("/home/example/project"), + Some(home) + )); + assert!(!is_forbidden_default_root(home, None)); + } + + #[test] + fn implicit_selection_refuses_canonical_home_and_allows_explicit_home() { + let directory = tempdir().expect("temporary directory"); + let home = directory.path().join("home"); + let child = home.join("child"); + fs::create_dir_all(&child).expect("home directories"); + let canonical_home = fs::canonicalize(&home).expect("canonical home"); + let discovered_home = home.join("child").join(".."); + + match select_project_with_home(&request(None, None), &home, || { + Some(discovered_home.clone()) + }) { + Err(CliError::ProjectPath { path, reason }) => { + assert_eq!(path, canonical_home); + assert!(reason.contains("pass --root ")); + } + result => panic!("implicit home selection must be refused, got {result:?}"), + } + + let selection = select_project_with_home(&request(Some(&home), None), &home, || { + panic!("explicit root must skip home discovery") + }) + .expect("explicit home selection"); + assert_eq!(selection.canonical_root, canonical_home); + assert_eq!(selection.provenance, SelectionProvenance::RootArgument); + + let selection = + select_project_with_home(&request(None, None), &child, || Some(discovered_home)) + .expect("implicit child selection"); + assert_eq!( + selection.canonical_root, + fs::canonicalize(&child).expect("canonical child") + ); + assert_eq!(selection.provenance, SelectionProvenance::CurrentDirectory); + } + + #[cfg(windows)] + #[test] + fn implicit_selection_matches_verbatim_cwd_to_unprefixed_home() { + use std::path::{Component, Prefix}; + + let directory = tempdir().expect("temporary directory"); + let canonical_home = fs::canonicalize(directory.path()).expect("canonical home"); + assert!(matches!( + canonical_home.components().next(), + Some(Component::Prefix(prefix)) if matches!(prefix.kind(), Prefix::VerbatimDisk(_)) + )); + let canonical_text = canonical_home.to_str().expect("temporary path text"); + let home = Path::new( + canonical_text + .strip_prefix(r"\\?\") + .expect("verbatim disk prefix"), + ); + assert!(home.is_dir()); + assert!(matches!( + select_project_with_home(&request(None, None), &canonical_home, || { + Some(home.to_path_buf()) + }), + Err(CliError::ProjectPath { path, .. }) if path == canonical_home + )); + } + #[test] fn rejects_invalid_cwd_before_other_selection_inputs() { let directory = tempdir().expect("temporary directory"); diff --git a/cli/src/result.rs b/cli/src/result.rs index 17fc6c5..bb0d3c1 100644 --- a/cli/src/result.rs +++ b/cli/src/result.rs @@ -3,6 +3,10 @@ use code2graph::{Confidence, Provenance, RefRole, SymbolId, SymbolKind, TypeRefContext}; use serde::{Deserialize, Serialize}; +mod omissions; +pub(crate) use omissions::cache_omission_reasons; +pub use omissions::{CacheReasonCountOutput, capped_omissions}; + use crate::cache::{CacheCompleteness, CacheOmission, LoadedSnapshot}; use crate::config::{ResolverTier, ResourceLimits}; use crate::exit::ExitCode; @@ -92,7 +96,19 @@ pub struct ProjectOutput { pub completeness: CacheCompletenessOutput, #[serde(rename = "omittedFiles")] pub omitted_files: usize, + /// Capped to [`crate::config::DEFAULT_MAX_OMISSIONS`] entries; `omittedFiles` carries the total. pub omissions: Vec, + /// Full snapshot counts, independent of the capped entry list. + #[serde(rename = "omissionReasons", default)] + pub omission_reasons: Vec, + /// Present only when `omissions` was capped, so a consumer can tell a short + /// list from a complete one. + #[serde( + rename = "omissionsTruncated", + default, + skip_serializing_if = "is_false" + )] + pub omissions_truncated: bool, /// Why a previously cached snapshot was discarded and rebuilt, when that /// happened during this run. A cache whose stored facts no longer satisfy /// their validation contract — after an upgrade changes that contract, say @@ -509,6 +525,12 @@ impl From<&CacheOmission> for CacheOmissionOutput { } } +/// `skip_serializing_if` for the additive truncation flags: an untruncated +/// envelope keeps the exact spelling it had before the flag existed. +const fn is_false(value: &bool) -> bool { + !*value +} + /// Counts of decisions made by the refresh planner. #[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)] pub struct PlanDecisionCountsOutput { @@ -544,7 +566,17 @@ pub struct IndexOutput { pub completeness: CacheCompletenessOutput, pub inventory_file_count: u64, pub inventory_total_bytes: u64, + /// Total extracted-and-omitted files, independent of `omissions` being capped. + #[serde(default)] + pub omitted_files: usize, + /// Capped to [`crate::config::DEFAULT_MAX_OMISSIONS`] entries; `omitted_files` carries the total. pub omissions: Vec, + /// Full snapshot counts, independent of the capped entry list. + #[serde(default)] + pub omission_reasons: Vec, + /// Present only when `omissions` was capped. + #[serde(default, skip_serializing_if = "is_false")] + pub omissions_truncated: bool, pub changed: usize, pub deleted: usize, pub ignored_omissions: usize, @@ -563,6 +595,8 @@ impl IndexOutput { attempts: u8, plan_decisions: PlanDecisionCountsOutput, ) -> Self { + let omission_reasons = cache_omission_reasons(&snapshot.omissions); + let (omissions, omissions_truncated) = capped_omissions(&snapshot.omissions); Self { candidate: snapshot.candidate_id.to_string(), snapshot: snapshot.candidate_id.to_string(), @@ -570,7 +604,10 @@ impl IndexOutput { completeness: snapshot.completeness.into(), inventory_file_count: snapshot.inventory_file_count, inventory_total_bytes: snapshot.inventory_total_bytes, - omissions: snapshot.omissions.iter().map(Into::into).collect(), + omitted_files: snapshot.omissions.len(), + omissions, + omission_reasons, + omissions_truncated, changed, deleted, ignored_omissions, @@ -618,7 +655,10 @@ impl StatusOutput { omitted_files: snapshot.omissions.len(), omission_reasons: Vec::new(), }, - cached_omissions: snapshot.omissions.iter().map(Into::into).collect(), + // The cached entries mirror `project.omissions` and are capped the + // same way; `project.omitted_files` carries the full count and + // `project.omissions_truncated` says whether either list is short. + cached_omissions: capped_omissions(&snapshot.omissions).0, max_files: limits.max_files, max_file_bytes: limits.max_file_bytes, max_total_bytes: limits.max_total_bytes, @@ -798,6 +838,7 @@ impl ErrorEnvelope { #[cfg(test)] mod tests { use super::*; + use crate::config::DEFAULT_MAX_OMISSIONS; fn loaded_snapshot(completeness: CacheCompleteness) -> LoadedSnapshot { let language = crate::cache::LanguageFeatureFingerprint::current(); @@ -909,10 +950,32 @@ mod tests { completeness: snapshot.completeness.into(), omitted_files: snapshot.omissions.len(), omissions: snapshot.omissions.iter().map(Into::into).collect(), + omission_reasons: cache_omission_reasons(&snapshot.omissions), + omissions_truncated: false, cache_recovery: None, } } + #[test] + fn omission_entry_lists_are_capped_while_the_truncation_is_reported() { + let omissions = (0..(DEFAULT_MAX_OMISSIONS + 5)) + .map(|index| CacheOmission { + path: format!("src/file{index:04}.rs"), + reason: "file-count-limit".into(), + detail: "limit=10000".into(), + }) + .collect::>(); + + let (reported, truncated) = capped_omissions(&omissions); + assert_eq!(reported.len(), DEFAULT_MAX_OMISSIONS); + assert!(truncated); + assert_eq!(reported[0].path, "src/file0000.rs"); + + let (short, truncated) = capped_omissions(&omissions[..3]); + assert_eq!(short.len(), 3); + assert!(!truncated); + } + #[test] fn index_output_and_cached_status_are_owned_stable_contracts() { let snapshot = loaded_snapshot(CacheCompleteness::Partial); @@ -962,6 +1025,11 @@ mod tests { reason: "file-too-large".into(), detail: "limit=1024".into(), }], + omitted_files: 1, + omission_reasons: cache_omission_reasons( + &loaded_snapshot(CacheCompleteness::Partial).omissions, + ), + omissions_truncated: false, changed: 2, deleted: 1, ignored_omissions: 4, @@ -986,6 +1054,8 @@ mod tests { "omissions": [{ "path": "src/large.rs", "reason": "file-too-large", "detail": "limit=1024" }], + "omitted_files": 1, + "omission_reasons": [{"reason": "file-too-large", "count": 1}], "changed": 2, "deleted": 1, "ignored_omissions": 4, @@ -1009,6 +1079,7 @@ mod tests { "root": "/project", "snapshot": "snapshot", "tier": "scope", "freshness": "stale", "cache": "hit", "completeness": "partial", "omittedFiles": 1, + "omissionReasons": [{"reason": "file-too-large", "count": 1}], "omissions": [{ "path": "src/large.rs", "reason": "file-too-large", "detail": "limit=1024" }] @@ -1045,6 +1116,7 @@ mod tests { "root": "/project", "snapshot": "snapshot", "tier": "scope", "freshness": spelling, "cache": "hit", "completeness": "partial", "omittedFiles": 1, + "omissionReasons": [{"reason": "file-too-large", "count": 1}], "omissions": [{ "path": "src/large.rs", "reason": "file-too-large", "detail": "limit=1024" }] @@ -1060,13 +1132,15 @@ mod tests { completeness: CacheCompletenessOutput::Complete, omitted_files: 0, omissions: Vec::new(), + omission_reasons: Vec::new(), + omissions_truncated: false, cache_recovery: None, }; assert_eq!( serde_json::to_value(complete).unwrap(), serde_json::json!({ "root": "/project", "snapshot": "snapshot", "tier": "scope", "freshness": "fresh", - "cache": "hit", "completeness": "complete", "omittedFiles": 0, "omissions": [] + "cache": "hit", "completeness": "complete", "omittedFiles": 0, "omissions": [], "omissionReasons": [] }) ); } @@ -1132,6 +1206,8 @@ mod tests { completeness: snapshot.completeness.into(), omitted_files: snapshot.omissions.len(), omissions: snapshot.omissions.iter().map(Into::into).collect(), + omission_reasons: cache_omission_reasons(&snapshot.omissions), + omissions_truncated: false, cache_recovery: None, }, &snapshot, diff --git a/cli/src/result/omissions.rs b/cli/src/result/omissions.rs new file mode 100644 index 0000000..5af545e --- /dev/null +++ b/cli/src/result/omissions.rs @@ -0,0 +1,47 @@ +// SPDX-License-Identifier: Apache-2.0 + +use std::collections::BTreeMap; + +use serde::{Deserialize, Serialize}; + +use crate::cache::CacheOmission; +use crate::config::DEFAULT_MAX_OMISSIONS; + +use super::CacheOmissionOutput; + +/// One persisted omission-reason count, ordered by its verbatim cache reason. +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub struct CacheReasonCountOutput { + pub reason: String, + pub count: usize, +} + +/// Counts every cached omission before callers cap diagnostic entries. +pub(crate) fn cache_omission_reasons(omissions: &[CacheOmission]) -> Vec { + let mut counts = BTreeMap::<&str, usize>::new(); + for omission in omissions { + *counts.entry(&omission.reason).or_default() += 1; + } + counts + .into_iter() + .map(|(reason, count)| CacheReasonCountOutput { + reason: reason.to_owned(), + count, + }) + .collect() +} + +/// Deterministically ordered, capped view of an omission list. +/// +/// Returns the reported entries (at most [`DEFAULT_MAX_OMISSIONS`]) and whether +/// entries were held back. Callers keep the full total in their own count field, +/// so capping the entry list never hides how many files were omitted. +pub fn capped_omissions(omissions: &[CacheOmission]) -> (Vec, bool) { + let mut sorted = omissions.iter().collect::>(); + sorted.sort_by(|left, right| { + (&left.path, &left.reason, &left.detail).cmp(&(&right.path, &right.reason, &right.detail)) + }); + let truncated = sorted.len() > DEFAULT_MAX_OMISSIONS; + sorted.truncate(DEFAULT_MAX_OMISSIONS); + (sorted.into_iter().map(Into::into).collect(), truncated) +} diff --git a/docs/agent-integration.md b/docs/agent-integration.md index 28d3893..2035616 100644 --- a/docs/agent-integration.md +++ b/docs/agent-integration.md @@ -32,6 +32,9 @@ strings, config values, comments, error text, non-source files, unsupported lang - ALWAYS pass `--allow-partial`: real codebases have files that fail extraction, and without it any such file aborts the command. +- ALWAYS pass `--root`: the implicit root is the working directory, and a home directory + or filesystem root is refused because walking one costs minutes and describes no + project. - `--root` a single package for tight results, or the workspace root for cross-package questions. `--json` for machine-readable output. - `--tier scope` (default) is precise; `--tier name` is recall-first; `--tier dense`