diff --git a/.github/workflows/test.yml b/.github/workflows/test.yml
index 5e18718..5849ef8 100644
--- a/.github/workflows/test.yml
+++ b/.github/workflows/test.yml
@@ -138,7 +138,9 @@ jobs:
set -euo pipefail
INSTALL_ROOT="$(mktemp -d)"
cargo install --path cli --root "$INSTALL_ROOT"
- "$INSTALL_ROOT/bin/code2graph" --help
+ for CLI_BIN in "$INSTALL_ROOT"/bin/*; do
+ "$CLI_BIN" --help
+ done
bindings:
if: ${{ !inputs.skip_bindings }}
diff --git a/README.md b/README.md
index 1e2239e..014e0c6 100644
--- a/README.md
+++ b/README.md
@@ -111,9 +111,11 @@ c2g callers helper
c2g impact helper --depth 3
```
-By default the CLI rejects an incomplete index. `--allow-partial` explicitly permits
-publishing and querying a partial source set; inspect the reported omissions before
-relying on its results.
+By default the CLI rejects an incomplete index. `--allow-partial` explicitly permits publishing and querying a partial source set. Inspect the reported omissions before relying on its results. Each reported omission list is capped at 256 entries. Project metadata retains the full count in `omittedFiles` and full reason totals in `omissionReasons`. Index results use `omitted_files` and `omission_reasons`. Status inventory retains the full count in `inventory.omitted_files`. The `omissionsTruncated` project flag and `omissions_truncated` index flag identify capped lists.
+
+Without `--root`, the selected project is the working directory. A working directory
+that is a home directory or the filesystem root is refused: walking one costs minutes
+and describes no project. Name the project (`--root
`) to proceed.
Driving the CLI from a coding agent: [`docs/agent-integration.md`](docs/agent-integration.md) carries a copy-pasteable rule block for `CLAUDE.md` / `AGENTS.md` and explains why a mechanical trigger is the only kind an agent reliably follows.
diff --git a/cli/src/config.rs b/cli/src/config.rs
index 302b636..c81b979 100644
--- a/cli/src/config.rs
+++ b/cli/src/config.rs
@@ -95,6 +95,13 @@ pub const DEFAULT_MAX_TOTAL_BYTES: usize = 256 * 1_024 * 1_024;
pub const DEFAULT_MAX_DEPTH: u32 = 32;
/// Default number of rows rendered by a command.
pub const DEFAULT_LIMIT: usize = 50;
+/// Default maximum number of individual omission entries reported in any one
+/// list. An over-broad root (a home directory, a parent of many repositories)
+/// omits tens of thousands of files, and a JSON envelope carrying one entry per
+/// omitted file grows to megabytes. The entry lists are diagnostics: each list
+/// is capped to this many entries while the totals (`omittedFiles`,
+/// `inventory.omitted_files`) and the rendered reason counts stay complete.
+pub const DEFAULT_MAX_OMISSIONS: usize = 256;
/// Default reverse-reachability depth for `impact`.
pub const DEFAULT_IMPACT_DEPTH: u32 = 2;
diff --git a/cli/src/execution/lifecycle.rs b/cli/src/execution/lifecycle.rs
index 190af01..5322578 100644
--- a/cli/src/execution/lifecycle.rs
+++ b/cli/src/execution/lifecycle.rs
@@ -1878,13 +1878,15 @@ fn graph_from_snapshot(
})
}
-fn project_output(
+pub(super) fn project_output(
selection: &crate::ProjectSelection,
snapshot: &LoadedSnapshot,
tier: crate::ResolverTier,
freshness: Freshness,
cache: CacheDisposition,
) -> ProjectOutput {
+ let omission_reasons = crate::result::cache_omission_reasons(&snapshot.omissions);
+ let (omissions, omissions_truncated) = crate::result::capped_omissions(&snapshot.omissions);
ProjectOutput {
root: selection.canonical_root.to_string_lossy().into_owned(),
snapshot: snapshot.candidate_id.to_string(),
@@ -1893,7 +1895,9 @@ fn project_output(
cache,
completeness: snapshot.completeness.into(),
omitted_files: snapshot.omissions.len(),
- omissions: snapshot.omissions.iter().map(Into::into).collect(),
+ omissions,
+ omission_reasons,
+ omissions_truncated,
// Only the paths that actually refreshed against a store can observe a
// recovery; they fill this in from the store afterwards.
cache_recovery: None,
@@ -2675,11 +2679,11 @@ mod tests {
// One project root disappears; one cache is left on an older schema.
fs::remove_dir_all(temp.path().join("deleted")).expect("remove project");
- let stale_key = crate::cache::CacheLocation::for_project(
- Some(cache.as_path()),
- &temp.path().join("stale"),
- )
- .expect("stale location");
+ let stale_root =
+ fs::canonicalize(temp.path().join(".").join("stale")).expect("canonical stale project");
+ let stale_key =
+ crate::cache::CacheLocation::for_project(Some(cache.as_path()), &stale_root)
+ .expect("stale location");
let connection =
rusqlite::Connection::open(&stale_key.database_path).expect("open stale cache");
connection
diff --git a/cli/src/execution/omission_output_tests.rs b/cli/src/execution/omission_output_tests.rs
new file mode 100644
index 0000000..512aabc
--- /dev/null
+++ b/cli/src/execution/omission_output_tests.rs
@@ -0,0 +1,175 @@
+// SPDX-License-Identifier: Apache-2.0
+
+use crate::cache::{
+ CacheCompleteness, CacheOmission, CandidateId, CompatibilityFingerprint, CompatibilityRecord,
+ LanguageFeatureFingerprint, LoadedSnapshot, PackageFingerprint, ProjectInputDigest,
+};
+use crate::config::{DEFAULT_MAX_OMISSIONS, ResourceLimits};
+use crate::result::{CacheReasonCountOutput, IndexOutput, PlanDecisionCountsOutput};
+use crate::{
+ CacheDisposition, Freshness, OutputEnvelope, OutputStatus, ProjectSelection, ResolverTier,
+ SelectionProvenance, StatusOutput,
+};
+
+use super::super::lifecycle::{CommandOutput, project_output};
+use super::render_human;
+
+fn mixed_snapshot() -> LoadedSnapshot {
+ let omissions = (0..265)
+ .rev()
+ .map(|index| CacheOmission {
+ path: format!("src/file{index:04}.rs"),
+ reason: match index {
+ 0..250 => "file-count-limit",
+ 250..260 => "file-too-large",
+ _ => "read-error:other",
+ }
+ .into(),
+ detail: "resource limit".into(),
+ })
+ .collect::>();
+ let language = LanguageFeatureFingerprint::current();
+ let package = PackageFingerprint::from_normalized(["test"]);
+ let compatibility = CompatibilityFingerprint::new(language, package);
+ let digest = ProjectInputDigest::from_inputs([] as [(&str, &str, [u8; 32]); 0]);
+ LoadedSnapshot {
+ candidate_id: CandidateId::new(
+ compatibility,
+ digest,
+ CacheCompleteness::Partial,
+ &omissions,
+ ),
+ compatibility: CompatibilityRecord {
+ id: compatibility,
+ language_fingerprint: language,
+ package_fingerprint: package,
+ created_at_ns: 1,
+ },
+ input_digest: digest,
+ completeness: CacheCompleteness::Partial,
+ omissions,
+ created_at_ns: 2,
+ inventory_file_count: 3,
+ inventory_total_bytes: 42,
+ files: Vec::new(),
+ tier_graphs: Vec::new(),
+ }
+}
+
+#[test]
+fn index_and_cached_status_count_reasons_outside_capped_entries() {
+ let snapshot = mixed_snapshot();
+ let expected = vec![
+ CacheReasonCountOutput {
+ reason: "file-count-limit".into(),
+ count: 250,
+ },
+ CacheReasonCountOutput {
+ reason: "file-too-large".into(),
+ count: 10,
+ },
+ CacheReasonCountOutput {
+ reason: "read-error:other".into(),
+ count: 5,
+ },
+ ];
+ let index = IndexOutput::from_loaded_snapshot(
+ &snapshot,
+ ResolverTier::Scope,
+ 0,
+ 0,
+ 0,
+ 1,
+ PlanDecisionCountsOutput::default(),
+ );
+ let selection = ProjectSelection {
+ canonical_root: "/project".into(),
+ canonical_source: None,
+ provenance: SelectionProvenance::RootArgument,
+ };
+ let project = project_output(
+ &selection,
+ &snapshot,
+ ResolverTier::Scope,
+ Freshness::Frozen,
+ CacheDisposition::Hit,
+ );
+ assert_eq!(index.omission_reasons, expected);
+ assert_eq!(project.omission_reasons, expected);
+ assert_eq!(index.omissions.len(), DEFAULT_MAX_OMISSIONS);
+ assert_eq!(index.omissions, project.omissions);
+ assert_eq!(index.omitted_files, 265);
+ assert_eq!(project.omitted_files, 265);
+ assert!(index.omissions_truncated);
+ assert!(project.omissions_truncated);
+ assert!(
+ index
+ .omissions
+ .iter()
+ .all(|entry| entry.reason != "read-error:other")
+ );
+ assert_eq!(index.omissions[0].path, "src/file0000.rs");
+ assert_eq!(index.omissions[255].path, "src/file0255.rs");
+
+ let status = StatusOutput::from_loaded_snapshot(project, &snapshot, &ResourceLimits::default());
+ assert_eq!(status.cached_omissions, index.omissions);
+ assert_eq!(status.project.omission_reasons, expected);
+ assert_eq!(status.inventory.omitted_files, 265);
+ let expected_json = serde_json::json!([
+ {"reason": "file-count-limit", "count": 250},
+ {"reason": "file-too-large", "count": 10},
+ {"reason": "read-error:other", "count": 5},
+ ]);
+ let mut index_json = serde_json::to_value(&index).expect("index JSON");
+ let mut project_json = serde_json::to_value(&status.project).expect("project JSON");
+ assert_eq!(index_json["omission_reasons"], expected_json);
+ assert_eq!(project_json["omissionReasons"], expected_json);
+ index_json
+ .as_object_mut()
+ .expect("index object")
+ .remove("omission_reasons");
+ project_json
+ .as_object_mut()
+ .expect("project object")
+ .remove("omissionReasons");
+ assert!(
+ serde_json::from_value::(index_json)
+ .expect("older index contract")
+ .omission_reasons
+ .is_empty()
+ );
+ assert!(
+ serde_json::from_value::(project_json)
+ .expect("older project contract")
+ .omission_reasons
+ .is_empty()
+ );
+
+ for output in [
+ CommandOutput::Index(OutputEnvelope::new(OutputStatus::Partial, index)),
+ CommandOutput::Status(OutputEnvelope::new(OutputStatus::Partial, status)),
+ ] {
+ let rendered = render_human(&output);
+ assert!(rendered.contains("warning: omission entries truncated; listing 256 of 265\n"));
+ let reasons = rendered
+ .lines()
+ .filter(|line| line.starts_with("omission reason="))
+ .collect::>();
+ assert_eq!(
+ reasons,
+ [
+ "omission reason=file-count-limit count=250",
+ "omission reason=file-too-large count=10",
+ "omission reason=read-error:other count=5",
+ ]
+ );
+ assert_eq!(
+ rendered
+ .lines()
+ .filter(|line| line.starts_with("omitted src/"))
+ .count(),
+ DEFAULT_MAX_OMISSIONS
+ );
+ assert!(!rendered.contains("omitted src/file0260.rs"));
+ }
+}
diff --git a/cli/src/execution/output.rs b/cli/src/execution/output.rs
index c7727f7..a15d036 100644
--- a/cli/src/execution/output.rs
+++ b/cli/src/execution/output.rs
@@ -70,6 +70,13 @@ fn query_warning(project: Option<&ProjectOutput>) -> String {
"warning: partial snapshot; {} source files omitted\n",
project.omitted_files
));
+ if project.omissions_truncated {
+ output.push_str(&format!(
+ "warning: omission entries truncated; listing {} of {}\n",
+ project.omissions.len(),
+ project.omitted_files
+ ));
+ }
for omission in sorted_omissions(&project.omissions) {
output.push_str(&format!(
"warning: omitted {} reason={} detail={}\n",
@@ -107,15 +114,21 @@ fn render_index(envelope: &crate::OutputEnvelope) -> String
}
output.push_str(&format!(
"omitted files={}\n",
- envelope.results.omissions.len()
+ envelope.results.omitted_files
));
- let omissions = sorted_omissions(&envelope.results.omissions);
- let mut counts = std::collections::BTreeMap::<&str, usize>::new();
- for omission in &omissions {
- *counts.entry(&omission.reason).or_default() += 1;
+ if envelope.results.omissions_truncated {
+ output.push_str(&format!(
+ "warning: omission entries truncated; listing {} of {}\n",
+ envelope.results.omissions.len(),
+ envelope.results.omitted_files
+ ));
}
- for (reason, count) in counts {
- output.push_str(&format!("omission reason={} count={}\n", reason, count));
+ let omissions = sorted_omissions(&envelope.results.omissions);
+ for reason in &envelope.results.omission_reasons {
+ output.push_str(&format!(
+ "omission reason={} count={}\n",
+ reason.reason, reason.count
+ ));
}
for omission in omissions {
output.push_str(&format!(
@@ -153,13 +166,19 @@ fn render_status(status: &crate::StatusOutput) -> String {
.timeout_millis
.map_or_else(|| "none".into(), |value| value.to_string()),
);
- let mut counts = std::collections::BTreeMap::<&str, usize>::new();
- let omissions = sorted_omissions(&status.project.omissions);
- for omission in &omissions {
- *counts.entry(&omission.reason).or_default() += 1;
+ if status.project.omissions_truncated {
+ output.push_str(&format!(
+ "warning: omission entries truncated; listing {} of {}\n",
+ status.project.omissions.len(),
+ status.project.omitted_files
+ ));
}
- for (reason, count) in counts {
- output.push_str(&format!("omission reason={} count={}\n", reason, count));
+ let omissions = sorted_omissions(&status.project.omissions);
+ for reason in &status.project.omission_reasons {
+ output.push_str(&format!(
+ "omission reason={} count={}\n",
+ reason.reason, reason.count
+ ));
}
for omission in omissions {
output.push_str(&format!(
@@ -547,6 +566,17 @@ mod tests {
detail: "limit=12".into(),
},
],
+ omission_reasons: vec![
+ crate::result::CacheReasonCountOutput {
+ reason: "file-too-large".into(),
+ count: 1,
+ },
+ crate::result::CacheReasonCountOutput {
+ reason: "read-error:other".into(),
+ count: 1,
+ },
+ ],
+ omissions_truncated: false,
cache_recovery: None,
}
}
@@ -595,6 +625,10 @@ mod tests {
inventory_file_count: 3,
inventory_total_bytes: 42,
omissions: project(Freshness::Fresh, CacheCompletenessOutput::Partial).omissions,
+ omitted_files: 2,
+ omission_reasons: project(Freshness::Fresh, CacheCompletenessOutput::Partial)
+ .omission_reasons,
+ omissions_truncated: false,
changed: 2,
deleted: 1,
ignored_omissions: 0,
@@ -617,6 +651,7 @@ mod tests {
let mut project = project(Freshness::Fresh, CacheCompletenessOutput::Complete);
project.omitted_files = 0;
project.omissions = Vec::new();
+ project.omission_reasons = Vec::new();
project.cache_recovery = Some(detail.into());
let mut envelope = OutputEnvelope::new(
@@ -629,6 +664,9 @@ mod tests {
inventory_file_count: 1,
inventory_total_bytes: 42,
omissions: Vec::new(),
+ omitted_files: 0,
+ omission_reasons: Vec::new(),
+ omissions_truncated: false,
changed: 1,
deleted: 0,
ignored_omissions: 0,
@@ -703,3 +741,7 @@ mod tests {
assert_eq!(render_human(&CommandOutput::Impact(impact)), expected);
}
}
+
+#[cfg(test)]
+#[path = "omission_output_tests.rs"]
+mod omission_output_tests;
diff --git a/cli/src/lib.rs b/cli/src/lib.rs
index fa681ea..d37d344 100644
--- a/cli/src/lib.rs
+++ b/cli/src/lib.rs
@@ -54,14 +54,14 @@ pub use refresh::{
pub use request::{CacheOp, CliRequest, CommandRequest, Selector, SourcePosition};
pub use result::{
CacheClearScope, CacheCompletenessOutput, CacheDetail, CacheDisposition, CacheOmissionOutput,
- CacheProjectOutput, CacheProjectState, CacheReport, CacheSnapshotOutput, ConfidenceOutput,
- ErrorEnvelope, Freshness, ImpactOutput, IndexOutput, InventoryCompletenessOutput,
- InventoryOmissionReasonOutput, InventoryReasonCountOutput, InventorySummaryOutput,
- ModuleDependencyOutput, ModuleDependencyTargetOutput, OUTPUT_SCHEMA_VERSION, OccurrenceOutput,
- OutputEnvelope, OutputStatus, PlanDecisionCountsOutput, ProjectOutput, ProvenanceOutput,
- RefRoleOutput, ReferenceOutput, RelationOutput, SelectorOutput, StableIoErrorOutput,
- StatusOutput, SymbolKindOutput, SymbolOutput, TypeRefContextOutput, success_exit_code,
- success_status,
+ CacheProjectOutput, CacheProjectState, CacheReasonCountOutput, CacheReport,
+ CacheSnapshotOutput, ConfidenceOutput, ErrorEnvelope, Freshness, ImpactOutput, IndexOutput,
+ InventoryCompletenessOutput, InventoryOmissionReasonOutput, InventoryReasonCountOutput,
+ InventorySummaryOutput, ModuleDependencyOutput, ModuleDependencyTargetOutput,
+ OUTPUT_SCHEMA_VERSION, OccurrenceOutput, OutputEnvelope, OutputStatus,
+ PlanDecisionCountsOutput, ProjectOutput, ProvenanceOutput, RefRoleOutput, ReferenceOutput,
+ RelationOutput, SelectorOutput, StableIoErrorOutput, StatusOutput, SymbolKindOutput,
+ SymbolOutput, TypeRefContextOutput, success_exit_code, success_status,
};
pub use selector::{
SelectorContext, SelectorOptions, SelectorPurpose, SelectorRequest, SelectorResolution,
diff --git a/cli/src/project/select.rs b/cli/src/project/select.rs
index 5fc885c..fe03bbe 100644
--- a/cli/src/project/select.rs
+++ b/cli/src/project/select.rs
@@ -33,6 +33,14 @@ pub struct ProjectSelection {
/// An explicit index file may be outside `cwd`; marker discovery climbs from its
/// parent through the filesystem root.
pub fn select_project(request: &CliRequest, cwd: &Path) -> Result {
+ select_project_with_home(request, cwd, home_directory)
+}
+
+fn select_project_with_home(
+ request: &CliRequest,
+ cwd: &Path,
+ home_directory: impl FnOnce() -> Option,
+) -> Result {
let cwd = validated_cwd(cwd)?;
if let Some(root) = &request.global.root {
@@ -68,7 +76,33 @@ pub fn select_project(request: &CliRequest, cwd: &Path) -> Result) -> bool {
+ path.parent().is_none() || home.is_some_and(|home| path == home)
+}
+
+fn home_directory() -> Option {
+ directories::BaseDirs::new().map(|dirs| dirs.home_dir().to_path_buf())
+}
+
+fn select_implicit_directory(cwd: &ValidatedCwd, home: Option<&Path>) -> Result {
+ let home = home.and_then(|path| fs::canonicalize(path).ok());
+ if is_forbidden_default_root(&cwd.canonical, home.as_deref()) {
+ return Err(CliError::ProjectPath {
+ path: cwd.canonical.clone(),
+ reason: "refusing the current directory as an implicit project root \
+ (home or filesystem root); pass --root "
+ .into(),
+ });
+ }
+ select_directory(&cwd.canonical, cwd, SelectionProvenance::CurrentDirectory)
}
struct ValidatedCwd {
@@ -222,7 +256,9 @@ mod tests {
#[cfg(target_os = "macos")]
use super::is_trusted_system_ancestor;
- use super::{SelectionProvenance, select_project};
+ use super::{
+ SelectionProvenance, is_forbidden_default_root, select_project, select_project_with_home,
+ };
use crate::config::GlobalOptions;
use crate::error::CliError;
use crate::request::{CliRequest, CommandRequest};
@@ -355,6 +391,81 @@ mod tests {
);
}
+ #[test]
+ fn implicit_cwd_root_refuses_home_and_filesystem_root_only() {
+ let home = Path::new("/home/example");
+ assert!(is_forbidden_default_root(Path::new("/"), Some(home)));
+ assert!(is_forbidden_default_root(Path::new("/"), None));
+ assert!(is_forbidden_default_root(home, Some(home)));
+ assert!(!is_forbidden_default_root(
+ Path::new("/home/example/project"),
+ Some(home)
+ ));
+ assert!(!is_forbidden_default_root(home, None));
+ }
+
+ #[test]
+ fn implicit_selection_refuses_canonical_home_and_allows_explicit_home() {
+ let directory = tempdir().expect("temporary directory");
+ let home = directory.path().join("home");
+ let child = home.join("child");
+ fs::create_dir_all(&child).expect("home directories");
+ let canonical_home = fs::canonicalize(&home).expect("canonical home");
+ let discovered_home = home.join("child").join("..");
+
+ match select_project_with_home(&request(None, None), &home, || {
+ Some(discovered_home.clone())
+ }) {
+ Err(CliError::ProjectPath { path, reason }) => {
+ assert_eq!(path, canonical_home);
+ assert!(reason.contains("pass --root "));
+ }
+ result => panic!("implicit home selection must be refused, got {result:?}"),
+ }
+
+ let selection = select_project_with_home(&request(Some(&home), None), &home, || {
+ panic!("explicit root must skip home discovery")
+ })
+ .expect("explicit home selection");
+ assert_eq!(selection.canonical_root, canonical_home);
+ assert_eq!(selection.provenance, SelectionProvenance::RootArgument);
+
+ let selection =
+ select_project_with_home(&request(None, None), &child, || Some(discovered_home))
+ .expect("implicit child selection");
+ assert_eq!(
+ selection.canonical_root,
+ fs::canonicalize(&child).expect("canonical child")
+ );
+ assert_eq!(selection.provenance, SelectionProvenance::CurrentDirectory);
+ }
+
+ #[cfg(windows)]
+ #[test]
+ fn implicit_selection_matches_verbatim_cwd_to_unprefixed_home() {
+ use std::path::{Component, Prefix};
+
+ let directory = tempdir().expect("temporary directory");
+ let canonical_home = fs::canonicalize(directory.path()).expect("canonical home");
+ assert!(matches!(
+ canonical_home.components().next(),
+ Some(Component::Prefix(prefix)) if matches!(prefix.kind(), Prefix::VerbatimDisk(_))
+ ));
+ let canonical_text = canonical_home.to_str().expect("temporary path text");
+ let home = Path::new(
+ canonical_text
+ .strip_prefix(r"\\?\")
+ .expect("verbatim disk prefix"),
+ );
+ assert!(home.is_dir());
+ assert!(matches!(
+ select_project_with_home(&request(None, None), &canonical_home, || {
+ Some(home.to_path_buf())
+ }),
+ Err(CliError::ProjectPath { path, .. }) if path == canonical_home
+ ));
+ }
+
#[test]
fn rejects_invalid_cwd_before_other_selection_inputs() {
let directory = tempdir().expect("temporary directory");
diff --git a/cli/src/result.rs b/cli/src/result.rs
index 17fc6c5..bb0d3c1 100644
--- a/cli/src/result.rs
+++ b/cli/src/result.rs
@@ -3,6 +3,10 @@
use code2graph::{Confidence, Provenance, RefRole, SymbolId, SymbolKind, TypeRefContext};
use serde::{Deserialize, Serialize};
+mod omissions;
+pub(crate) use omissions::cache_omission_reasons;
+pub use omissions::{CacheReasonCountOutput, capped_omissions};
+
use crate::cache::{CacheCompleteness, CacheOmission, LoadedSnapshot};
use crate::config::{ResolverTier, ResourceLimits};
use crate::exit::ExitCode;
@@ -92,7 +96,19 @@ pub struct ProjectOutput {
pub completeness: CacheCompletenessOutput,
#[serde(rename = "omittedFiles")]
pub omitted_files: usize,
+ /// Capped to [`crate::config::DEFAULT_MAX_OMISSIONS`] entries; `omittedFiles` carries the total.
pub omissions: Vec,
+ /// Full snapshot counts, independent of the capped entry list.
+ #[serde(rename = "omissionReasons", default)]
+ pub omission_reasons: Vec,
+ /// Present only when `omissions` was capped, so a consumer can tell a short
+ /// list from a complete one.
+ #[serde(
+ rename = "omissionsTruncated",
+ default,
+ skip_serializing_if = "is_false"
+ )]
+ pub omissions_truncated: bool,
/// Why a previously cached snapshot was discarded and rebuilt, when that
/// happened during this run. A cache whose stored facts no longer satisfy
/// their validation contract — after an upgrade changes that contract, say
@@ -509,6 +525,12 @@ impl From<&CacheOmission> for CacheOmissionOutput {
}
}
+/// `skip_serializing_if` for the additive truncation flags: an untruncated
+/// envelope keeps the exact spelling it had before the flag existed.
+const fn is_false(value: &bool) -> bool {
+ !*value
+}
+
/// Counts of decisions made by the refresh planner.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)]
pub struct PlanDecisionCountsOutput {
@@ -544,7 +566,17 @@ pub struct IndexOutput {
pub completeness: CacheCompletenessOutput,
pub inventory_file_count: u64,
pub inventory_total_bytes: u64,
+ /// Total extracted-and-omitted files, independent of `omissions` being capped.
+ #[serde(default)]
+ pub omitted_files: usize,
+ /// Capped to [`crate::config::DEFAULT_MAX_OMISSIONS`] entries; `omitted_files` carries the total.
pub omissions: Vec,
+ /// Full snapshot counts, independent of the capped entry list.
+ #[serde(default)]
+ pub omission_reasons: Vec,
+ /// Present only when `omissions` was capped.
+ #[serde(default, skip_serializing_if = "is_false")]
+ pub omissions_truncated: bool,
pub changed: usize,
pub deleted: usize,
pub ignored_omissions: usize,
@@ -563,6 +595,8 @@ impl IndexOutput {
attempts: u8,
plan_decisions: PlanDecisionCountsOutput,
) -> Self {
+ let omission_reasons = cache_omission_reasons(&snapshot.omissions);
+ let (omissions, omissions_truncated) = capped_omissions(&snapshot.omissions);
Self {
candidate: snapshot.candidate_id.to_string(),
snapshot: snapshot.candidate_id.to_string(),
@@ -570,7 +604,10 @@ impl IndexOutput {
completeness: snapshot.completeness.into(),
inventory_file_count: snapshot.inventory_file_count,
inventory_total_bytes: snapshot.inventory_total_bytes,
- omissions: snapshot.omissions.iter().map(Into::into).collect(),
+ omitted_files: snapshot.omissions.len(),
+ omissions,
+ omission_reasons,
+ omissions_truncated,
changed,
deleted,
ignored_omissions,
@@ -618,7 +655,10 @@ impl StatusOutput {
omitted_files: snapshot.omissions.len(),
omission_reasons: Vec::new(),
},
- cached_omissions: snapshot.omissions.iter().map(Into::into).collect(),
+ // The cached entries mirror `project.omissions` and are capped the
+ // same way; `project.omitted_files` carries the full count and
+ // `project.omissions_truncated` says whether either list is short.
+ cached_omissions: capped_omissions(&snapshot.omissions).0,
max_files: limits.max_files,
max_file_bytes: limits.max_file_bytes,
max_total_bytes: limits.max_total_bytes,
@@ -798,6 +838,7 @@ impl ErrorEnvelope {
#[cfg(test)]
mod tests {
use super::*;
+ use crate::config::DEFAULT_MAX_OMISSIONS;
fn loaded_snapshot(completeness: CacheCompleteness) -> LoadedSnapshot {
let language = crate::cache::LanguageFeatureFingerprint::current();
@@ -909,10 +950,32 @@ mod tests {
completeness: snapshot.completeness.into(),
omitted_files: snapshot.omissions.len(),
omissions: snapshot.omissions.iter().map(Into::into).collect(),
+ omission_reasons: cache_omission_reasons(&snapshot.omissions),
+ omissions_truncated: false,
cache_recovery: None,
}
}
+ #[test]
+ fn omission_entry_lists_are_capped_while_the_truncation_is_reported() {
+ let omissions = (0..(DEFAULT_MAX_OMISSIONS + 5))
+ .map(|index| CacheOmission {
+ path: format!("src/file{index:04}.rs"),
+ reason: "file-count-limit".into(),
+ detail: "limit=10000".into(),
+ })
+ .collect::>();
+
+ let (reported, truncated) = capped_omissions(&omissions);
+ assert_eq!(reported.len(), DEFAULT_MAX_OMISSIONS);
+ assert!(truncated);
+ assert_eq!(reported[0].path, "src/file0000.rs");
+
+ let (short, truncated) = capped_omissions(&omissions[..3]);
+ assert_eq!(short.len(), 3);
+ assert!(!truncated);
+ }
+
#[test]
fn index_output_and_cached_status_are_owned_stable_contracts() {
let snapshot = loaded_snapshot(CacheCompleteness::Partial);
@@ -962,6 +1025,11 @@ mod tests {
reason: "file-too-large".into(),
detail: "limit=1024".into(),
}],
+ omitted_files: 1,
+ omission_reasons: cache_omission_reasons(
+ &loaded_snapshot(CacheCompleteness::Partial).omissions,
+ ),
+ omissions_truncated: false,
changed: 2,
deleted: 1,
ignored_omissions: 4,
@@ -986,6 +1054,8 @@ mod tests {
"omissions": [{
"path": "src/large.rs", "reason": "file-too-large", "detail": "limit=1024"
}],
+ "omitted_files": 1,
+ "omission_reasons": [{"reason": "file-too-large", "count": 1}],
"changed": 2,
"deleted": 1,
"ignored_omissions": 4,
@@ -1009,6 +1079,7 @@ mod tests {
"root": "/project", "snapshot": "snapshot", "tier": "scope",
"freshness": "stale", "cache": "hit", "completeness": "partial",
"omittedFiles": 1,
+ "omissionReasons": [{"reason": "file-too-large", "count": 1}],
"omissions": [{
"path": "src/large.rs", "reason": "file-too-large", "detail": "limit=1024"
}]
@@ -1045,6 +1116,7 @@ mod tests {
"root": "/project", "snapshot": "snapshot", "tier": "scope",
"freshness": spelling, "cache": "hit", "completeness": "partial",
"omittedFiles": 1,
+ "omissionReasons": [{"reason": "file-too-large", "count": 1}],
"omissions": [{
"path": "src/large.rs", "reason": "file-too-large", "detail": "limit=1024"
}]
@@ -1060,13 +1132,15 @@ mod tests {
completeness: CacheCompletenessOutput::Complete,
omitted_files: 0,
omissions: Vec::new(),
+ omission_reasons: Vec::new(),
+ omissions_truncated: false,
cache_recovery: None,
};
assert_eq!(
serde_json::to_value(complete).unwrap(),
serde_json::json!({
"root": "/project", "snapshot": "snapshot", "tier": "scope", "freshness": "fresh",
- "cache": "hit", "completeness": "complete", "omittedFiles": 0, "omissions": []
+ "cache": "hit", "completeness": "complete", "omittedFiles": 0, "omissions": [], "omissionReasons": []
})
);
}
@@ -1132,6 +1206,8 @@ mod tests {
completeness: snapshot.completeness.into(),
omitted_files: snapshot.omissions.len(),
omissions: snapshot.omissions.iter().map(Into::into).collect(),
+ omission_reasons: cache_omission_reasons(&snapshot.omissions),
+ omissions_truncated: false,
cache_recovery: None,
},
&snapshot,
diff --git a/cli/src/result/omissions.rs b/cli/src/result/omissions.rs
new file mode 100644
index 0000000..5af545e
--- /dev/null
+++ b/cli/src/result/omissions.rs
@@ -0,0 +1,47 @@
+// SPDX-License-Identifier: Apache-2.0
+
+use std::collections::BTreeMap;
+
+use serde::{Deserialize, Serialize};
+
+use crate::cache::CacheOmission;
+use crate::config::DEFAULT_MAX_OMISSIONS;
+
+use super::CacheOmissionOutput;
+
+/// One persisted omission-reason count, ordered by its verbatim cache reason.
+#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
+pub struct CacheReasonCountOutput {
+ pub reason: String,
+ pub count: usize,
+}
+
+/// Counts every cached omission before callers cap diagnostic entries.
+pub(crate) fn cache_omission_reasons(omissions: &[CacheOmission]) -> Vec {
+ let mut counts = BTreeMap::<&str, usize>::new();
+ for omission in omissions {
+ *counts.entry(&omission.reason).or_default() += 1;
+ }
+ counts
+ .into_iter()
+ .map(|(reason, count)| CacheReasonCountOutput {
+ reason: reason.to_owned(),
+ count,
+ })
+ .collect()
+}
+
+/// Deterministically ordered, capped view of an omission list.
+///
+/// Returns the reported entries (at most [`DEFAULT_MAX_OMISSIONS`]) and whether
+/// entries were held back. Callers keep the full total in their own count field,
+/// so capping the entry list never hides how many files were omitted.
+pub fn capped_omissions(omissions: &[CacheOmission]) -> (Vec, bool) {
+ let mut sorted = omissions.iter().collect::>();
+ sorted.sort_by(|left, right| {
+ (&left.path, &left.reason, &left.detail).cmp(&(&right.path, &right.reason, &right.detail))
+ });
+ let truncated = sorted.len() > DEFAULT_MAX_OMISSIONS;
+ sorted.truncate(DEFAULT_MAX_OMISSIONS);
+ (sorted.into_iter().map(Into::into).collect(), truncated)
+}
diff --git a/docs/agent-integration.md b/docs/agent-integration.md
index 28d3893..2035616 100644
--- a/docs/agent-integration.md
+++ b/docs/agent-integration.md
@@ -32,6 +32,9 @@ strings, config values, comments, error text, non-source files, unsupported lang
- ALWAYS pass `--allow-partial`: real codebases have files that fail extraction, and
without it any such file aborts the command.
+- ALWAYS pass `--root`: the implicit root is the working directory, and a home directory
+ or filesystem root is refused because walking one costs minutes and describes no
+ project.
- `--root` a single package for tight results, or the workspace root for cross-package
questions. `--json` for machine-readable output.
- `--tier scope` (default) is precise; `--tier name` is recall-first; `--tier dense`