Files
Genarrative/apps/ai-game-creator-shell/src-tauri/src/repository_context.rs
T
kdletters 97d8a6c39a
Project CI / AI game creator shell Rust shard 4/4 (push) Failing after 7m34s
Project CI / AI game creator shell Rust shard 2/4 (push) Failing after 7m42s
Project CI / AI game creator shell Rust shard 3/4 (push) Failing after 7m43s
Project CI / AI game creator shell Rust shard 1/4 (push) Failing after 8m1s
Project CI / AI game creator shell Rust smoke (push) Successful in 1m58s
Project CI / AI game creator shell Rust crates (push) Successful in 3m9s
Project CI / Backend tests (push) Successful in 12m4s
Project CI / Frontend tests (push) Successful in 11m37s
Project CI / Native shell tests (push) Successful in 15m37s
Project CI / Repository checks (push) Successful in 13m51s
Project CI / AI game creator shell web tests (push) Successful in 6m52s
清理并统一外置智能体提示词
将智能体系统指令、工具说明和恢复模板集中到外置提示词目录
精简重复否定、历史对照与无关实现说明,保留必要执行约束
扩展提示词构建校验和源码边界检查并接入现有本地与持续集成门禁
同步内置技能、协作规范、技术文档与对应验证断言
2026-09-20 15:32:48 +08:00

3135 lines
107 KiB
Rust

use serde::{Deserialize, Serialize};
use sha2::{Digest, Sha256};
use std::collections::{BTreeMap, BTreeSet, VecDeque};
use std::fmt::Write as _;
use std::fs::{self, Metadata, OpenOptions};
use std::io::Read;
use std::path::{Component, Path, PathBuf};
use std::process::{Command, Stdio};
use std::thread;
use std::time::{Duration, Instant};
const REPOSITORY_STARTUP_CONTEXT_SCHEMA_VERSION: &str = "repository-startup-context-v3";
const MAX_SCANNED_ENTRIES: usize = 10_000;
const MAX_CANDIDATE_FILES: usize = 2_000;
const MAX_SCAN_DEPTH: usize = 12;
const MAX_TOP_LEVEL_ENTRIES: usize = 96;
const MAX_MANIFESTS: usize = 96;
const MAX_MANIFEST_BYTES: usize = 256 * 1024;
const MAX_PACKAGE_SCRIPTS: usize = 64;
const MAX_SCRIPT_NAME_BYTES: usize = 128;
const MAX_SCRIPT_COMMAND_BYTES: usize = 512;
const MAX_PROJECT_NAME_BYTES: usize = 256;
const MAX_DOCUMENTS: usize = 64;
const MAX_DOCUMENT_BYTES: usize = 24 * 1024;
const MAX_DOCUMENT_BODY_BYTES: usize = 64 * 1024;
const MAX_SKILLS: usize = 64;
const MAX_SKILL_FILE_BYTES: usize = 128 * 1024;
const MAX_SKILL_NAME_BYTES: usize = 64;
const MAX_SKILL_DESCRIPTION_BYTES: usize = 2 * 1024;
const MAX_LANGUAGE_SUMMARIES: usize = 32;
const MAX_ENTRY_POINTS: usize = 96;
const MAX_SOURCE_PATHS: usize = 128;
const MAX_CONTEXT_OBJECTS: usize = 512;
const MAX_PROMPT_BYTES: usize = 24 * 1024;
const MAX_PROMPT_SOURCE_BYTES: usize = 2_500;
const MAX_PROMPT_INVENTORY_BYTES: usize = 3_500;
const MAX_PROMPT_MANIFEST_BYTES: usize = 3_000;
const MAX_PROMPT_SKILL_BYTES: usize = 4 * 1024;
const MAX_PROMPT_DOCUMENT_BYTES: usize = 10_000;
const MAX_PROMPT_DOCUMENT_BODY_BYTES: usize = 3_072;
const GIT_STATUS_TIMEOUT: Duration = Duration::from_millis(1_500);
const MAX_GIT_STATUS_BYTES: usize = 256 * 1024;
const REDACTED_SECRET: &str = "[REDACTED]";
#[derive(Clone, Debug, Default, Deserialize, Eq, PartialEq, Serialize)]
#[serde(rename_all = "camelCase")]
pub(crate) struct RepositoryStartupContext {
pub(crate) schema_version: String,
pub(crate) scan: RepositoryScanSummary,
pub(crate) manifests: Vec<RepositoryManifestSummary>,
#[serde(default)]
pub(crate) skills: Vec<RepositorySkillSummary>,
pub(crate) documents: Vec<RepositoryContextDocument>,
pub(crate) git_status: RepositoryGitStatusSummary,
pub(crate) languages: Vec<RepositoryLanguageSummary>,
pub(crate) entry_points: Vec<String>,
pub(crate) source_paths: Vec<String>,
pub(crate) fingerprint: String,
pub(crate) truncated: bool,
}
#[derive(Clone, Debug, Default, Deserialize, Eq, PartialEq, Serialize)]
#[serde(rename_all = "camelCase")]
pub(crate) struct RepositoryScanSummary {
pub(crate) scanned_entries: usize,
pub(crate) candidate_files: usize,
pub(crate) files: usize,
pub(crate) directories: usize,
pub(crate) max_depth_reached: usize,
pub(crate) top_level: Vec<RepositoryTopLevelEntry>,
}
#[derive(Clone, Debug, Default, Deserialize, Eq, PartialEq, Serialize)]
#[serde(rename_all = "camelCase")]
pub(crate) struct RepositoryTopLevelEntry {
pub(crate) path: String,
pub(crate) kind: String,
}
#[derive(Clone, Debug, Default, Deserialize, Eq, PartialEq, Serialize)]
#[serde(rename_all = "camelCase")]
pub(crate) struct RepositoryManifestSummary {
pub(crate) path: String,
pub(crate) kind: String,
pub(crate) name: Option<String>,
pub(crate) scripts: BTreeMap<String, String>,
pub(crate) size: u64,
pub(crate) parsed: bool,
pub(crate) truncated: bool,
}
#[derive(Clone, Debug, Default, Deserialize, Eq, PartialEq, Serialize)]
#[serde(rename_all = "camelCase")]
pub(crate) struct RepositoryContextDocument {
pub(crate) path: String,
pub(crate) kind: String,
#[serde(default)]
pub(crate) scope: String,
pub(crate) content: String,
pub(crate) content_sha256: String,
pub(crate) truncated: bool,
}
#[derive(Clone, Debug, Default, Deserialize, Eq, PartialEq, Serialize)]
#[serde(rename_all = "camelCase")]
pub(crate) struct RepositorySkillSummary {
pub(crate) name: String,
pub(crate) description: String,
pub(crate) entry_path: String,
pub(crate) source_root: String,
pub(crate) content_sha256: String,
pub(crate) truncated: bool,
}
#[derive(Clone, Debug, Default, Deserialize, Eq, PartialEq, Serialize)]
#[serde(rename_all = "camelCase")]
pub(crate) struct RepositoryGitStatusSummary {
pub(crate) available: bool,
pub(crate) is_repository: bool,
pub(crate) branch: Option<String>,
pub(crate) tracked_dirty_count: usize,
pub(crate) timed_out: bool,
}
#[derive(Clone, Debug, Default, Deserialize, Eq, PartialEq, Serialize)]
#[serde(rename_all = "camelCase")]
pub(crate) struct RepositoryLanguageSummary {
pub(crate) language: String,
pub(crate) files: usize,
pub(crate) bytes: u64,
}
#[derive(Clone, Debug)]
struct DiscoveredFile {
path: PathBuf,
relative_path: String,
size: u64,
}
#[derive(Debug, Default)]
struct RepositoryScanOutput {
summary: RepositoryScanSummary,
files: Vec<DiscoveredFile>,
truncated: bool,
}
#[derive(Debug)]
struct BoundedFileContent {
bytes: Vec<u8>,
truncated: bool,
}
#[derive(Debug, Deserialize)]
struct RepositorySkillFrontmatter {
name: String,
description: String,
}
#[derive(Debug, Default)]
struct BoundedGitOutput {
spawned: bool,
success: bool,
stdout: Vec<u8>,
timed_out: bool,
}
#[derive(Debug)]
struct BoundedPromptSection {
content: String,
max_bytes: usize,
truncated: bool,
}
impl BoundedPromptSection {
fn new(max_bytes: usize) -> Self {
Self {
content: String::new(),
max_bytes,
truncated: false,
}
}
fn finish(mut self) -> (String, bool) {
if self.truncated {
append_marker_within_budget(
&mut self.content,
self.max_bytes,
"\n[section truncated]\n",
);
}
(self.content, self.truncated)
}
}
impl std::fmt::Write for BoundedPromptSection {
fn write_str(&mut self, value: &str) -> std::fmt::Result {
if self.truncated {
return Ok(());
}
let remaining = self.max_bytes.saturating_sub(self.content.len());
if value.len() <= remaining {
self.content.push_str(value);
return Ok(());
}
self.content
.push_str(utf8_prefix(value, remaining.saturating_sub(24)));
self.truncated = true;
Ok(())
}
}
pub(crate) fn build_repository_startup_context_at(
root: &Path,
) -> Result<RepositoryStartupContext, String> {
let root = validate_repository_root(root)?;
let scan = scan_repository(&root)?;
let (manifests, manifests_truncated) = build_manifest_summaries(&root, &scan.files);
let (skills, skill_source_paths, skills_truncated) = build_skill_summaries(&root, &scan.files);
let document_source_paths = scan
.files
.iter()
.filter(|file| document_kind(&file.relative_path).is_some())
.map(|file| file.relative_path.clone())
.collect::<Vec<_>>();
let (documents, documents_truncated) = build_context_documents(&root, &scan.files);
let languages = collect_language_distribution(&scan.files);
let (entry_points, entry_points_truncated) = collect_entry_points(&scan.files);
let git_status = collect_git_status(&root);
let mut source_paths = manifests
.iter()
.map(|manifest| manifest.path.clone())
.chain(skill_source_paths)
.chain(documents.iter().map(|document| document.path.clone()))
.chain(document_source_paths)
.collect::<Vec<_>>();
sort_root_to_specific(&mut source_paths);
source_paths.dedup();
let source_paths_truncated = source_paths.len() > MAX_SOURCE_PATHS;
source_paths.truncate(MAX_SOURCE_PATHS);
let mut context = RepositoryStartupContext {
schema_version: REPOSITORY_STARTUP_CONTEXT_SCHEMA_VERSION.to_string(),
scan: scan.summary,
manifests,
skills,
documents,
git_status,
languages,
entry_points,
source_paths,
fingerprint: String::new(),
truncated: scan.truncated
|| manifests_truncated
|| skills_truncated
|| documents_truncated
|| entry_points_truncated
|| source_paths_truncated,
};
context.truncated |= enforce_context_object_budget(&mut context);
context.fingerprint = repository_startup_context_fingerprint(&context);
Ok(context)
}
pub(crate) fn repository_startup_context_fingerprint(context: &RepositoryStartupContext) -> String {
let mut canonical = context.clone();
canonical.fingerprint.clear();
canonical.git_status = RepositoryGitStatusSummary::default();
for entry in &mut canonical.scan.top_level {
entry.path = sanitize_repository_text(&entry.path, None);
entry.kind = sanitize_repository_text(&entry.kind, None);
}
canonical.scan.top_level.sort_by(|left, right| {
left.path
.cmp(&right.path)
.then_with(|| left.kind.cmp(&right.kind))
});
for manifest in &mut canonical.manifests {
manifest.path = sanitize_repository_text(&manifest.path, None);
manifest.kind = sanitize_repository_text(&manifest.kind, None);
manifest.name = manifest
.name
.as_deref()
.map(|name| sanitize_repository_text(name, None));
manifest.scripts = manifest
.scripts
.iter()
.map(|(name, command)| {
(
sanitize_repository_text(name, None),
sanitize_repository_text(command, None),
)
})
.collect();
manifest.size = 0;
}
canonical.manifests.sort_by(|left, right| {
root_to_specific_cmp(&left.path, &right.path).then_with(|| left.kind.cmp(&right.kind))
});
for skill in &mut canonical.skills {
skill.name = sanitize_repository_text(&skill.name, None);
skill.description = sanitize_repository_text(&skill.description, None);
skill.entry_path = sanitize_repository_text(&skill.entry_path, None);
skill.source_root = sanitize_repository_text(&skill.source_root, None);
skill.content_sha256 = sanitize_repository_text(&skill.content_sha256, None);
}
canonical.skills.sort_by(|left, right| {
left.name
.cmp(&right.name)
.then_with(|| {
skill_source_priority(&left.source_root)
.cmp(&skill_source_priority(&right.source_root))
})
.then_with(|| left.entry_path.cmp(&right.entry_path))
});
for document in &mut canonical.documents {
document.path = sanitize_repository_text(&document.path, None);
document.kind = sanitize_repository_text(&document.kind, None);
document.scope = sanitize_repository_text(&document.scope, None);
document.content = sanitize_repository_text(&document.content, None);
document.content_sha256 = format!("{:x}", Sha256::digest(document.content.as_bytes()));
}
canonical.documents.sort_by(|left, right| {
document_order_key(&left.path, &left.kind)
.cmp(&document_order_key(&right.path, &right.kind))
});
for language in &mut canonical.languages {
language.language = sanitize_repository_text(&language.language, None);
language.bytes = 0;
}
canonical
.languages
.sort_by(|left, right| left.language.cmp(&right.language));
canonical.entry_points = canonical
.entry_points
.iter()
.map(|path| sanitize_repository_text(path, None))
.collect();
canonical.entry_points.sort();
canonical.entry_points.dedup();
canonical.source_paths = canonical
.source_paths
.iter()
.map(|path| sanitize_repository_text(path, None))
.collect();
sort_root_to_specific(&mut canonical.source_paths);
canonical.source_paths.dedup();
let encoded = serde_json::to_vec(&canonical)
.expect("RepositoryStartupContext contains only JSON-serializable values");
format!("{:x}", Sha256::digest(encoded))
}
fn enforce_context_object_budget(context: &mut RepositoryStartupContext) -> bool {
fn retain_within_budget<T>(
values: &mut Vec<T>,
item_limit: usize,
remaining: &mut usize,
) -> bool {
let retained = values.len().min(item_limit).min(*remaining);
let truncated = retained < values.len();
values.truncate(retained);
*remaining = remaining.saturating_sub(retained);
truncated
}
let mut remaining = MAX_CONTEXT_OBJECTS;
let mut truncated = false;
truncated |= retain_within_budget(&mut context.documents, MAX_DOCUMENTS, &mut remaining);
truncated |= retain_within_budget(&mut context.skills, MAX_SKILLS, &mut remaining);
truncated |= retain_within_budget(&mut context.manifests, MAX_MANIFESTS, &mut remaining);
truncated |= retain_within_budget(&mut context.source_paths, MAX_SOURCE_PATHS, &mut remaining);
truncated |= retain_within_budget(
&mut context.scan.top_level,
MAX_TOP_LEVEL_ENTRIES,
&mut remaining,
);
truncated |= retain_within_budget(&mut context.entry_points, MAX_ENTRY_POINTS, &mut remaining);
truncated |= retain_within_budget(
&mut context.languages,
MAX_LANGUAGE_SUMMARIES,
&mut remaining,
);
truncated
}
pub(crate) fn render_repository_startup_context_for_prompt(
context: &RepositoryStartupContext,
) -> String {
let (sources, sources_truncated) = render_prompt_sources(context);
let (inventory, inventory_truncated) = render_prompt_inventory(context);
let (manifests, manifests_truncated) = render_prompt_manifests(context);
let (skills, skills_truncated) = render_prompt_skills(context);
let (documents, documents_truncated) = render_prompt_documents(context);
let prompt_truncated = sources_truncated
|| inventory_truncated
|| manifests_truncated
|| skills_truncated
|| documents_truncated;
let mut prompt = String::with_capacity(MAX_PROMPT_BYTES);
let _ = writeln!(prompt, prompt_text!("goalContext.repository_heading"));
let _ = writeln!(
prompt,
prompt_text!("goalContext.repository_agents_instructions")
);
let _ = writeln!(
prompt,
prompt_text!("goalContext.repository_reference_trust")
);
let _ = writeln!(
prompt,
prompt_text!("goalContext.repository_truncated_agents")
);
let _ = writeln!(
prompt,
prompt_text!("goalContext.repository_skill_instructions")
);
let _ = writeln!(
prompt,
"schemaVersion: {}",
safe_prompt_value(&context.schema_version)
);
let _ = writeln!(
prompt,
"fingerprint: {}",
repository_startup_context_fingerprint(context)
);
let _ = writeln!(
prompt,
"truncated: {}",
context.truncated || prompt_truncated
);
prompt.push_str(&sources);
prompt.push_str(&inventory);
prompt.push_str(&manifests);
prompt.push_str(&skills);
prompt.push_str(&documents);
prompt.push_str(prompt_text!("goalContext.repository_end"));
if prompt.len() <= MAX_PROMPT_BYTES {
return prompt;
}
truncate_with_marker(
&prompt,
MAX_PROMPT_BYTES,
"\n[repository startup context prompt truncated]\n",
)
}
fn validate_repository_root(root: &Path) -> Result<PathBuf, String> {
if !root.is_absolute() {
return Err("Repository root must be an absolute path".to_string());
}
if root
.components()
.any(|component| matches!(component, Component::ParentDir))
{
return Err("Repository root must not contain parent-directory components".to_string());
}
let canonical_temp_root = fs::canonicalize(std::env::temp_dir()).ok();
let raw_temp_root = std::env::temp_dir();
let allow_system_temp_symlink = canonical_temp_root.as_ref().is_some_and(|temp| {
fs::canonicalize(root).is_ok_and(|canonical| canonical.starts_with(temp))
}) && root.starts_with(&raw_temp_root);
for ancestor in root.ancestors() {
let metadata = fs::symlink_metadata(ancestor)
.map_err(|error| format!("Unable to inspect repository root ancestor: {error}"))?;
let is_system_temp_ancestor = allow_system_temp_symlink
&& raw_temp_root
.ancestors()
.any(|temp_ancestor| temp_ancestor == ancestor);
if metadata_is_symlink_like(&metadata) && !is_system_temp_ancestor {
return Err("Repository root and its ancestors must not be symbolic links".to_string());
}
}
let metadata = fs::symlink_metadata(root)
.map_err(|error| format!("Unable to inspect repository root: {error}"))?;
if !metadata.is_dir() {
return Err("Repository root must be a directory".to_string());
}
let canonical = fs::canonicalize(root)
.map_err(|error| format!("Unable to canonicalize repository root: {error}"))?;
let canonical_metadata = fs::symlink_metadata(&canonical)
.map_err(|error| format!("Unable to inspect canonical repository root: {error}"))?;
if metadata_is_symlink_like(&canonical_metadata) || !canonical_metadata.is_dir() {
return Err("Canonical repository root must be a regular directory".to_string());
}
Ok(canonical)
}
fn metadata_is_symlink_like(metadata: &Metadata) -> bool {
if metadata.file_type().is_symlink() {
return true;
}
#[cfg(windows)]
{
use std::os::windows::fs::MetadataExt;
const FILE_ATTRIBUTE_REPARSE_POINT: u32 = 0x0000_0400;
return metadata.file_attributes() & FILE_ATTRIBUTE_REPARSE_POINT != 0;
}
#[cfg(not(windows))]
false
}
fn scan_repository(root: &Path) -> Result<RepositoryScanOutput, String> {
let mut output = RepositoryScanOutput::default();
let mut queue = VecDeque::from([(root.to_path_buf(), 0_usize)]);
'scan: while let Some((directory, depth)) = queue.pop_front() {
let read_dir = match fs::read_dir(&directory) {
Ok(read_dir) => read_dir,
Err(error) if depth == 0 => {
return Err(format!("Unable to read repository root: {error}"));
}
Err(_) => {
output.truncated = true;
continue;
}
};
let mut entries = Vec::new();
let mut read_dir = read_dir;
while output.summary.scanned_entries < MAX_SCANNED_ENTRIES {
let Some(entry) = read_dir.next() else {
break;
};
output.summary.scanned_entries += 1;
match entry {
Ok(entry) => entries.push(entry),
Err(_) => output.truncated = true,
}
}
let scan_budget_exhausted = output.summary.scanned_entries >= MAX_SCANNED_ENTRIES;
output.truncated |= scan_budget_exhausted;
entries.sort_by_key(|entry| entry.file_name());
for entry in entries {
let path = entry.path();
let metadata = match fs::symlink_metadata(&path) {
Ok(metadata) => metadata,
Err(_) => {
output.truncated = true;
continue;
}
};
if metadata_is_symlink_like(&metadata) {
continue;
}
let relative = match path.strip_prefix(root) {
Ok(relative) => relative,
Err(_) => {
output.truncated = true;
continue;
}
};
let Some(relative_path) = normalized_relative_path(relative) else {
output.truncated = true;
continue;
};
let child_depth = depth + 1;
if metadata.is_dir() {
if should_ignore_directory(relative) {
continue;
}
output.summary.directories += 1;
output.summary.max_depth_reached =
output.summary.max_depth_reached.max(child_depth);
record_top_level_entry(&mut output, child_depth, &relative_path, "directory");
if child_depth >= MAX_SCAN_DEPTH {
output.truncated = true;
} else {
queue.push_back((path, child_depth));
}
continue;
}
if !metadata.is_file() || should_ignore_file(relative) {
continue;
}
output.summary.files += 1;
output.summary.max_depth_reached = output.summary.max_depth_reached.max(child_depth);
record_top_level_entry(&mut output, child_depth, &relative_path, "file");
if output.files.len() >= MAX_CANDIDATE_FILES {
output.truncated = true;
continue;
}
let size = metadata.len();
output.files.push(DiscoveredFile {
path,
relative_path,
size,
});
}
if scan_budget_exhausted {
break 'scan;
}
}
output.summary.candidate_files = output.files.len();
output
.summary
.top_level
.sort_by(|left, right| left.path.cmp(&right.path));
Ok(output)
}
fn record_top_level_entry(output: &mut RepositoryScanOutput, depth: usize, path: &str, kind: &str) {
if depth != 1 {
return;
}
if output.summary.top_level.len() >= MAX_TOP_LEVEL_ENTRIES {
output.truncated = true;
return;
}
output.summary.top_level.push(RepositoryTopLevelEntry {
path: path.to_string(),
kind: kind.to_string(),
});
}
fn normalized_relative_path(path: &Path) -> Option<String> {
let mut parts = Vec::new();
for component in path.components() {
match component {
Component::Normal(part) => {
let part = part.to_string_lossy();
let sanitized = part
.chars()
.map(|character| {
if character == '\\' || character.is_control() {
'_'
} else {
character
}
})
.collect::<String>();
if sanitized.is_empty() {
return None;
}
parts.push(sanitized);
}
Component::CurDir => {}
Component::ParentDir | Component::RootDir | Component::Prefix(_) => return None,
}
}
if parts.is_empty() {
None
} else {
Some(parts.join("/"))
}
}
fn relative_components_lower(path: &Path) -> Vec<String> {
path.components()
.filter_map(|component| match component {
Component::Normal(part) => Some(part.to_string_lossy().to_ascii_lowercase()),
_ => None,
})
.collect()
}
fn should_ignore_directory(path: &Path) -> bool {
const IGNORED_DIRECTORY_NAMES: &[&str] = &[
".git",
".hg",
".svn",
".agent",
"node_modules",
"bower_components",
"target",
"build",
"dist",
"out",
"coverage",
".coverage",
".cache",
"cache",
"__pycache__",
".pytest_cache",
".mypy_cache",
".ruff_cache",
".tox",
".venv",
"venv",
"vendor",
"pods",
".gradle",
".next",
".nuxt",
".svelte-kit",
".parcel-cache",
".turbo",
".codegraph",
".rag",
".idea",
".vscode",
".ssh",
".aws",
".azure",
".gnupg",
".kube",
".secrets",
"secrets",
"credentials",
];
let components = relative_components_lower(path);
if components
.iter()
.any(|component| IGNORED_DIRECTORY_NAMES.contains(&component.as_str()))
{
return true;
}
if components.first().is_some_and(|component| {
matches!(component.as_str(), "memory" | "logs" | ".app" | "exports")
}) {
return true;
}
components.windows(2).any(|pair| {
pair[0] == ".agent"
&& matches!(
pair[1].as_str(),
"runtime" | "conversations" | "browser-validations" | "locks"
)
})
}
fn should_ignore_file(path: &Path) -> bool {
let Some(file_name) = path.file_name().and_then(|name| name.to_str()) else {
return true;
};
let lower = file_name.to_ascii_lowercase();
if lower == ".env" || lower.starts_with(".env.") || lower == ".envrc" {
return true;
}
if matches!(
lower.as_str(),
".npmrc"
| ".pypirc"
| ".netrc"
| ".git-credentials"
| "credentials.json"
| "credentials.toml"
| "secrets.json"
| "secrets.toml"
| "cookies.txt"
| "cookies.json"
| "token"
| "token.txt"
| "game-creator.config.json"
) {
return true;
}
if lower.starts_with("id_rsa") || lower.starts_with("id_ed25519") {
return true;
}
let extension = Path::new(&lower)
.extension()
.and_then(|extension| extension.to_str())
.unwrap_or_default();
if matches!(
extension,
"pem" | "key" | "p12" | "pfx" | "sqlite" | "sqlite3" | "db" | "dump"
) {
return true;
}
(lower.contains("cookie") || lower.contains("credential"))
&& matches!(extension, "json" | "txt" | "yaml" | "yml" | "toml")
}
fn build_skill_summaries(
root: &Path,
files: &[DiscoveredFile],
) -> (Vec<RepositorySkillSummary>, Vec<String>, bool) {
let mut candidates = files
.iter()
.filter_map(|file| {
skill_entry_identity(&file.relative_path)
.map(|(source_root, directory_name)| (file, source_root, directory_name))
})
.collect::<Vec<_>>();
candidates.sort_by(
|(left, left_root, left_name), (right, right_root, right_name)| {
skill_source_priority(left_root)
.cmp(&skill_source_priority(right_root))
.then_with(|| left_name.cmp(right_name))
.then_with(|| left.relative_path.cmp(&right.relative_path))
},
);
let mut summaries = Vec::new();
let mut active_names = BTreeSet::new();
let mut truncated = false;
for (file, source_root, directory_name) in candidates {
if summaries.len() >= MAX_SKILLS {
truncated = true;
break;
}
if file.size > MAX_SKILL_FILE_BYTES as u64 {
truncated = true;
continue;
}
let content = match read_bounded_regular_file(&file.path, MAX_SKILL_FILE_BYTES) {
Ok(Some(content)) if !content.truncated => content,
Ok(Some(_)) | Ok(None) | Err(_) => {
truncated = true;
continue;
}
};
let Ok(text) = std::str::from_utf8(&content.bytes) else {
continue;
};
let Some(frontmatter) = extract_skill_frontmatter(text) else {
continue;
};
let Ok(frontmatter) = serde_yaml::from_str::<RepositorySkillFrontmatter>(frontmatter)
else {
continue;
};
if !is_valid_skill_name(directory_name)
|| frontmatter.name != directory_name
|| frontmatter.description.trim().is_empty()
|| active_names.contains(directory_name)
{
continue;
}
let (description, description_truncated) =
summarize_one_line(root, &frontmatter.description, MAX_SKILL_DESCRIPTION_BYTES);
if description.trim().is_empty() {
continue;
}
let sanitized_content = sanitize_repository_text(text, Some(root));
summaries.push(RepositorySkillSummary {
name: directory_name.to_string(),
description,
entry_path: file.relative_path.clone(),
source_root: source_root.to_string(),
content_sha256: format!("{:x}", Sha256::digest(sanitized_content.as_bytes())),
truncated: description_truncated,
});
active_names.insert(directory_name.to_string());
truncated |= description_truncated;
}
summaries.sort_by(|left, right| left.name.cmp(&right.name));
let source_paths = summaries
.iter()
.map(|skill| skill.entry_path.clone())
.collect::<Vec<_>>();
(summaries, source_paths, truncated)
}
fn skill_entry_identity(relative_path: &str) -> Option<(&str, &str)> {
let parts = relative_path.split('/').collect::<Vec<_>>();
if parts.len() != 4
|| !matches!(parts[0], ".codex" | ".agents")
|| parts[1] != "skills"
|| parts[3] != "SKILL.md"
{
return None;
}
Some((parts[0], parts[2]))
}
fn skill_source_priority(source_root: &str) -> usize {
match source_root {
".codex" => 0,
".agents" => 1,
_ => 2,
}
}
fn is_valid_skill_name(name: &str) -> bool {
if name.is_empty()
|| name.len() > MAX_SKILL_NAME_BYTES
|| name.starts_with('-')
|| name.ends_with('-')
|| name.contains("--")
{
return false;
}
name.bytes()
.all(|byte| byte.is_ascii_lowercase() || byte.is_ascii_digit() || byte == b'-')
}
fn extract_skill_frontmatter(content: &str) -> Option<&str> {
let body = content
.strip_prefix("---\n")
.or_else(|| content.strip_prefix("---\r\n"))?;
let mut offset = 0;
for line in body.split_inclusive('\n') {
if line.trim_end_matches(['\r', '\n']) == "---" {
return Some(&body[..offset]);
}
offset += line.len();
}
None
}
fn build_manifest_summaries(
root: &Path,
files: &[DiscoveredFile],
) -> (Vec<RepositoryManifestSummary>, bool) {
let mut candidates = files
.iter()
.filter(|file| manifest_kind(&file.relative_path).is_some())
.collect::<Vec<_>>();
candidates
.sort_by(|left, right| root_to_specific_cmp(&left.relative_path, &right.relative_path));
let mut truncated = candidates.len() > MAX_MANIFESTS;
let mut summaries = Vec::new();
for file in candidates.into_iter().take(MAX_MANIFESTS) {
let kind = manifest_kind(&file.relative_path).unwrap_or_default();
let mut summary = RepositoryManifestSummary {
path: file.relative_path.clone(),
kind: kind.to_string(),
size: file.size,
..RepositoryManifestSummary::default()
};
match read_bounded_regular_file(&file.path, MAX_MANIFEST_BYTES) {
Ok(Some(content)) => {
summary.truncated = content.truncated;
truncated |= content.truncated;
let text = String::from_utf8_lossy(&content.bytes);
parse_manifest_summary(root, &text, &mut summary, &mut truncated);
}
Ok(None) | Err(_) => {
summary.truncated = true;
truncated = true;
}
}
summaries.push(summary);
}
(summaries, truncated)
}
fn manifest_kind(relative_path: &str) -> Option<&'static str> {
let file_name = relative_path
.rsplit('/')
.next()
.unwrap_or(relative_path)
.to_ascii_lowercase();
match file_name.as_str() {
"package.json" => Some("package.json"),
"cargo.toml" => Some("Cargo.toml"),
"pyproject.toml" => Some("pyproject.toml"),
"go.mod" => Some("go.mod"),
_ => None,
}
}
fn parse_manifest_summary(
root: &Path,
content: &str,
summary: &mut RepositoryManifestSummary,
context_truncated: &mut bool,
) {
match summary.kind.as_str() {
"package.json" => {
let Ok(package) = serde_json::from_str::<serde_json::Value>(content) else {
return;
};
summary.parsed = true;
summary.name = package
.get("name")
.and_then(serde_json::Value::as_str)
.map(|name| summarize_one_line(root, name, MAX_PROJECT_NAME_BYTES).0)
.filter(|name| !name.is_empty());
let Some(scripts) = package
.get("scripts")
.and_then(serde_json::Value::as_object)
else {
return;
};
let mut scripts = scripts
.iter()
.filter_map(|(name, command)| {
command.as_str().map(|command| (name.as_str(), command))
})
.collect::<Vec<_>>();
scripts.sort_by(|left, right| {
package_script_priority(left.0)
.cmp(&package_script_priority(right.0))
.then_with(|| left.0.cmp(right.0))
});
if scripts.len() > MAX_PACKAGE_SCRIPTS {
summary.truncated = true;
*context_truncated = true;
}
for (name, command) in scripts.into_iter().take(MAX_PACKAGE_SCRIPTS) {
let (name, name_truncated) = summarize_one_line(root, name, MAX_SCRIPT_NAME_BYTES);
let (command, command_truncated) =
summarize_one_line(root, command, MAX_SCRIPT_COMMAND_BYTES);
if name.is_empty() {
continue;
}
if summary.scripts.insert(name, command).is_some()
|| name_truncated
|| command_truncated
{
summary.truncated = true;
*context_truncated = true;
}
}
}
"Cargo.toml" => {
summary.parsed = true;
summary.name = parse_toml_project_name(content, &["package"])
.map(|name| summarize_one_line(root, &name, MAX_PROJECT_NAME_BYTES).0)
.filter(|name| !name.is_empty());
}
"pyproject.toml" => {
summary.parsed = true;
summary.name = parse_toml_project_name(content, &["project", "tool.poetry"])
.map(|name| summarize_one_line(root, &name, MAX_PROJECT_NAME_BYTES).0)
.filter(|name| !name.is_empty());
}
"go.mod" => {
summary.parsed = true;
summary.name = content
.lines()
.map(str::trim)
.find_map(|line| line.strip_prefix("module ").map(str::trim))
.map(|name| summarize_one_line(root, name, MAX_PROJECT_NAME_BYTES).0)
.filter(|name| !name.is_empty());
}
_ => {}
}
}
fn package_script_priority(name: &str) -> usize {
const PRIORITY_NAMES: &[&str] = &[
"check",
"test",
"lint",
"typecheck",
"build",
"verify",
"validate",
];
let lower = name.to_ascii_lowercase();
PRIORITY_NAMES
.iter()
.position(|candidate| {
lower == *candidate
|| lower.starts_with(&format!("{candidate}:"))
|| lower.starts_with(&format!("pre{candidate}"))
|| lower.starts_with(&format!("post{candidate}"))
})
.unwrap_or(PRIORITY_NAMES.len())
}
fn parse_toml_project_name(content: &str, allowed_sections: &[&str]) -> Option<String> {
let mut section = String::new();
for line in content.lines() {
let line = line.trim();
if line.starts_with('[') && line.ends_with(']') {
section = line[1..line.len() - 1].trim().to_ascii_lowercase();
continue;
}
if !allowed_sections.contains(&section.as_str()) {
continue;
}
let Some((key, value)) = line.split_once('=') else {
continue;
};
if key.trim() != "name" {
continue;
}
let value = value.trim();
if let Some(quote) = value
.chars()
.next()
.filter(|quote| matches!(quote, '"' | '\''))
{
let rest = &value[quote.len_utf8()..];
if let Some(end) = rest.find(quote) {
return Some(rest[..end].to_string());
}
}
let bare = value.split('#').next().unwrap_or_default().trim();
if !bare.is_empty() {
return Some(bare.to_string());
}
}
None
}
fn build_context_documents(
root: &Path,
files: &[DiscoveredFile],
) -> (Vec<RepositoryContextDocument>, bool) {
let mut candidates = files
.iter()
.filter_map(|file| document_kind(&file.relative_path).map(|kind| (file, kind.to_string())))
.collect::<Vec<_>>();
candidates.sort_by(|(left, left_kind), (right, right_kind)| {
document_order_key(&left.relative_path, left_kind)
.cmp(&document_order_key(&right.relative_path, right_kind))
});
let mut remaining_body_bytes = MAX_DOCUMENT_BODY_BYTES;
let mut truncated = candidates.len() > MAX_DOCUMENTS;
let mut documents = Vec::with_capacity(candidates.len().min(MAX_DOCUMENTS));
for (file, kind) in candidates.into_iter().take(MAX_DOCUMENTS) {
let allowed = remaining_body_bytes.min(MAX_DOCUMENT_BYTES);
let content = if allowed == 0 {
BoundedFileContent {
bytes: Vec::new(),
truncated: file.size > 0,
}
} else {
match read_bounded_regular_file(&file.path, allowed) {
Ok(Some(content)) => content,
Ok(None) | Err(_) => {
truncated = true;
continue;
}
}
};
let sanitized =
sanitize_repository_text(&String::from_utf8_lossy(&content.bytes), Some(root));
let (sanitized, sanitized_truncated) = truncate_utf8_owned(sanitized, allowed);
let content_sha256 = format!("{:x}", Sha256::digest(sanitized.as_bytes()));
remaining_body_bytes = remaining_body_bytes.saturating_sub(sanitized.len());
let document_truncated = content.truncated || sanitized_truncated;
truncated |= document_truncated;
documents.push(RepositoryContextDocument {
path: file.relative_path.clone(),
kind,
scope: document_scope(&file.relative_path),
content: sanitized,
content_sha256,
truncated: document_truncated,
});
}
(documents, truncated)
}
fn document_kind(relative_path: &str) -> Option<&'static str> {
let mut components = relative_path.split('/');
let first = components.next().unwrap_or_default();
let is_root = components.next().is_none();
let file_name = relative_path
.rsplit('/')
.next()
.unwrap_or(relative_path)
.to_ascii_lowercase();
if file_name == "agents.md" {
return Some("agents");
}
if is_root && first.eq_ignore_ascii_case("CONTEXT.md") {
return Some("context");
}
if is_root
&& matches!(
file_name.as_str(),
"readme" | "readme.md" | "readme.txt" | "readme.rst"
)
{
return Some("readme");
}
None
}
fn document_scope(relative_path: &str) -> String {
relative_path
.rsplit_once('/')
.map(|(parent, _)| parent)
.filter(|parent| !parent.is_empty())
.unwrap_or(".")
.to_string()
}
fn document_order_key<'a>(relative_path: &'a str, kind: &str) -> (usize, usize, &'a str) {
let kind_priority = match kind {
"agents" => 0,
"context" => 1,
"readme" => 2,
_ => 3,
};
(kind_priority, path_depth(relative_path), relative_path)
}
fn read_bounded_regular_file(
path: &Path,
max_bytes: usize,
) -> std::io::Result<Option<BoundedFileContent>> {
let path_metadata = fs::symlink_metadata(path)?;
if metadata_is_symlink_like(&path_metadata) || !path_metadata.is_file() {
return Ok(None);
}
let mut options = OpenOptions::new();
options.read(true);
#[cfg(unix)]
{
use std::os::unix::fs::OpenOptionsExt;
options.custom_flags(libc::O_CLOEXEC | libc::O_NOFOLLOW);
}
#[cfg(windows)]
{
use std::os::windows::fs::OpenOptionsExt;
const FILE_FLAG_OPEN_REPARSE_POINT: u32 = 0x0020_0000;
options.custom_flags(FILE_FLAG_OPEN_REPARSE_POINT);
}
let mut file = options.open(path)?;
let opened_metadata = file.metadata()?;
if metadata_is_symlink_like(&opened_metadata) || !opened_metadata.is_file() {
return Ok(None);
}
#[cfg(unix)]
{
use std::os::unix::fs::MetadataExt;
if path_metadata.dev() != opened_metadata.dev()
|| path_metadata.ino() != opened_metadata.ino()
{
return Ok(None);
}
}
let mut bytes = Vec::with_capacity(max_bytes.min(16 * 1024).saturating_add(1));
file.by_ref()
.take(max_bytes.saturating_add(1) as u64)
.read_to_end(&mut bytes)?;
let truncated = opened_metadata.len() > max_bytes as u64 || bytes.len() > max_bytes;
bytes.truncate(max_bytes);
Ok(Some(BoundedFileContent { bytes, truncated }))
}
fn collect_language_distribution(files: &[DiscoveredFile]) -> Vec<RepositoryLanguageSummary> {
let mut counts = BTreeMap::<&'static str, (usize, u64)>::new();
for file in files {
let Some(language) = language_for_path(&file.relative_path) else {
continue;
};
let entry = counts.entry(language).or_default();
entry.0 += 1;
entry.1 = entry.1.saturating_add(file.size);
}
let mut languages = counts
.into_iter()
.map(|(language, (files, bytes))| RepositoryLanguageSummary {
language: language.to_string(),
files,
bytes,
})
.collect::<Vec<_>>();
languages.sort_by(|left, right| {
right
.files
.cmp(&left.files)
.then_with(|| left.language.cmp(&right.language))
});
languages
}
fn language_for_path(relative_path: &str) -> Option<&'static str> {
let path = Path::new(relative_path);
let file_name = path
.file_name()
.and_then(|name| name.to_str())
.unwrap_or_default()
.to_ascii_lowercase();
if file_name == "dockerfile" {
return Some("Dockerfile");
}
if matches!(file_name.as_str(), "makefile" | "gnumakefile") {
return Some("Make");
}
let extension = path
.extension()
.and_then(|extension| extension.to_str())?
.to_ascii_lowercase();
match extension.as_str() {
"rs" => Some("Rust"),
"ts" | "tsx" | "mts" | "cts" => Some("TypeScript"),
"js" | "jsx" | "mjs" | "cjs" => Some("JavaScript"),
"py" | "pyi" => Some("Python"),
"go" => Some("Go"),
"c" | "h" => Some("C"),
"cc" | "cpp" | "cxx" | "hpp" | "hh" => Some("C++"),
"cs" => Some("C#"),
"java" => Some("Java"),
"kt" | "kts" => Some("Kotlin"),
"swift" => Some("Swift"),
"rb" => Some("Ruby"),
"php" => Some("PHP"),
"dart" => Some("Dart"),
"zig" => Some("Zig"),
"lua" => Some("Lua"),
"gd" => Some("GDScript"),
"html" | "htm" => Some("HTML"),
"css" | "scss" | "sass" | "less" => Some("CSS"),
"vue" => Some("Vue"),
"svelte" => Some("Svelte"),
"json" | "jsonc" => Some("JSON"),
"toml" => Some("TOML"),
"yaml" | "yml" => Some("YAML"),
"md" | "mdx" => Some("Markdown"),
"sh" | "bash" | "zsh" | "fish" | "ps1" => Some("Shell"),
"sql" => Some("SQL"),
_ => None,
}
}
fn collect_entry_points(files: &[DiscoveredFile]) -> (Vec<String>, bool) {
let mut candidates = BTreeSet::new();
for file in files {
if is_common_entry_point(&file.relative_path) {
candidates.insert(file.relative_path.clone());
}
}
let mut candidates = candidates.into_iter().collect::<Vec<_>>();
candidates.sort_by(|left, right| {
entry_point_priority(left)
.cmp(&entry_point_priority(right))
.then_with(|| left.cmp(right))
});
let truncated = candidates.len() > MAX_ENTRY_POINTS;
candidates.truncate(MAX_ENTRY_POINTS);
(candidates, truncated)
}
fn is_common_entry_point(relative_path: &str) -> bool {
let lower = relative_path.to_ascii_lowercase();
let file_name = lower.rsplit('/').next().unwrap_or(&lower);
if matches!(
file_name,
"main.rs"
| "lib.rs"
| "main.py"
| "app.py"
| "manage.py"
| "__main__.py"
| "main.go"
| "program.cs"
| "index.html"
| "index.htm"
| "index.js"
| "index.jsx"
| "index.ts"
| "index.tsx"
| "main.js"
| "main.jsx"
| "main.ts"
| "main.tsx"
| "app.js"
| "app.jsx"
| "app.ts"
| "app.tsx"
| "server.js"
| "server.ts"
) {
return true;
}
lower.starts_with("cmd/") && lower.ends_with("/main.go")
}
fn entry_point_priority(relative_path: &str) -> (usize, usize) {
let depth = path_depth(relative_path);
let lower = relative_path.to_ascii_lowercase();
let location = if depth == 1 {
0
} else if lower.starts_with("src/") || lower.contains("/src/") {
1
} else if lower.starts_with("app/") || lower.starts_with("apps/") {
2
} else {
3
};
(location, depth)
}
fn collect_git_status(root: &Path) -> RepositoryGitStatusSummary {
let Ok(root) = validate_repository_root(root) else {
return RepositoryGitStatusSummary::default();
};
let top_level = run_isolated_git(
&root,
&["rev-parse", "--is-inside-work-tree", "--show-prefix"],
);
if !top_level.spawned {
return RepositoryGitStatusSummary::default();
}
if top_level.timed_out {
return RepositoryGitStatusSummary {
available: true,
timed_out: true,
..RepositoryGitStatusSummary::default()
};
}
if !top_level.success || !git_output_confirms_top_level(&top_level.stdout) {
return RepositoryGitStatusSummary {
available: true,
..RepositoryGitStatusSummary::default()
};
}
let status = run_isolated_git(
&root,
&[
"status",
"--porcelain=v1",
"--branch",
"--untracked-files=no",
"--ignore-submodules=all",
"--no-renames",
],
);
if !status.spawned || status.timed_out || !status.success {
return RepositoryGitStatusSummary {
available: true,
is_repository: true,
timed_out: status.timed_out,
..RepositoryGitStatusSummary::default()
};
}
let output = String::from_utf8_lossy(&status.stdout);
let mut lines = output.lines();
let branch = lines
.next()
.and_then(parse_git_status_branch)
.map(|branch| sanitize_repository_text(&branch, None))
.filter(|branch| !branch.is_empty());
let tracked_dirty_count = lines.filter(|line| !line.trim().is_empty()).count();
RepositoryGitStatusSummary {
available: true,
is_repository: true,
branch,
tracked_dirty_count,
timed_out: false,
}
}
fn git_output_confirms_top_level(stdout: &[u8]) -> bool {
let output = String::from_utf8_lossy(stdout);
let Some((inside_work_tree, prefix)) = output.split_once('\n') else {
return false;
};
inside_work_tree.trim() == "true" && prefix.trim_matches(['\r', '\n']).is_empty()
}
fn run_isolated_git(root: &Path, args: &[&str]) -> BoundedGitOutput {
let mut command = isolated_git_command(root);
command.args(args);
run_bounded_git_command(command)
}
fn isolated_git_command(root: &Path) -> Command {
let inherited_environment = ["PATH", "SystemRoot", "WINDIR", "PATHEXT"]
.into_iter()
.filter_map(|key| std::env::var_os(key).map(|value| (key, value)))
.collect::<Vec<_>>();
let null_device = if cfg!(windows) { "NUL" } else { "/dev/null" };
let mut command = Command::new("git");
crate::configure_windows_background_std_command(&mut command, false);
command.env_clear();
for (key, value) in inherited_environment {
command.env(key, value);
}
command
.env("GIT_CONFIG_NOSYSTEM", "1")
.env("GIT_CONFIG_SYSTEM", null_device)
.env("GIT_CONFIG_GLOBAL", null_device)
.env("GIT_OPTIONAL_LOCKS", "0")
.env("GIT_TERMINAL_PROMPT", "0")
.env("GIT_PAGER", "cat")
.env("PAGER", "cat")
.env("TERM", "dumb")
.current_dir(root)
.arg("--no-pager")
.arg("--no-optional-locks")
.arg("-c")
.arg("core.fsmonitor=false")
.arg("-c")
.arg(format!("core.hooksPath={null_device}"))
.arg("-c")
.arg("core.pager=cat")
.arg("-c")
.arg("pager.status=false")
.stdout(Stdio::piped())
.stderr(Stdio::piped());
if let Some(parent) = root.parent() {
command.env("GIT_CEILING_DIRECTORIES", parent);
}
command
}
fn run_bounded_git_command(mut command: Command) -> BoundedGitOutput {
let Ok(mut child) = command.spawn() else {
return BoundedGitOutput::default();
};
let stdout = child.stdout.take();
let stderr = child.stderr.take();
let stdout_reader =
thread::spawn(move || read_bounded_process_output(stdout, MAX_GIT_STATUS_BYTES));
let stderr_reader =
thread::spawn(move || read_bounded_process_output(stderr, MAX_GIT_STATUS_BYTES));
let started = Instant::now();
let mut timed_out = false;
let status = loop {
match child.try_wait() {
Ok(Some(status)) => break Some(status),
Ok(None) if started.elapsed() < GIT_STATUS_TIMEOUT => {
thread::sleep(Duration::from_millis(10));
}
Ok(None) => {
timed_out = true;
let _ = child.kill();
let _ = child.wait();
break None;
}
Err(_) => {
let _ = child.kill();
let _ = child.wait();
break None;
}
}
};
let stdout = stdout_reader.join().unwrap_or_default();
let _ = stderr_reader.join();
BoundedGitOutput {
spawned: true,
success: status.is_some_and(|status| status.success()),
stdout,
timed_out,
}
}
fn read_bounded_process_output<R: Read + Send + 'static>(
stream: Option<R>,
max_bytes: usize,
) -> Vec<u8> {
let Some(mut stream) = stream else {
return Vec::new();
};
let mut collected = Vec::new();
let mut buffer = [0_u8; 8 * 1024];
while let Ok(read) = stream.read(&mut buffer) {
if read == 0 {
break;
}
let remaining = max_bytes.saturating_sub(collected.len());
collected.extend_from_slice(&buffer[..read.min(remaining)]);
}
collected
}
fn parse_git_status_branch(line: &str) -> Option<String> {
let value = line.strip_prefix("## ")?.trim();
if let Some(branch) = value.strip_prefix("No commits yet on ") {
return Some(branch.trim().to_string());
}
if let Some(branch) = value.strip_prefix("Initial commit on ") {
return Some(branch.trim().to_string());
}
if matches!(value, "HEAD (no branch)" | "HEAD (detached)") {
return None;
}
Some(
value
.split("...")
.next()
.unwrap_or(value)
.trim()
.to_string(),
)
}
fn render_prompt_sources(context: &RepositoryStartupContext) -> (String, bool) {
let mut section = BoundedPromptSection::new(MAX_PROMPT_SOURCE_BYTES);
let _ = writeln!(section, "\nsourcePaths:");
if context.source_paths.is_empty() {
let _ = writeln!(section, "- (none)");
} else {
for path in &context.source_paths {
let _ = writeln!(section, "- {}", safe_prompt_value(path));
}
}
section.finish()
}
fn render_prompt_inventory(context: &RepositoryStartupContext) -> (String, bool) {
let mut section = BoundedPromptSection::new(MAX_PROMPT_INVENTORY_BYTES);
let _ = writeln!(section, "\nscan:");
let _ = writeln!(
section,
"- scannedEntries: {}",
context.scan.scanned_entries
);
let _ = writeln!(
section,
"- candidateFiles: {}",
context.scan.candidate_files
);
let _ = writeln!(section, "- files: {}", context.scan.files);
let _ = writeln!(section, "- directories: {}", context.scan.directories);
let _ = writeln!(
section,
"- maxDepthReached: {}",
context.scan.max_depth_reached
);
let _ = writeln!(section, "topLevel:");
if context.scan.top_level.is_empty() {
let _ = writeln!(section, "- (empty)");
} else {
for entry in &context.scan.top_level {
let _ = writeln!(
section,
"- {}: {}",
safe_prompt_value(&entry.kind),
safe_prompt_value(&entry.path)
);
}
}
let _ = writeln!(section, "gitStatus:");
let _ = writeln!(section, "- available: {}", context.git_status.available);
let _ = writeln!(
section,
"- isRepository: {}",
context.git_status.is_repository
);
let _ = writeln!(
section,
"- branch: {}",
context
.git_status
.branch
.as_deref()
.map(safe_prompt_value)
.unwrap_or_else(|| "(unknown)".to_string())
);
let _ = writeln!(
section,
"- trackedDirtyCount: {}",
context.git_status.tracked_dirty_count
);
let _ = writeln!(section, "- timedOut: {}", context.git_status.timed_out);
let _ = writeln!(section, "languages:");
if context.languages.is_empty() {
let _ = writeln!(section, "- (none detected)");
} else {
for language in &context.languages {
let _ = writeln!(
section,
"- {}: {} files, {} bytes",
safe_prompt_value(&language.language),
language.files,
language.bytes
);
}
}
let _ = writeln!(section, "entryPoints:");
if context.entry_points.is_empty() {
let _ = writeln!(section, "- (none detected)");
} else {
for path in &context.entry_points {
let _ = writeln!(section, "- {}", safe_prompt_value(path));
}
}
section.finish()
}
fn render_prompt_manifests(context: &RepositoryStartupContext) -> (String, bool) {
let mut section = BoundedPromptSection::new(MAX_PROMPT_MANIFEST_BYTES);
let _ = writeln!(section, "\nmanifests:");
if context.manifests.is_empty() {
let _ = writeln!(section, "- (none detected)");
} else {
for manifest in &context.manifests {
let _ = writeln!(
section,
"- {} [{}] name={} parsed={} truncated={}",
safe_prompt_value(&manifest.path),
safe_prompt_value(&manifest.kind),
manifest
.name
.as_deref()
.map(safe_prompt_value)
.unwrap_or_else(|| "(unknown)".to_string()),
manifest.parsed,
manifest.truncated
);
for (name, command) in &manifest.scripts {
let _ = writeln!(
section,
" script {}: {}",
safe_prompt_value(name),
safe_prompt_value(command)
);
}
}
}
section.finish()
}
fn render_prompt_skills(context: &RepositoryStartupContext) -> (String, bool) {
let mut section = BoundedPromptSection::new(MAX_PROMPT_SKILL_BYTES);
let _ = writeln!(
section,
"\nPROJECT SKILL CATALOG (UNTRUSTED DISCOVERY METADATA):"
);
if context.skills.is_empty() {
let _ = writeln!(section, "- (none detected)");
return section.finish();
}
for skill in &context.skills {
let entry = serde_json::json!({
"name": safe_prompt_value(&skill.name),
"description": safe_prompt_value(&skill.description),
"entryPath": safe_prompt_value(&skill.entry_path),
"sourceRoot": safe_prompt_value(&skill.source_root),
"contentSha256": safe_prompt_value(&skill.content_sha256),
"metadataTruncated": skill.truncated,
"bodyLoaded": false,
});
let _ = writeln!(section, "- {entry}");
}
section.finish()
}
fn render_prompt_documents(context: &RepositoryStartupContext) -> (String, bool) {
let mut section = BoundedPromptSection::new(MAX_PROMPT_DOCUMENT_BYTES);
let mut body_truncated = false;
let _ = writeln!(section, "\nrepositoryDocuments:");
if context.documents.is_empty() {
let _ = writeln!(section, "- (none detected)");
return section.finish();
}
for document in &context.documents {
let role = if document.kind == "agents" {
"scoped-instructions"
} else {
"untrusted-reference"
};
let _ = writeln!(
section,
"- {} [{}] scope={} role={} sha256={} truncated={}",
safe_prompt_value(&document.path),
safe_prompt_value(&document.kind),
safe_prompt_value(&document.scope),
role,
safe_prompt_value(&document.content_sha256),
document.truncated
);
}
let _ = writeln!(section, "repositoryDocumentBodies:");
for document in &context.documents {
let safe_content = safe_prompt_value(&document.content);
let (content, clipped) = truncate_utf8_owned(safe_content, MAX_PROMPT_DOCUMENT_BODY_BYTES);
body_truncated |= clipped;
let path = safe_prompt_value(&document.path);
let scope = safe_prompt_value(&document.scope);
if document.kind == "agents" {
let _ = writeln!(
section,
"BEGIN SCOPED REPOSITORY INSTRUCTIONS path={path} scope={scope}"
);
} else {
let _ = writeln!(
section,
"BEGIN UNTRUSTED REPOSITORY REFERENCE path={path} kind={}",
safe_prompt_value(&document.kind)
);
}
let _ = writeln!(section, "{content}");
if clipped {
let _ = writeln!(section, "[file body truncated for prompt]");
}
if document.kind == "agents" {
let _ = writeln!(
section,
"END SCOPED REPOSITORY INSTRUCTIONS path={path} scope={scope}"
);
} else {
let _ = writeln!(section, "END UNTRUSTED REPOSITORY REFERENCE path={path}");
}
}
let (content, section_truncated) = section.finish();
(content, section_truncated || body_truncated)
}
fn summarize_one_line(root: &Path, value: &str, max_bytes: usize) -> (String, bool) {
let collapsed = value.split_whitespace().collect::<Vec<_>>().join(" ");
truncate_utf8_owned(sanitize_repository_text(&collapsed, Some(root)), max_bytes)
}
fn sanitize_repository_text(value: &str, root: Option<&Path>) -> String {
let mut sanitized = redact_repository_secrets(value);
if let Some(root) = root {
if root.is_absolute() {
let root = root.to_string_lossy();
if root.len() > 1 {
sanitized = sanitized.replace(root.as_ref(), "<repository-root>");
}
}
}
redact_absolute_path_tokens(&sanitized)
}
fn redact_repository_secrets(value: &str) -> String {
let value = redact_url_userinfo_secrets(value);
let mut output = String::with_capacity(value.len());
let mut index = 0;
while index < value.len() {
let redaction = known_secret_prefix_redaction_at(&value, index)
.or_else(|| sensitive_assignment_redaction_at(&value, index))
.or_else(|| bearer_redaction_at(&value, index));
if let Some((end, replacement)) = redaction {
output.push_str(&replacement);
index = end;
continue;
}
let character = value[index..].chars().next().unwrap_or_default();
output.push(character);
index += character.len_utf8();
}
output
}
fn known_secret_prefix_redaction_at(value: &str, index: usize) -> Option<(usize, String)> {
const SECRET_PREFIXES: &[&str] = &[
"github_pat_",
"ghp_",
"gho_",
"ghu_",
"ghs_",
"xoxa-",
"xoxb-",
"xoxp-",
"sk-",
"AIza",
];
if index > 0
&& value
.as_bytes()
.get(index - 1)
.is_some_and(|byte| byte.is_ascii_alphanumeric() || *byte == b'_')
{
return None;
}
let prefix = SECRET_PREFIXES
.iter()
.find(|prefix| starts_with_ignore_ascii_case(value, index, prefix))?;
let mut end = index + prefix.len();
while end < value.len() {
let character = value[end..].chars().next().unwrap_or_default();
if !is_secret_token_character(character) {
break;
}
end += character.len_utf8();
}
if end < index + prefix.len() + 4 {
return None;
}
Some((end, REDACTED_SECRET.to_string()))
}
fn sensitive_assignment_redaction_at(value: &str, index: usize) -> Option<(usize, String)> {
let bytes = value.as_bytes();
let first = *bytes.get(index)?;
if !is_secret_identifier_byte(first)
|| (index > 0 && is_secret_identifier_byte(bytes[index - 1]))
{
return None;
}
let mut identifier_end = index;
while bytes
.get(identifier_end)
.is_some_and(|byte| is_secret_identifier_byte(*byte))
{
identifier_end += 1;
}
let identifier = &value[index..identifier_end];
let normalized = normalized_secret_identifier(identifier);
if !is_sensitive_secret_identifier(&normalized) {
return None;
}
let mut separator = identifier_end;
if bytes
.get(separator)
.is_some_and(|byte| matches!(byte, b'\'' | b'"' | b'`'))
{
separator += 1;
}
let before_whitespace = separator;
separator = consume_horizontal_whitespace(bytes, separator);
let had_whitespace = separator > before_whitespace;
if bytes
.get(separator)
.is_some_and(|byte| matches!(byte, b':' | b'='))
{
separator += 1;
if bytes.get(separator) == Some(&b'>') {
separator += 1;
}
separator = consume_horizontal_whitespace(bytes, separator);
} else if identifier.starts_with("--") && had_whitespace {
// CLI flags commonly use `--api-key VALUE` without '='.
} else {
return None;
}
if separator >= value.len() || starts_redaction_marker(value, separator) {
return None;
}
let consume_line = normalized.ends_with("cookie");
let (end, redacted_value) = if normalized.ends_with("authorization") {
redact_authorization_value(value, separator)?
} else {
redact_secret_value(value, separator, consume_line)?
};
let mut replacement = String::with_capacity(end.saturating_sub(index));
replacement.push_str(&value[index..separator]);
replacement.push_str(&redacted_value);
Some((end, replacement))
}
fn bearer_redaction_at(value: &str, index: usize) -> Option<(usize, String)> {
const BEARER: &str = "Bearer";
if !starts_with_ignore_ascii_case(value, index, BEARER)
|| (index > 0 && value.as_bytes()[index - 1].is_ascii_alphanumeric())
|| value
.as_bytes()
.get(index + BEARER.len())
.is_some_and(|byte| byte.is_ascii_alphanumeric() || *byte == b'_')
{
return None;
}
let value_start = consume_horizontal_whitespace(value.as_bytes(), index + BEARER.len());
if value_start == index + BEARER.len()
|| value_start >= value.len()
|| starts_redaction_marker(value, value_start)
{
return None;
}
let (end, redacted_value) = redact_secret_value(value, value_start, false)?;
let mut replacement = String::with_capacity(end.saturating_sub(index));
replacement.push_str(&value[index..value_start]);
replacement.push_str(&redacted_value);
Some((end, replacement))
}
fn redact_authorization_value(value: &str, value_start: usize) -> Option<(usize, String)> {
if value
.as_bytes()
.get(value_start)
.is_some_and(|byte| matches!(byte, b'\'' | b'"' | b'`'))
{
return redact_secret_value(value, value_start, false);
}
for scheme in ["Bearer", "Basic", "Digest"] {
if !starts_with_ignore_ascii_case(value, value_start, scheme) {
continue;
}
let credential_start =
consume_horizontal_whitespace(value.as_bytes(), value_start + scheme.len());
if credential_start == value_start + scheme.len()
|| credential_start >= value.len()
|| starts_redaction_marker(value, credential_start)
{
return None;
}
let (end, redacted_value) = redact_secret_value(value, credential_start, false)?;
let mut replacement = String::with_capacity(end.saturating_sub(value_start));
replacement.push_str(&value[value_start..credential_start]);
replacement.push_str(&redacted_value);
return Some((end, replacement));
}
redact_secret_value(value, value_start, false)
}
fn redact_secret_value(
value: &str,
value_start: usize,
consume_line: bool,
) -> Option<(usize, String)> {
if value_start >= value.len() || starts_redaction_marker(value, value_start) {
return None;
}
let first = value[value_start..].chars().next()?;
if matches!(first, '\'' | '"' | '`') {
let content_start = value_start + first.len_utf8();
let end = find_closing_quote(value, content_start, first)
.map(|closing| closing + first.len_utf8())
.unwrap_or_else(|| line_end(value, content_start));
let content_end = if end > content_start && value[..end].ends_with(first) {
end - first.len_utf8()
} else {
end
};
if content_start == content_end || starts_redaction_marker(value, content_start) {
return None;
}
let mut replacement = String::with_capacity(REDACTED_SECRET.len() + 2);
replacement.push(first);
replacement.push_str(REDACTED_SECRET);
if content_end < end {
replacement.push(first);
}
return Some((end, replacement));
}
let mut end = if consume_line {
line_end(value, value_start)
} else {
unquoted_secret_value_end(value, value_start)
};
while end > value_start
&& value[..end]
.chars()
.next_back()
.is_some_and(char::is_whitespace)
{
end -= value[..end]
.chars()
.next_back()
.map(char::len_utf8)
.unwrap_or(1);
}
if end == value_start || starts_redaction_marker(value, value_start) {
return None;
}
Some((end, REDACTED_SECRET.to_string()))
}
fn redact_url_userinfo_secrets(value: &str) -> String {
let mut output = String::with_capacity(value.len());
let mut copied = 0;
let mut search = 0;
while let Some(relative_scheme) = value[search..].find("://") {
let authority_start = search + relative_scheme + 3;
let authority_end = value[authority_start..]
.char_indices()
.find_map(|(offset, character)| {
(character.is_whitespace()
|| matches!(character, '/' | '?' | '#' | '\'' | '"' | '`'))
.then_some(authority_start + offset)
})
.unwrap_or(value.len());
let authority = &value[authority_start..authority_end];
let Some(at_offset) = authority.rfind('@') else {
search = authority_end.max(authority_start);
continue;
};
let Some(colon_offset) = authority[..at_offset].rfind(':') else {
search = authority_end.max(authority_start);
continue;
};
let secret_start = authority_start + colon_offset + 1;
let secret_end = authority_start + at_offset;
if secret_start >= secret_end || starts_redaction_marker(value, secret_start) {
search = authority_end.max(authority_start);
continue;
}
output.push_str(&value[copied..secret_start]);
output.push_str(REDACTED_SECRET);
copied = secret_end;
search = authority_end.max(secret_end);
}
if copied == 0 {
return value.to_string();
}
output.push_str(&value[copied..]);
output
}
fn normalized_secret_identifier(identifier: &str) -> String {
identifier
.bytes()
.filter(|byte| byte.is_ascii_alphanumeric())
.map(|byte| byte.to_ascii_lowercase() as char)
.collect()
}
fn is_sensitive_secret_identifier(identifier: &str) -> bool {
[
"authorization",
"token",
"apikey",
"password",
"passwd",
"pwd",
"cookie",
"setcookie",
"clientsecret",
"secret",
"secretkey",
"accesskey",
"privatekey",
"credential",
"credentials",
]
.iter()
.any(|suffix| identifier == *suffix || identifier.ends_with(suffix))
}
fn is_secret_identifier_byte(byte: u8) -> bool {
byte.is_ascii_alphanumeric() || matches!(byte, b'_' | b'-' | b'.')
}
fn is_secret_token_character(character: char) -> bool {
character.is_ascii_alphanumeric()
|| matches!(character, '_' | '-' | '.' | '/' | '+' | '=' | '~')
}
fn starts_with_ignore_ascii_case(value: &str, index: usize, needle: &str) -> bool {
value
.get(index..index.saturating_add(needle.len()))
.is_some_and(|candidate| candidate.eq_ignore_ascii_case(needle))
}
fn starts_redaction_marker(value: &str, index: usize) -> bool {
value
.get(index..)
.is_some_and(|rest| rest.starts_with(REDACTED_SECRET))
}
fn consume_horizontal_whitespace(bytes: &[u8], mut index: usize) -> usize {
while bytes
.get(index)
.is_some_and(|byte| matches!(byte, b' ' | b'\t'))
{
index += 1;
}
index
}
fn find_closing_quote(value: &str, mut index: usize, quote: char) -> Option<usize> {
let mut escaped = false;
while index < value.len() {
let character = value[index..].chars().next()?;
if matches!(character, '\n' | '\r') {
return None;
}
if character == quote && !escaped {
return Some(index);
}
escaped = character == '\\' && !escaped;
if character != '\\' {
escaped = false;
}
index += character.len_utf8();
}
None
}
fn line_end(value: &str, start: usize) -> usize {
value[start..]
.find(['\n', '\r'])
.map(|offset| start + offset)
.unwrap_or(value.len())
}
fn unquoted_secret_value_end(value: &str, mut index: usize) -> usize {
while index < value.len() {
let character = value[index..].chars().next().unwrap_or_default();
if character.is_whitespace()
|| matches!(
character,
'\'' | '"' | '`' | ',' | ';' | '&' | ']' | '}' | ')'
)
{
break;
}
index += character.len_utf8();
}
index
}
pub(crate) fn redact_absolute_path_tokens(value: &str) -> String {
let bytes = value.as_bytes();
let mut output = String::with_capacity(value.len());
let mut index = 0;
while index < bytes.len() {
if starts_absolute_path(bytes, index) || starts_file_uri(bytes, index) {
output.push_str("<absolute-path>");
index = consume_path_token(bytes, index);
continue;
}
let character = value[index..].chars().next().unwrap_or_default();
if character.is_control() && !matches!(character, '\n' | '\r' | '\t') {
output.push(' ');
} else {
output.push(character);
}
index += character.len_utf8();
}
output
}
fn starts_absolute_path(bytes: &[u8], index: usize) -> bool {
let boundary = index == 0 || is_path_boundary(bytes[index - 1]);
if !boundary {
return false;
}
if bytes[index] == b'/' {
return bytes.get(index + 1) != Some(&b'/') || starts_forward_slash_unc_path(bytes, index);
}
if bytes[index] == b'\\' && bytes.get(index + 1) == Some(&b'\\') {
return true;
}
bytes[index].is_ascii_alphabetic()
&& bytes.get(index + 1) == Some(&b':')
&& bytes
.get(index + 2)
.is_some_and(|separator| matches!(separator, b'/' | b'\\'))
}
fn starts_forward_slash_unc_path(bytes: &[u8], index: usize) -> bool {
let server_start = index + 2;
let token_end = consume_path_token(bytes, server_start);
let Some(server_end) = bytes[server_start..token_end]
.iter()
.position(|byte| *byte == b'/')
.map(|offset| server_start + offset)
else {
return false;
};
let share_start = server_end + 1;
server_end > server_start && share_start < token_end && bytes.get(share_start) != Some(&b'/')
}
fn starts_file_uri(bytes: &[u8], index: usize) -> bool {
const FILE_URI_PREFIX: &[u8] = b"file:";
let boundary = index == 0 || is_path_boundary(bytes[index - 1]);
if !boundary {
return false;
}
let Some(prefix_end) = index.checked_add(FILE_URI_PREFIX.len()) else {
return false;
};
if !bytes
.get(index..prefix_end)
.is_some_and(|prefix| prefix.eq_ignore_ascii_case(FILE_URI_PREFIX))
{
return false;
}
let token_end = consume_path_token(bytes, prefix_end);
bytes.get(prefix_end) == Some(&b'/')
&& token_end > prefix_end
&& bytes.get(prefix_end..token_end) != Some(b"//")
}
fn is_path_boundary(byte: u8) -> bool {
byte.is_ascii_whitespace()
|| matches!(
byte,
b'"' | b'\'' | b'`' | b'=' | b'(' | b'[' | b'{' | b'<' | b',' | b';'
)
}
fn consume_path_token(bytes: &[u8], mut index: usize) -> usize {
while index < bytes.len()
&& !bytes[index].is_ascii_whitespace()
&& !matches!(
bytes[index],
b'"' | b'\'' | b'`' | b')' | b']' | b'}' | b'>' | b',' | b';'
)
{
index += 1;
}
index
}
fn safe_prompt_value(value: &str) -> String {
sanitize_repository_text(value, None)
}
fn path_depth(path: &str) -> usize {
path.split('/').filter(|part| !part.is_empty()).count()
}
fn root_to_specific_cmp(left: &str, right: &str) -> std::cmp::Ordering {
path_depth(left)
.cmp(&path_depth(right))
.then_with(|| left.cmp(right))
}
fn sort_root_to_specific(paths: &mut [String]) {
paths.sort_by(|left, right| root_to_specific_cmp(left, right));
}
fn truncate_utf8_owned(mut value: String, max_bytes: usize) -> (String, bool) {
if value.len() <= max_bytes {
return (value, false);
}
let boundary = utf8_prefix_boundary(&value, max_bytes);
value.truncate(boundary);
(value, true)
}
fn truncate_with_marker(value: &str, max_bytes: usize, marker: &str) -> String {
if value.len() <= max_bytes {
return value.to_string();
}
let content_budget = max_bytes.saturating_sub(marker.len());
let mut truncated = utf8_prefix(value, content_budget).to_string();
truncated.push_str(utf8_prefix(
marker,
max_bytes.saturating_sub(truncated.len()),
));
truncated
}
fn append_marker_within_budget(value: &mut String, max_bytes: usize, marker: &str) {
if value.len().saturating_add(marker.len()) > max_bytes {
let boundary = utf8_prefix_boundary(value, max_bytes.saturating_sub(marker.len()));
value.truncate(boundary);
}
value.push_str(utf8_prefix(marker, max_bytes.saturating_sub(value.len())));
}
fn utf8_prefix(value: &str, max_bytes: usize) -> &str {
&value[..utf8_prefix_boundary(value, max_bytes)]
}
fn utf8_prefix_boundary(value: &str, max_bytes: usize) -> usize {
let mut boundary = value.len().min(max_bytes);
while boundary > 0 && !value.is_char_boundary(boundary) {
boundary -= 1;
}
boundary
}
#[cfg(test)]
mod tests {
use super::*;
use std::sync::atomic::{AtomicU64, Ordering};
use std::time::{SystemTime, UNIX_EPOCH};
static TEST_DIRECTORY_COUNTER: AtomicU64 = AtomicU64::new(0);
struct TestDirectory {
path: PathBuf,
}
impl TestDirectory {
fn new(label: &str) -> Self {
let nonce = TEST_DIRECTORY_COUNTER.fetch_add(1, Ordering::Relaxed);
let timestamp = SystemTime::now()
.duration_since(UNIX_EPOCH)
.unwrap_or_default()
.as_nanos();
let temp_root = fs::canonicalize(std::env::temp_dir()).expect("canonical temp root");
let path = temp_root.join(format!(
"repository-context-{label}-{}-{timestamp}-{nonce}",
std::process::id()
));
fs::create_dir_all(&path).expect("create test repository");
Self { path }
}
fn write(&self, relative_path: &str, content: impl AsRef<[u8]>) {
let path = self.path.join(relative_path);
if let Some(parent) = path.parent() {
fs::create_dir_all(parent).expect("create fixture parent");
}
fs::write(path, content).expect("write fixture");
}
fn init_git(&self) -> bool {
let output = run_isolated_git(&self.path, &["init", "--quiet", "."]);
output.spawned && output.success
}
}
impl Drop for TestDirectory {
fn drop(&mut self) {
let _ = fs::remove_dir_all(&self.path);
}
}
#[test]
fn discovers_nested_agents_in_root_to_specific_order() {
let repository = TestDirectory::new("nested-agents");
repository.write("AGENTS.md", "root rule");
repository.write("art/AGENTS.md", "art sibling rule");
repository.write("game/AGENTS.md", "game rule");
repository.write("game/feature/AGENTS.md", "feature rule");
repository.write("CONTEXT.md", "project context");
repository.write(
"README.md",
format!("read from {}", repository.path.display()),
);
let context = build_repository_startup_context_at(&repository.path).unwrap();
let agent_documents = context
.documents
.iter()
.filter(|document| document.kind == "agents")
.map(|document| (document.path.as_str(), document.scope.as_str()))
.collect::<Vec<_>>();
assert_eq!(
agent_documents,
vec![
("AGENTS.md", "."),
("art/AGENTS.md", "art"),
("game/AGENTS.md", "game"),
("game/feature/AGENTS.md", "game/feature")
]
);
assert_eq!(context.schema_version, "repository-startup-context-v3");
assert!(context.source_paths.contains(&"CONTEXT.md".to_string()));
assert!(context.source_paths.contains(&"README.md".to_string()));
assert_eq!(context.fingerprint.len(), 64);
assert!(context
.fingerprint
.bytes()
.all(|byte| byte.is_ascii_hexdigit()));
let prompt = render_repository_startup_context_for_prompt(&context);
let root_rule = prompt.find("root rule").unwrap();
let game_rule = prompt.find("game rule").unwrap();
let feature_rule = prompt.find("feature rule").unwrap();
assert!(root_rule < game_rule && game_rule < feature_rule);
assert!(prompt.contains("AGENTS.md documents are scoped repository instructions"));
assert!(prompt.contains("BEGIN SCOPED REPOSITORY INSTRUCTIONS path=AGENTS.md scope=."));
assert!(
prompt.contains("BEGIN SCOPED REPOSITORY INSTRUCTIONS path=game/AGENTS.md scope=game")
);
assert!(prompt.contains(
"BEGIN SCOPED REPOSITORY INSTRUCTIONS path=game/feature/AGENTS.md scope=game/feature"
));
assert!(
prompt.contains("BEGIN SCOPED REPOSITORY INSTRUCTIONS path=art/AGENTS.md scope=art")
);
assert!(
prompt.contains("BEGIN UNTRUSTED REPOSITORY REFERENCE path=CONTEXT.md kind=context")
);
assert!(prompt.contains("sibling scopes never apply"));
assert!(!prompt.contains(repository.path.to_string_lossy().as_ref()));
assert!(prompt.contains("<repository-root>"));
}
#[test]
fn scoped_agents_path_and_content_changes_advance_the_fingerprint() {
let repository = TestDirectory::new("scoped-agents-fingerprint");
repository.write("AGENTS.md", "root rule\n");
repository.write("game/AGENTS.md", "game rule\n");
let baseline = build_repository_startup_context_at(&repository.path).unwrap();
repository.write("game/AGENTS.md", "changed game rule\n");
let content_changed = build_repository_startup_context_at(&repository.path).unwrap();
assert_ne!(baseline.fingerprint, content_changed.fingerprint);
fs::create_dir_all(repository.path.join("art")).expect("create sibling scope");
fs::rename(
repository.path.join("game/AGENTS.md"),
repository.path.join("art/AGENTS.md"),
)
.expect("move scoped instructions");
let scope_changed = build_repository_startup_context_at(&repository.path).unwrap();
assert_ne!(content_changed.fingerprint, scope_changed.fingerprint);
assert!(scope_changed.documents.iter().any(|document| {
document.path == "art/AGENTS.md" && document.kind == "agents" && document.scope == "art"
}));
assert!(!scope_changed
.documents
.iter()
.any(|document| document.scope == "game"));
}
#[test]
fn discovers_project_skill_metadata_without_preloading_bodies() {
let repository = TestDirectory::new("project-skills");
let body_marker = "SKILL_BODY_MUST_NOT_BE_PRELOADED";
let shadow_marker = "SHADOWED_SKILL_BODY_MUST_NOT_APPEAR";
repository.write(
".codex/skills/release-capsule/SKILL.md",
format!(
"---\nname: release-capsule\ndescription: >-\n Use for release capsule tasks with apiKey=sk-description-123456 and path {}.\nmetadata:\n ignored: true\n---\n# Workflow\n{body_marker}\n",
repository.path.display()
),
);
repository.write(
".agents/skills/release-capsule/SKILL.md",
format!(
"---\nname: release-capsule\ndescription: Shadowed compatibility copy.\n---\n{shadow_marker}\n"
),
);
repository.write(
".agents/skills/art-audit/SKILL.md",
"---\nname: art-audit\ndescription: Audit generated art handoffs.\n---\nART_AUDIT_BODY\n",
);
repository.write(
".codex/skills/release-capsule/references/SKILL.md",
"---\nname: nested-reference\ndescription: Must not become a skill.\n---\nNESTED_BODY\n",
);
repository.write(
".codex/skills/Bad-Name/SKILL.md",
"---\nname: Bad-Name\ndescription: Invalid directory name.\n---\nINVALID_BODY\n",
);
repository.write(
".codex/skills/mismatch/SKILL.md",
"---\nname: another-name\ndescription: Mismatched name.\n---\nMISMATCH_BODY\n",
);
let context = build_repository_startup_context_at(&repository.path).unwrap();
assert_eq!(context.schema_version, "repository-startup-context-v3");
assert_eq!(
context
.skills
.iter()
.map(|skill| (
skill.name.as_str(),
skill.entry_path.as_str(),
skill.source_root.as_str()
))
.collect::<Vec<_>>(),
vec![
("art-audit", ".agents/skills/art-audit/SKILL.md", ".agents"),
(
"release-capsule",
".codex/skills/release-capsule/SKILL.md",
".codex"
)
]
);
let release = context
.skills
.iter()
.find(|skill| skill.name == "release-capsule")
.unwrap();
assert!(release.description.contains(REDACTED_SECRET));
assert!(release.description.contains("<repository-root>"));
assert!(!release.description.contains("sk-description-123456"));
assert!(!release
.description
.contains(repository.path.to_string_lossy().as_ref()));
assert_eq!(release.content_sha256.len(), 64);
assert!(context
.source_paths
.contains(&release.entry_path.to_string()));
assert!(!context
.source_paths
.contains(&".agents/skills/release-capsule/SKILL.md".to_string()));
let prompt = render_repository_startup_context_for_prompt(&context);
assert!(prompt.contains("PROJECT SKILL CATALOG (UNTRUSTED DISCOVERY METADATA)"));
assert!(prompt.contains("release-capsule"));
assert!(prompt.contains(".codex/skills/release-capsule/SKILL.md"));
assert!(prompt.contains("\"bodyLoaded\":false"));
assert!(prompt.contains("use approved file.read on the exact entryPath"));
for forbidden in [
body_marker,
shadow_marker,
"ART_AUDIT_BODY",
"NESTED_BODY",
"INVALID_BODY",
"MISMATCH_BODY",
"sk-description-123456",
] {
assert!(
!prompt.contains(forbidden),
"skill body leaked: {forbidden}"
);
}
}
#[test]
fn active_skill_content_and_precedence_control_the_fingerprint() {
let repository = TestDirectory::new("skill-fingerprint");
repository.write(
".codex/skills/release-capsule/SKILL.md",
"---\nname: release-capsule\ndescription: Primary skill.\n---\nPRIMARY_V1\n",
);
repository.write(
".agents/skills/release-capsule/SKILL.md",
"---\nname: release-capsule\ndescription: Compatibility skill.\n---\nSHADOW_V1\n",
);
let baseline = build_repository_startup_context_at(&repository.path).unwrap();
assert_eq!(baseline.skills.len(), 1);
assert_eq!(baseline.skills[0].source_root, ".codex");
repository.write(
".agents/skills/release-capsule/SKILL.md",
"---\nname: release-capsule\ndescription: Compatibility skill.\n---\nSHADOW_V2\n",
);
let shadow_changed = build_repository_startup_context_at(&repository.path).unwrap();
assert_eq!(baseline.fingerprint, shadow_changed.fingerprint);
repository.write(
".codex/skills/release-capsule/SKILL.md",
"---\nname: release-capsule\ndescription: Primary skill.\n---\nPRIMARY_V2\n",
);
let active_changed = build_repository_startup_context_at(&repository.path).unwrap();
assert_ne!(shadow_changed.fingerprint, active_changed.fingerprint);
fs::remove_file(
repository
.path
.join(".codex/skills/release-capsule/SKILL.md"),
)
.unwrap();
let fallback = build_repository_startup_context_at(&repository.path).unwrap();
assert_eq!(fallback.skills.len(), 1);
assert_eq!(fallback.skills[0].source_root, ".agents");
assert_eq!(
fallback.skills[0].entry_path,
".agents/skills/release-capsule/SKILL.md"
);
assert_ne!(active_changed.fingerprint, fallback.fingerprint);
}
#[test]
fn enforces_project_skill_count_and_file_budgets() {
let count_repository = TestDirectory::new("skill-count-budget");
for index in 0..=MAX_SKILLS {
count_repository.write(
&format!(".codex/skills/skill-{index:03}/SKILL.md"),
format!(
"---\nname: skill-{index:03}\ndescription: Skill {index}.\n---\nBODY_{index}\n"
),
);
}
let count_context = build_repository_startup_context_at(&count_repository.path).unwrap();
assert_eq!(count_context.skills.len(), MAX_SKILLS);
assert!(count_context.truncated);
let size_repository = TestDirectory::new("skill-size-budget");
let mut oversized = "---\nname: oversized\ndescription: Oversized skill.\n---\n"
.as_bytes()
.to_vec();
oversized.resize(MAX_SKILL_FILE_BYTES + 1, b'x');
size_repository.write(".codex/skills/oversized/SKILL.md", oversized);
let size_context = build_repository_startup_context_at(&size_repository.path).unwrap();
assert!(size_context.skills.is_empty());
assert!(size_context.truncated);
}
#[cfg(unix)]
#[test]
fn project_skill_discovery_rejects_symbolic_link_entries() {
use std::os::unix::fs::symlink;
let repository = TestDirectory::new("skill-symlink");
repository.write(
"outside-skill.md",
"---\nname: linked\ndescription: Linked skill.\n---\nLINKED_BODY\n",
);
fs::create_dir_all(repository.path.join(".codex/skills/linked")).unwrap();
symlink(
repository.path.join("outside-skill.md"),
repository.path.join(".codex/skills/linked/SKILL.md"),
)
.unwrap();
let context = build_repository_startup_context_at(&repository.path).unwrap();
assert!(context.skills.is_empty());
assert!(!render_repository_startup_context_for_prompt(&context).contains("LINKED_BODY"));
}
#[cfg(unix)]
#[test]
fn skips_sensitive_paths_and_symbolic_links() {
use std::os::unix::fs::symlink;
let repository = TestDirectory::new("sensitive-symlink");
repository.write(".env.local", "ENV_SECRET_MARKER");
repository.write("secrets/AGENTS.md", "DIRECTORY_SECRET_MARKER");
repository.write("actual-secret.txt", "SYMLINK_SECRET_MARKER");
fs::create_dir_all(repository.path.join("nested")).unwrap();
symlink(
"../actual-secret.txt",
repository.path.join("nested/AGENTS.md"),
)
.unwrap();
repository.write("src/main.rs", "fn main() {}");
let context = build_repository_startup_context_at(&repository.path).unwrap();
let prompt = render_repository_startup_context_for_prompt(&context);
assert!(context.documents.is_empty());
assert!(!context
.scan
.top_level
.iter()
.any(|entry| matches!(entry.path.as_str(), ".env.local" | "secrets")));
assert!(!context
.source_paths
.iter()
.any(|path| path.contains("nested/AGENTS.md")));
assert!(!prompt.contains("ENV_SECRET_MARKER"));
assert!(!prompt.contains("DIRECTORY_SECRET_MARKER"));
assert!(!prompt.contains("SYMLINK_SECRET_MARKER"));
}
#[test]
fn requires_an_absolute_repository_root() {
let error = build_repository_startup_context_at(Path::new("relative-project"))
.expect_err("relative roots must be rejected");
assert!(error.contains("absolute path"));
}
#[cfg(unix)]
#[test]
fn rejects_symbolic_link_ancestors_and_no_follow_reads_link_targets() {
use std::os::unix::fs::symlink;
let target = TestDirectory::new("symlink-target");
fs::create_dir_all(target.path.join("project")).unwrap();
target.write("secret.txt", "NO_FOLLOW_SECRET");
let container = TestDirectory::new("symlink-container");
symlink(&target.path, container.path.join("linked-parent")).unwrap();
let linked_root = container.path.join("linked-parent/project");
let error = build_repository_startup_context_at(&linked_root)
.expect_err("a symlinked ancestor must be rejected");
assert!(error.contains("ancestors"));
symlink(
target.path.join("secret.txt"),
container.path.join("secret-link.txt"),
)
.unwrap();
assert!(
read_bounded_regular_file(&container.path.join("secret-link.txt"), 1024)
.unwrap()
.is_none()
);
assert!(
read_bounded_regular_file(&target.path.join("secret.txt"), 1024)
.unwrap()
.is_some()
);
}
#[test]
fn git_status_requires_the_canonical_root_to_be_the_top_level() {
let repository = TestDirectory::new("git-top-level");
if !repository.init_git() {
return;
}
fs::create_dir_all(repository.path.join("nested")).unwrap();
let root_status = collect_git_status(&repository.path);
let nested_status = collect_git_status(&repository.path.join("nested"));
assert!(root_status.available);
assert!(root_status.is_repository);
assert!(nested_status.available);
assert!(!nested_status.is_repository);
}
#[cfg(unix)]
#[test]
fn git_status_disables_repository_fsmonitor_and_external_process_hooks() {
use std::os::unix::fs::PermissionsExt;
let repository = TestDirectory::new("git-isolation");
if !repository.init_git() {
return;
}
repository.write("tracked.txt", "tracked");
assert!(run_isolated_git(&repository.path, &["add", "tracked.txt"]).success);
let marker = repository.path.join("fsmonitor-invoked");
let script = repository.path.join("malicious-fsmonitor.sh");
repository.write(
"malicious-fsmonitor.sh",
format!(
"#!/bin/sh\nprintf invoked > '{}'\nprintf '0\\n'\n",
marker.display()
),
);
let mut permissions = fs::metadata(&script).unwrap().permissions();
permissions.set_mode(0o755);
fs::set_permissions(&script, permissions).unwrap();
assert!(
run_isolated_git(
&repository.path,
&[
"config",
"--local",
"core.fsmonitor",
script.to_string_lossy().as_ref(),
],
)
.success
);
let status = collect_git_status(&repository.path);
assert!(status.is_repository);
assert!(!marker.exists(), "repository fsmonitor must never execute");
let command = isolated_git_command(&repository.path);
let args = command
.get_args()
.map(|argument| argument.to_string_lossy().into_owned())
.collect::<Vec<_>>();
assert!(args.iter().any(|argument| argument == "--no-pager"));
assert!(args
.iter()
.any(|argument| argument == "core.fsmonitor=false"));
assert!(args
.iter()
.any(|argument| argument.starts_with("core.hooksPath=")));
let environment = command
.get_envs()
.filter_map(|(key, value)| {
value.map(|value| {
(
key.to_string_lossy().into_owned(),
value.to_string_lossy().into_owned(),
)
})
})
.collect::<BTreeMap<_, _>>();
assert_eq!(
environment.get("GIT_CONFIG_NOSYSTEM").map(String::as_str),
Some("1")
);
assert_eq!(
environment.get("GIT_CONFIG_GLOBAL").map(String::as_str),
Some("/dev/null")
);
assert_eq!(
environment.get("GIT_PAGER").map(String::as_str),
Some("cat")
);
}
#[test]
fn enforces_document_and_prompt_budgets() {
let repository = TestDirectory::new("budgets");
repository.write("AGENTS.md", "A".repeat(30 * 1024));
repository.write("nested/AGENTS.md", "B".repeat(30 * 1024));
repository.write("CONTEXT.md", "C".repeat(30 * 1024));
let context = build_repository_startup_context_at(&repository.path).unwrap();
let total_body_bytes = context
.documents
.iter()
.map(|document| document.content.len())
.sum::<usize>();
assert!(context
.documents
.iter()
.all(|document| document.content.len() <= MAX_DOCUMENT_BYTES));
assert!(total_body_bytes <= MAX_DOCUMENT_BODY_BYTES);
assert!(context.truncated);
let prompt = render_repository_startup_context_for_prompt(&context);
assert!(prompt.len() <= MAX_PROMPT_BYTES);
assert!(prompt.contains("truncated: true"));
}
#[test]
fn redacts_manifest_and_document_credentials_without_redacting_lookalikes() {
let repository = TestDirectory::new("secret-redaction");
repository.write(
"package.json",
r#"{
"name": "redaction-fixture",
"scripts": {
"check": "curl -H 'Authorization: Bearer bearer-script-secret' --api-key=sk-script-123456789 --cookie='session=cookie-script-secret' https://example.test",
"test": "PASSWORD=script-password cargo test"
}
}"#,
);
repository.write(
"AGENTS.md",
r#"Authorization: Bearer bearer-document-secret
OPENAI_API_KEY=sk-document-123456789
password = "document-password"
cookie: session=cookie-document-secret; theme=dark
DATABASE_URL=postgres://user:database-password@localhost/game
token_budget = 4096
passwordless = true
cookie_policy: strict
sketch-color = green
"#,
);
repository.write(
"CONTEXT.md",
"Authorization: Basic QWxhZGRpbjpvcGVuIHNlc2FtZQ==\n",
);
repository.write(
"README.md",
"auth_token: readme-token-secret\nAuthorization mode: OAuth\n",
);
let context = build_repository_startup_context_at(&repository.path).unwrap();
let serialized = serde_json::to_string(&context).unwrap();
let prompt = render_repository_startup_context_for_prompt(&context);
let combined = format!("{serialized}\n{prompt}");
for secret in [
"bearer-script-secret",
"sk-script-123456789",
"cookie-script-secret",
"script-password",
"bearer-document-secret",
"sk-document-123456789",
"document-password",
"cookie-document-secret",
"database-password",
"QWxhZGRpbjpvcGVuIHNlc2FtZQ==",
"readme-token-secret",
] {
assert!(!combined.contains(secret), "credential leaked: {secret}");
}
assert!(combined.contains(REDACTED_SECRET));
assert!(combined.contains("token_budget = 4096"));
assert!(combined.contains("passwordless = true"));
assert!(combined.contains("cookie_policy: strict"));
assert!(combined.contains("sketch-color = green"));
assert!(combined.contains("Authorization mode: OAuth"));
for document in &context.documents {
assert_eq!(
document.content_sha256,
format!("{:x}", Sha256::digest(document.content.as_bytes()))
);
}
}
#[test]
fn redacts_file_uri_tokens_without_redacting_relative_text() {
assert_eq!(
redact_absolute_path_tokens(
"open file:///home/alice/project, FILE:///C:/Users/Alice/project; \
file://server/share/project file:///home/alice/My%20Project \
file:///%68ome/alice%2Fproject"
),
"open <absolute-path>, <absolute-path>; <absolute-path> <absolute-path> \
<absolute-path>"
);
assert_eq!(
redact_absolute_path_tokens(
"profile homeward docs/file.txt file:notes.txt file:// file:///"
),
"profile homeward docs/file.txt file:notes.txt file:// <absolute-path>"
);
}
#[test]
fn redacts_cross_platform_absolute_paths_without_redacting_urls() {
assert_eq!(
redact_absolute_path_tokens(concat!(
r#"open //server/share/private, \\server\share\private; "#,
"file:/home/user/private file:/C:/Users/Alice/private ",
r#"C:/Users/Alice/private C:\Users\Alice\private"#
)),
"open <absolute-path>, <absolute-path>; \
<absolute-path> <absolute-path> \
<absolute-path> <absolute-path>"
);
assert_eq!(
redact_absolute_path_tokens(
"keep docs/private file:notes.txt // not-a-path /// docs \
https://example.test/private http://localhost:3000/private"
),
"keep docs/private file:notes.txt // not-a-path /// docs \
https://example.test/private http://localhost:3000/private"
);
}
#[test]
fn secret_changes_and_git_status_do_not_change_the_fingerprint() {
let repository = TestDirectory::new("stable-fingerprint");
repository.write(
"package.json",
r#"{"scripts":{"check":"API_KEY=first-api-key-value cargo check"}}"#,
);
repository.write(
"AGENTS.md",
"Authorization: Bearer first-bearer-value\nrule = keep\n",
);
let first = build_repository_startup_context_at(&repository.path).unwrap();
repository.write(
"package.json",
r#"{"scripts":{"check":"API_KEY=a-much-longer-second-api-key-value cargo check"}}"#,
);
repository.write(
"AGENTS.md",
"Authorization: Bearer a-much-longer-second-bearer-value\nrule = keep\n",
);
let second = build_repository_startup_context_at(&repository.path).unwrap();
assert_eq!(first.fingerprint, second.fingerprint);
let mut volatile_git = second.clone();
volatile_git.git_status = RepositoryGitStatusSummary {
available: true,
is_repository: true,
branch: Some("another-branch".to_string()),
tracked_dirty_count: 999,
timed_out: true,
};
assert_eq!(
second.fingerprint,
repository_startup_context_fingerprint(&volatile_git)
);
repository.write(
"AGENTS.md",
"Authorization: Bearer third-bearer-value\nrule = changed\n",
);
let changed = build_repository_startup_context_at(&repository.path).unwrap();
assert_ne!(second.fingerprint, changed.fingerprint);
}
#[test]
fn agent_control_plane_changes_do_not_change_the_fingerprint() {
let repository = TestDirectory::new("stable-agent-control-plane-fingerprint");
repository.write("package.json", r#"{"scripts":{"check":"cargo check"}}"#);
repository.write("AGENTS.md", "rule = keep\n");
repository.write(".agent/manifest.json", r#"{"projectId":"project-1"}"#);
let before = build_repository_startup_context_at(&repository.path).unwrap();
repository.write(
".agent/checkpoints/checkpoint-1/manifest.json",
r#"{"checkpointId":"checkpoint-1"}"#,
);
repository.write(".agent/logs/command.log", "internal command output\n");
repository.write(
".agent/runtime/agents/code-prototype.json",
r#"{"status":"running"}"#,
);
let after = build_repository_startup_context_at(&repository.path).unwrap();
assert_eq!(before.fingerprint, after.fingerprint);
assert!(!after
.scan
.top_level
.iter()
.any(|entry| entry.path == ".agent"));
}
#[test]
fn caps_document_count_and_total_context_objects() {
let repository = TestDirectory::new("document-count");
for index in 0..=MAX_DOCUMENTS {
repository.write(&format!("scope-{index:03}/AGENTS.md"), "rule");
}
let context = build_repository_startup_context_at(&repository.path).unwrap();
assert_eq!(context.documents.len(), MAX_DOCUMENTS);
assert!(context.truncated);
assert!(context
.source_paths
.contains(&format!("scope-{MAX_DOCUMENTS:03}/AGENTS.md")));
let mut oversized = RepositoryStartupContext {
scan: RepositoryScanSummary {
top_level: vec![RepositoryTopLevelEntry::default(); MAX_CONTEXT_OBJECTS],
..RepositoryScanSummary::default()
},
manifests: vec![RepositoryManifestSummary::default(); MAX_CONTEXT_OBJECTS],
documents: vec![RepositoryContextDocument::default(); MAX_CONTEXT_OBJECTS],
languages: vec![RepositoryLanguageSummary::default(); MAX_CONTEXT_OBJECTS],
entry_points: vec![String::new(); MAX_CONTEXT_OBJECTS],
source_paths: vec![String::new(); MAX_CONTEXT_OBJECTS],
..RepositoryStartupContext::default()
};
assert!(enforce_context_object_budget(&mut oversized));
let object_count = oversized.scan.top_level.len()
+ oversized.manifests.len()
+ oversized.documents.len()
+ oversized.languages.len()
+ oversized.entry_points.len()
+ oversized.source_paths.len();
assert!(object_count <= MAX_CONTEXT_OBJECTS);
assert!(oversized.documents.len() <= MAX_DOCUMENTS);
assert!(oversized.source_paths.len() <= MAX_SOURCE_PATHS);
}
#[test]
fn enforces_candidate_file_budget() {
let repository = TestDirectory::new("candidate-budget");
for index in 0..=MAX_CANDIDATE_FILES {
repository.write(&format!("src/file-{index:04}.rs"), "pub fn fixture() {}");
}
let context = build_repository_startup_context_at(&repository.path).unwrap();
assert_eq!(context.scan.candidate_files, MAX_CANDIDATE_FILES);
assert_eq!(context.scan.files, MAX_CANDIDATE_FILES + 1);
assert!(context.truncated);
}
#[test]
fn parses_package_scripts_and_detects_supported_manifests() {
let repository = TestDirectory::new("manifests");
repository.write(
"package.json",
r#"{
"name": "context-fixture",
"scripts": {
"test": "cargo test --workspace",
"dev": "vite --host 127.0.0.1"
}
}"#,
);
repository.write(
"crates/tool/Cargo.toml",
"[package]\nname = \"fixture-tool\"\nversion = \"0.1.0\"\n",
);
repository.write(
"python/pyproject.toml",
"[project]\nname = \"fixture-python\"\n",
);
repository.write("go/go.mod", "module example.com/fixture\n\ngo 1.23\n");
let context = build_repository_startup_context_at(&repository.path).unwrap();
let package = context
.manifests
.iter()
.find(|manifest| manifest.kind == "package.json")
.unwrap();
assert_eq!(package.name.as_deref(), Some("context-fixture"));
assert_eq!(
package.scripts.get("test").map(String::as_str),
Some("cargo test --workspace")
);
assert_eq!(
context
.manifests
.iter()
.map(|manifest| manifest.kind.as_str())
.collect::<BTreeSet<_>>(),
BTreeSet::from(["Cargo.toml", "go.mod", "package.json", "pyproject.toml"])
);
assert!(render_repository_startup_context_for_prompt(&context)
.contains("script test: cargo test --workspace"));
}
}