Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
9 changes: 9 additions & 0 deletions docs/research-cache.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,9 @@
# Content-addressed research cache

Standard and Deep runs can reuse fresh specialist JSON artifacts from a prior equivalent run. The cache key includes normalized company identity, research depth, fixed model and browser-tool profile, the complete job prompt, and a digest of every agent template. A prompt, model, tool profile, depth, company, or template change therefore produces a cold miss.

Only specialist JSON objects are cached. Malformed JSON, stream logs, credentials, Apollo contact results, verifier output, and synthesizer output are excluded. The verifier and synthesizer always rerun over the selected specialist set.

Entries expire after seven days. Restoration rejects malformed keys, schema mismatches, stale or future-dated manifests, symlinks, non-files, logs, malformed JSON, and derived-agent files. Cache failure never converts a failed research job to success.

The unit fixture covers deterministic normalization, material-input invalidation, TTL expiry, and derived-output exclusion. Operational targets for a representative unchanged Standard rerun are at least 70% fewer specialist model calls and p95 under 30 seconds; measure those in release telemetry before making a public performance claim.
42 changes: 36 additions & 6 deletions src-tauri/src/jobs/queue.rs
Original file line number Diff line number Diff line change
Expand Up @@ -405,16 +405,31 @@ impl JobQueue {
crate::orchestration::ResearchDepth::Light
};

let cache_key = if execution_depth != crate::orchestration::ResearchDepth::Light {
Some(crate::orchestration::research_cache_key(
&entity_label,
execution_depth,
&format!("claude-opus-4-7;chrome={}", settings.use_chrome),
&prompt,
))
} else {
None
};
let effective_working_dir = if execution_depth
!= crate::orchestration::ResearchDepth::Light
{
let prepared =
crate::orchestration::prepare_job_workspace(&app, &job_id, execution_depth)?;
let prepared = crate::orchestration::prepare_job_workspace(
&app,
&job_id,
execution_depth,
cache_key.as_deref(),
)?;
eprintln!(
"[job_queue] job_id={} Prepared orchestration workspace at {:?} with agents: {:?}",
"[job_queue] job_id={} Prepared orchestration workspace at {:?} with agents: {:?}; cache hits: {:?}",
job_id,
prepared.path,
prepared.agent_names
prepared.agent_names,
prepared.cache_hit_agents,
);
prepared.path.to_string_lossy().to_string()
} else {
Expand All @@ -438,10 +453,10 @@ impl JobQueue {
output_path: Some(metadata.primary_output_path.to_string_lossy().to_string()),
};
crate::db::insert_job(&conn, &new_job).map_err(|e| e.to_string())?;
Ok((settings, effective_working_dir, execution_depth))
Ok((settings, effective_working_dir, execution_depth, cache_key))
};

let (settings, effective_working_dir, execution_depth) = match setup_result {
let (settings, effective_working_dir, execution_depth, cache_key) = match setup_result {
Ok(result) => result,
Err(e) => {
active_jobs.lock().await.remove(&job_id);
Expand Down Expand Up @@ -841,6 +856,21 @@ impl JobQueue {

update_job_status(&final_status, result.1, final_error_msg.as_deref());

if final_success {
if let Some(ref key) = cache_key {
if let Err(error) = crate::orchestration::persist_research_cache(
&app_clone,
key,
&PathBuf::from(&effective_working_dir),
) {
eprintln!(
"[job_queue] job_id={} Could not update research cache: {}",
job_id_clone, error
);
}
}
}

if let Err(e) = on_event.send(StreamEvent {
job_id: job_id_clone.clone(),
event_type: final_status.clone(),
Expand Down
1 change: 1 addition & 0 deletions src-tauri/src/lib.rs
Original file line number Diff line number Diff line change
Expand Up @@ -5,6 +5,7 @@ mod events;
mod jobs;
mod orchestration;
mod prompts;
mod research_cache;

use db::{get_db_path, DbState};
use jobs::JobQueue;
Expand Down
71 changes: 69 additions & 2 deletions src-tauri/src/orchestration.rs
Original file line number Diff line number Diff line change
Expand Up @@ -5,6 +5,9 @@ use tauri::{AppHandle, Manager};

use crate::db::Settings;
use crate::prompts;
use crate::research_cache::ResearchCache;

const RESEARCH_CACHE_TTL_MS: i64 = 7 * 24 * 60 * 60 * 1_000;

#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum ResearchDepth {
Expand Down Expand Up @@ -59,6 +62,7 @@ impl ResearchDepth {
pub struct PreparedWorkspace {
pub path: PathBuf,
pub agent_names: Vec<&'static str>,
pub cache_hit_agents: Vec<String>,
}

struct AgentTemplate {
Expand All @@ -75,6 +79,7 @@ pub fn prepare_job_workspace(
app: &AppHandle,
job_id: &str,
depth: ResearchDepth,
cache_key: Option<&str>,
) -> Result<PreparedWorkspace, String> {
let workspace = app
.path()
Expand All @@ -97,21 +102,82 @@ pub fn prepare_job_workspace(
fs::write(path, template.content).map_err(|e| e.to_string())?;
}

let cache_hit_agents = if let Some(cache_key) = cache_key {
let cache = ResearchCache::new(
app.path()
.app_data_dir()
.unwrap_or_else(|_| PathBuf::from("."))
.join("research-cache"),
RESEARCH_CACHE_TTL_MS,
)?;
cache
.restore(
cache_key,
&workspace.join("outputs").join("specialists"),
chrono::Utc::now().timestamp_millis(),
)?
.agents
} else {
Vec::new()
};

fs::write(
workspace.join("README.md"),
format!(
"# Augur OS Research Job\n\nDepth: `{}`\n\nSpecialist outputs must be written to `outputs/specialists/`.\n",
depth.as_str()
"# Augur OS Research Job\n\nDepth: `{}`\n\nSpecialist outputs must be written to `outputs/specialists/`.\n\nFresh cached specialists: {}. The verifier and synthesizer are never cached and must always run.\n",
depth.as_str(),
if cache_hit_agents.is_empty() { "none".to_string() } else { cache_hit_agents.join(", ") }
),
)
.map_err(|e| e.to_string())?;

Ok(PreparedWorkspace {
path: workspace,
agent_names,
cache_hit_agents,
})
}

pub fn research_cache_key(
company: &str,
depth: ResearchDepth,
model: &str,
prompt: &str,
) -> String {
use sha2::{Digest, Sha256};
let mut version = Sha256::new();
for template in templates_for_depth(depth) {
version.update(template.name.as_bytes());
version.update(template.content.as_bytes());
}
ResearchCache::key(
company,
depth.as_str(),
model,
prompt,
&format!("{:x}", version.finalize()),
)
}

pub fn persist_research_cache(
app: &AppHandle,
cache_key: &str,
workspace: &std::path::Path,
) -> Result<Vec<String>, String> {
let cache = ResearchCache::new(
app.path()
.app_data_dir()
.unwrap_or_else(|_| PathBuf::from("."))
.join("research-cache"),
RESEARCH_CACHE_TTL_MS,
)?;
cache.store(
cache_key,
&workspace.join("outputs").join("specialists"),
chrono::Utc::now().timestamp_millis(),
)
}

pub fn cleanup_job_workspace(app: &AppHandle, job_id: &str) {
let workspace = app
.path()
Expand Down Expand Up @@ -163,6 +229,7 @@ Execution plan:

Rules:
- Run independent Wave 1 specialists in parallel.
- Before launching a specialist, check whether its JSON artifact already exists from the fresh content-addressed cache. Reuse a valid cached specialist artifact; never reuse `verifier.json` or synthesizer output.
- Each specialist must write its strict JSON envelope to `outputs/specialists/<agent-name>.json`.
- Each specialist must write a short progress log to `outputs/specialists/<agent-name>.stream.log`.
- Each specialist must return only this compact pointer: `{{"agent":"<name>","status":"completed","path":"outputs/specialists/<name>.json","streamLog":"outputs/specialists/<name>.stream.log"}}`.
Expand Down
Loading