Files
iris/src/check.rs
T

226 lines
8.8 KiB
Rust
Raw Normal View History

//! Headless schema-check binary - loads content and `aggregates.yaml`
//! (from a Gitea repo URL or a local directory) and validates them
//! exactly the way `content::watch_for_reload`/`main.rs`'s boot path
//! do, with no NATS, OIDC, web server, or JetStream connection
//! involved. Built once by portal's own deploy workflow and run
//! directly by `questions`' own CI (same bare-metal runner/host,
//! published to a stable path - no artifact download needed), rather
//! than compiled there - keeps that repo's CI coupling to "run a
//! static binary," not "build a Rust workspace."
use crate::local::{load_aggregates_from_dir, load_from_dir, load_needs_from_dir, load_site_from_dir};
use crate::needs;
use portal::{aggregates, content};
pub async fn run(argv: Vec<String>) -> anyhow::Result<()> {
let mut args = argv.into_iter();
let mut repo: Option<String> = None;
let mut branch = "main".to_string();
let mut subdir = "questions".to_string();
let mut path: Option<String> = None;
let mut tasks_out: Option<String> = None;
let mut answers_in: Option<String> = None;
let mut min_accuracy: f64 = 0.0;
let mut skim = false;
let mut sim_out: Option<String> = None;
while let Some(arg) = args.next() {
match arg.as_str() {
"--repo" => repo = args.next(),
"--branch" => branch = args.next().unwrap_or(branch),
"--subdir" => subdir = args.next().unwrap_or(subdir),
"--path" => path = args.next(),
"--needs-tasks" => tasks_out = args.next(),
"--skim" => skim = true,
"--needs-sim" => sim_out = args.next(),
"--needs-score" => answers_in = args.next(),
"--min-accuracy" => {
min_accuracy = args.next().and_then(|v| v.parse().ok()).unwrap_or(min_accuracy)
}
other => {
eprintln!("unknown argument: {other}");
std::process::exit(2);
}
}
}
let (questions, aggregates_map, site, needs_raw) = if let Some(dir) = path {
(
load_from_dir(&dir)?,
load_aggregates_from_dir(&dir)?,
load_site_from_dir(&dir)?,
load_needs_from_dir(&dir)?,
)
} else if let Some(repo_url) = repo {
let questions = content::load_questions_from_gitea(&repo_url, &branch, &subdir).await?;
let aggregates_map = content::load_aggregates_from_gitea(&repo_url, &branch).await?;
// load_site_from_gitea already validates; a missing file is
// the default config, same as at portal boot.
let site = content::load_site_from_gitea(&repo_url, &branch).await?;
let needs_raw = content::load_needs_from_gitea(&repo_url, &branch).await?;
(questions, aggregates_map, site, needs_raw)
} else {
eprintln!(
"usage: iris check (--repo <gitea-url> [--branch main] [--subdir questions] | --path <local-dir>) [--needs-tasks <out.jsonl> [--skim]] [--needs-sim <model.json>] [--needs-score <answers.jsonl> [--min-accuracy 0.8]]"
);
std::process::exit(2);
};
if let Err(e) = site.validate() {
eprintln!("FAIL: {e}");
std::process::exit(1);
}
let aggregates_map = content::with_builtin_aggregates(aggregates_map);
match content::validate_questions(&questions, &aggregates_map) {
Ok(()) => {
// The runtime's own event desk: a repo announcing pages
// (Question.event) should read portal_events somewhere, or
// "post what happened" tasks pile up unseen. Warn, don't
// fail - the runtime creates the records either way.
if questions.values().any(|q| q.event.is_some())
&& !content::bucket_is_read(&questions, content::EVENTS_BUCKET)
{
eprintln!(
"WARN: pages carry `event:` but no kv resource reads bucket {:?} - add a desk so summaries get posted",
content::EVENTS_BUCKET
);
}
println!(
"OK: {} question(s), {} aggregate(s) valid",
questions.len(),
aggregates_map.len()
);
check_needs(
needs_raw.as_deref(),
&questions,
&aggregates_map,
tasks_out.as_deref(),
skim,
sim_out.as_deref(),
answers_in.as_deref(),
min_accuracy,
)
}
Err(e) => {
eprintln!("FAIL: {e}");
std::process::exit(1);
}
}
}
/// The business-needs pass (see `portal::needs`): structure always,
/// then the wording tasks written out and/or an engine's answers
/// scored, when asked. A repo without `needs.yaml` skips all of it.
fn check_needs(
raw: Option<&str>,
questions: &std::collections::HashMap<String, content::Question>,
aggregates_map: &std::collections::HashMap<String, aggregates::AggregateSchema>,
tasks_out: Option<&str>,
skim: bool,
sim_out: Option<&str>,
answers_in: Option<&str>,
min_accuracy: f64,
) -> anyhow::Result<()> {
let Some(raw) = raw else {
if tasks_out.is_some() || answers_in.is_some() {
eprintln!("FAIL: --needs-tasks/--needs-score given, but the repo has no needs.yaml");
std::process::exit(1);
}
return Ok(());
};
let file = match needs::parse(raw) {
Ok(file) => file,
Err(e) => {
eprintln!("FAIL: {e}");
std::process::exit(1);
}
};
let findings = needs::check(&file, questions, aggregates_map);
let mut failed = 0;
for finding in &findings {
let level = match finding.level {
needs::Level::Fail => {
failed += 1;
"FAIL"
}
needs::Level::Warn => "WARN",
};
eprintln!("{level}: need {:?}: {}", finding.need, finding.message);
}
for need in &file.needs {
if let Some(path) = needs::path_for(questions, need) {
let steps: Vec<String> = path
.iter()
.map(|s| format!("{} [{}]", s.page, s.alternative))
.collect();
println!("need {:?}: {} -> {:?}", need.id, steps.join(" -> "), need.lands_in);
}
}
if failed == 0 {
let planned = file.needs.iter().filter(|n| n.planned).count();
println!(
"OK: {} need(s) structurally met, {} planned, {} persona(s)",
file.needs.len() - planned,
planned,
file.personas.len()
);
}
// The wording pass still runs for the needs that do have a path -
// a closed flow elsewhere is no reason to stop grading the rest.
if let Some(out) = sim_out {
let model = needs::sim_model(&file, questions, aggregates_map);
std::fs::write(out, serde_json::to_string_pretty(&model)?)
.map_err(|e| anyhow::anyhow!("writing {out}: {e}"))?;
println!("wrote the simulation model to {out}");
}
let tasks = needs::tasks(&file, questions, skim);
if let Some(out) = tasks_out {
let mut lines = String::new();
for task in &tasks {
lines.push_str(&serde_json::to_string(task)?);
lines.push('\n');
}
std::fs::write(out, lines).map_err(|e| anyhow::anyhow!("writing {out}: {e}"))?;
println!("wrote {} wording task(s) to {out}", tasks.len());
}
if let Some(path) = answers_in {
let raw = std::fs::read_to_string(path).map_err(|e| anyhow::anyhow!("reading {path}: {e}"))?;
let mut answers = Vec::new();
for (n, line) in raw.lines().enumerate().filter(|(_, l)| !l.trim().is_empty()) {
answers.push(
serde_json::from_str::<needs::Answer>(line)
.map_err(|e| anyhow::anyhow!("{path}:{}: {e}", n + 1))?,
);
}
let score = needs::score(&tasks, &answers, 0.6);
for (id, picked, expected, confidence) in &score.misroutes {
eprintln!(
"MISROUTE: {id}: picked {picked:?}{}, the need is met by {expected:?}",
confidence.map(|c| format!(" ({c:.2})")).unwrap_or_default()
);
}
for (id, confidence) in &score.hesitant {
eprintln!("HESITANT: {id}: right, at {confidence:.2}");
}
for id in &score.unanswered {
eprintln!("UNANSWERED: {id}");
}
println!(
"wording: {}/{} routed right ({:.0}%)",
score.correct,
score.total,
score.accuracy() * 100.0
);
if score.accuracy() < min_accuracy {
eprintln!("FAIL: below --min-accuracy {min_accuracy}");
std::process::exit(1);
}
}
if failed > 0 {
eprintln!("FAIL: {failed} unmet business need finding(s)");
std::process::exit(1);
}
Ok(())
}