226 lines
8.8 KiB
Rust
226 lines
8.8 KiB
Rust
//! Headless schema-check binary - loads content and `aggregates.yaml`
|
|||
|
|
//! (from a Gitea repo URL or a local directory) and validates them
|
||
|
|
//! exactly the way `content::watch_for_reload`/`main.rs`'s boot path
|
||
|
|
//! do, with no NATS, OIDC, web server, or JetStream connection
|
||
|
|
//! involved. Built once by portal's own deploy workflow and run
|
||
|
|
//! directly by `questions`' own CI (same bare-metal runner/host,
|
||
|
|
//! published to a stable path - no artifact download needed), rather
|
||
|
|
//! than compiled there - keeps that repo's CI coupling to "run a
|
||
|
|
//! static binary," not "build a Rust workspace."
|
||
|
|
|
||
|
|
use crate::local::{load_aggregates_from_dir, load_from_dir, load_needs_from_dir, load_site_from_dir};
|
||
|
|
use crate::needs;
|
||
|
|
use portal::{aggregates, content};
|
||
|
|
|
||
|
|
pub async fn run(argv: Vec<String>) -> anyhow::Result<()> {
|
||
|
|
let mut args = argv.into_iter();
|
||
|
|
let mut repo: Option<String> = None;
|
||
|
|
let mut branch = "main".to_string();
|
||
|
|
let mut subdir = "questions".to_string();
|
||
|
|
let mut path: Option<String> = None;
|
||
|
|
let mut tasks_out: Option<String> = None;
|
||
|
|
let mut answers_in: Option<String> = None;
|
||
|
|
let mut min_accuracy: f64 = 0.0;
|
||
|
|
let mut skim = false;
|
||
|
|
let mut sim_out: Option<String> = None;
|
||
|
|
|
||
|
|
while let Some(arg) = args.next() {
|
||
|
|
match arg.as_str() {
|
||
|
|
"--repo" => repo = args.next(),
|
||
|
|
"--branch" => branch = args.next().unwrap_or(branch),
|
||
|
|
"--subdir" => subdir = args.next().unwrap_or(subdir),
|
||
|
|
"--path" => path = args.next(),
|
||
|
|
"--needs-tasks" => tasks_out = args.next(),
|
||
|
|
"--skim" => skim = true,
|
||
|
|
"--needs-sim" => sim_out = args.next(),
|
||
|
|
"--needs-score" => answers_in = args.next(),
|
||
|
|
"--min-accuracy" => {
|
||
|
|
min_accuracy = args.next().and_then(|v| v.parse().ok()).unwrap_or(min_accuracy)
|
||
|
|
}
|
||
|
|
other => {
|
||
|
|
eprintln!("unknown argument: {other}");
|
||
|
|
std::process::exit(2);
|
||
|
|
}
|
||
|
|
}
|
||
|
|
}
|
||
|
|
|
||
|
|
let (questions, aggregates_map, site, needs_raw) = if let Some(dir) = path {
|
||
|
|
(
|
||
|
|
load_from_dir(&dir)?,
|
||
|
|
load_aggregates_from_dir(&dir)?,
|
||
|
|
load_site_from_dir(&dir)?,
|
||
|
|
load_needs_from_dir(&dir)?,
|
||
|
|
)
|
||
|
|
} else if let Some(repo_url) = repo {
|
||
|
|
let questions = content::load_questions_from_gitea(&repo_url, &branch, &subdir).await?;
|
||
|
|
let aggregates_map = content::load_aggregates_from_gitea(&repo_url, &branch).await?;
|
||
|
|
// load_site_from_gitea already validates; a missing file is
|
||
|
|
// the default config, same as at portal boot.
|
||
|
|
let site = content::load_site_from_gitea(&repo_url, &branch).await?;
|
||
|
|
let needs_raw = content::load_needs_from_gitea(&repo_url, &branch).await?;
|
||
|
|
(questions, aggregates_map, site, needs_raw)
|
||
|
|
} else {
|
||
|
|
eprintln!(
|
||
|
|
"usage: iris check (--repo <gitea-url> [--branch main] [--subdir questions] | --path <local-dir>) [--needs-tasks <out.jsonl> [--skim]] [--needs-sim <model.json>] [--needs-score <answers.jsonl> [--min-accuracy 0.8]]"
|
||
|
|
);
|
||
|
|
std::process::exit(2);
|
||
|
|
};
|
||
|
|
|
||
|
|
if let Err(e) = site.validate() {
|
||
|
|
eprintln!("FAIL: {e}");
|
||
|
|
std::process::exit(1);
|
||
|
|
}
|
||
|
|
let aggregates_map = content::with_builtin_aggregates(aggregates_map);
|
||
|
|
match content::validate_questions(&questions, &aggregates_map) {
|
||
|
|
Ok(()) => {
|
||
|
|
// The runtime's own event desk: a repo announcing pages
|
||
|
|
// (Question.event) should read portal_events somewhere, or
|
||
|
|
// "post what happened" tasks pile up unseen. Warn, don't
|
||
|
|
// fail - the runtime creates the records either way.
|
||
|
|
if questions.values().any(|q| q.event.is_some())
|
||
|
|
&& !content::bucket_is_read(&questions, content::EVENTS_BUCKET)
|
||
|
|
{
|
||
|
|
eprintln!(
|
||
|
|
"WARN: pages carry `event:` but no kv resource reads bucket {:?} - add a desk so summaries get posted",
|
||
|
|
content::EVENTS_BUCKET
|
||
|
|
);
|
||
|
|
}
|
||
|
|
println!(
|
||
|
|
"OK: {} question(s), {} aggregate(s) valid",
|
||
|
|
questions.len(),
|
||
|
|
aggregates_map.len()
|
||
|
|
);
|
||
|
|
check_needs(
|
||
|
|
needs_raw.as_deref(),
|
||
|
|
&questions,
|
||
|
|
&aggregates_map,
|
||
|
|
tasks_out.as_deref(),
|
||
|
|
skim,
|
||
|
|
sim_out.as_deref(),
|
||
|
|
answers_in.as_deref(),
|
||
|
|
min_accuracy,
|
||
|
|
)
|
||
|
|
}
|
||
|
|
Err(e) => {
|
||
|
|
eprintln!("FAIL: {e}");
|
||
|
|
std::process::exit(1);
|
||
|
|
}
|
||
|
|
}
|
||
|
|
}
|
||
|
|
|
||
|
|
/// The business-needs pass (see `portal::needs`): structure always,
|
||
|
|
/// then the wording tasks written out and/or an engine's answers
|
||
|
|
/// scored, when asked. A repo without `needs.yaml` skips all of it.
|
||
|
|
fn check_needs(
|
||
|
|
raw: Option<&str>,
|
||
|
|
questions: &std::collections::HashMap<String, content::Question>,
|
||
|
|
aggregates_map: &std::collections::HashMap<String, aggregates::AggregateSchema>,
|
||
|
|
tasks_out: Option<&str>,
|
||
|
|
skim: bool,
|
||
|
|
sim_out: Option<&str>,
|
||
|
|
answers_in: Option<&str>,
|
||
|
|
min_accuracy: f64,
|
||
|
|
) -> anyhow::Result<()> {
|
||
|
|
let Some(raw) = raw else {
|
||
|
|
if tasks_out.is_some() || answers_in.is_some() {
|
||
|
|
eprintln!("FAIL: --needs-tasks/--needs-score given, but the repo has no needs.yaml");
|
||
|
|
std::process::exit(1);
|
||
|
|
}
|
||
|
|
return Ok(());
|
||
|
|
};
|
||
|
|
let file = match needs::parse(raw) {
|
||
|
|
Ok(file) => file,
|
||
|
|
Err(e) => {
|
||
|
|
eprintln!("FAIL: {e}");
|
||
|
|
std::process::exit(1);
|
||
|
|
}
|
||
|
|
};
|
||
|
|
let findings = needs::check(&file, questions, aggregates_map);
|
||
|
|
let mut failed = 0;
|
||
|
|
for finding in &findings {
|
||
|
|
let level = match finding.level {
|
||
|
|
needs::Level::Fail => {
|
||
|
|
failed += 1;
|
||
|
|
"FAIL"
|
||
|
|
}
|
||
|
|
needs::Level::Warn => "WARN",
|
||
|
|
};
|
||
|
|
eprintln!("{level}: need {:?}: {}", finding.need, finding.message);
|
||
|
|
}
|
||
|
|
for need in &file.needs {
|
||
|
|
if let Some(path) = needs::path_for(questions, need) {
|
||
|
|
let steps: Vec<String> = path
|
||
|
|
.iter()
|
||
|
|
.map(|s| format!("{} [{}]", s.page, s.alternative))
|
||
|
|
.collect();
|
||
|
|
println!("need {:?}: {} -> {:?}", need.id, steps.join(" -> "), need.lands_in);
|
||
|
|
}
|
||
|
|
}
|
||
|
|
if failed == 0 {
|
||
|
|
let planned = file.needs.iter().filter(|n| n.planned).count();
|
||
|
|
println!(
|
||
|
|
"OK: {} need(s) structurally met, {} planned, {} persona(s)",
|
||
|
|
file.needs.len() - planned,
|
||
|
|
planned,
|
||
|
|
file.personas.len()
|
||
|
|
);
|
||
|
|
}
|
||
|
|
// The wording pass still runs for the needs that do have a path -
|
||
|
|
// a closed flow elsewhere is no reason to stop grading the rest.
|
||
|
|
|
||
|
|
if let Some(out) = sim_out {
|
||
|
|
let model = needs::sim_model(&file, questions, aggregates_map);
|
||
|
|
std::fs::write(out, serde_json::to_string_pretty(&model)?)
|
||
|
|
.map_err(|e| anyhow::anyhow!("writing {out}: {e}"))?;
|
||
|
|
println!("wrote the simulation model to {out}");
|
||
|
|
}
|
||
|
|
let tasks = needs::tasks(&file, questions, skim);
|
||
|
|
if let Some(out) = tasks_out {
|
||
|
|
let mut lines = String::new();
|
||
|
|
for task in &tasks {
|
||
|
|
lines.push_str(&serde_json::to_string(task)?);
|
||
|
|
lines.push('\n');
|
||
|
|
}
|
||
|
|
std::fs::write(out, lines).map_err(|e| anyhow::anyhow!("writing {out}: {e}"))?;
|
||
|
|
println!("wrote {} wording task(s) to {out}", tasks.len());
|
||
|
|
}
|
||
|
|
if let Some(path) = answers_in {
|
||
|
|
let raw = std::fs::read_to_string(path).map_err(|e| anyhow::anyhow!("reading {path}: {e}"))?;
|
||
|
|
let mut answers = Vec::new();
|
||
|
|
for (n, line) in raw.lines().enumerate().filter(|(_, l)| !l.trim().is_empty()) {
|
||
|
|
answers.push(
|
||
|
|
serde_json::from_str::<needs::Answer>(line)
|
||
|
|
.map_err(|e| anyhow::anyhow!("{path}:{}: {e}", n + 1))?,
|
||
|
|
);
|
||
|
|
}
|
||
|
|
let score = needs::score(&tasks, &answers, 0.6);
|
||
|
|
for (id, picked, expected, confidence) in &score.misroutes {
|
||
|
|
eprintln!(
|
||
|
|
"MISROUTE: {id}: picked {picked:?}{}, the need is met by {expected:?}",
|
||
|
|
confidence.map(|c| format!(" ({c:.2})")).unwrap_or_default()
|
||
|
|
);
|
||
|
|
}
|
||
|
|
for (id, confidence) in &score.hesitant {
|
||
|
|
eprintln!("HESITANT: {id}: right, at {confidence:.2}");
|
||
|
|
}
|
||
|
|
for id in &score.unanswered {
|
||
|
|
eprintln!("UNANSWERED: {id}");
|
||
|
|
}
|
||
|
|
println!(
|
||
|
|
"wording: {}/{} routed right ({:.0}%)",
|
||
|
|
score.correct,
|
||
|
|
score.total,
|
||
|
|
score.accuracy() * 100.0
|
||
|
|
);
|
||
|
|
if score.accuracy() < min_accuracy {
|
||
|
|
eprintln!("FAIL: below --min-accuracy {min_accuracy}");
|
||
|
|
std::process::exit(1);
|
||
|
|
}
|
||
|
|
}
|
||
|
|
if failed > 0 {
|
||
|
|
eprintln!("FAIL: {failed} unmet business need finding(s)");
|
||
|
|
std::process::exit(1);
|
||
|
|
}
|
||
|
|
Ok(())
|
||
|
|
}
|