//! Headless schema-check binary - loads content and `aggregates.yaml` //! (from a Gitea repo URL or a local directory) and validates them //! exactly the way `content::watch_for_reload`/`main.rs`'s boot path //! do, with no NATS, OIDC, web server, or JetStream connection //! involved. Built once by portal's own deploy workflow and run //! directly by `questions`' own CI (same bare-metal runner/host, //! published to a stable path - no artifact download needed), rather //! than compiled there - keeps that repo's CI coupling to "run a //! static binary," not "build a Rust workspace." use crate::local::{load_aggregates_from_dir, load_from_dir, load_needs_from_dir, load_site_from_dir}; use crate::needs; use portal::{aggregates, content}; pub async fn run(argv: Vec) -> anyhow::Result<()> { let mut args = argv.into_iter(); let mut repo: Option = None; let mut branch = "main".to_string(); let mut subdir = "questions".to_string(); let mut path: Option = None; let mut tasks_out: Option = None; let mut answers_in: Option = None; let mut min_accuracy: f64 = 0.0; let mut skim = false; let mut sim_out: Option = None; let mut show_access = false; while let Some(arg) = args.next() { match arg.as_str() { "--repo" => repo = args.next(), "--branch" => branch = args.next().unwrap_or(branch), "--subdir" => subdir = args.next().unwrap_or(subdir), "--path" => path = args.next(), "--needs-tasks" => tasks_out = args.next(), "--skim" => skim = true, "--needs-sim" => sim_out = args.next(), "--access" => show_access = true, "--needs-score" => answers_in = args.next(), "--min-accuracy" => { min_accuracy = args.next().and_then(|v| v.parse().ok()).unwrap_or(min_accuracy) } other => { eprintln!("unknown argument: {other}"); std::process::exit(2); } } } let (questions, aggregates_map, site, needs_raw) = if let Some(dir) = path { ( load_from_dir(&dir)?, load_aggregates_from_dir(&dir)?, load_site_from_dir(&dir)?, load_needs_from_dir(&dir)?, ) } else if let Some(repo_url) = repo { let questions = content::load_questions_from_gitea(&repo_url, &branch, &subdir).await?; let aggregates_map = content::load_aggregates_from_gitea(&repo_url, &branch).await?; // load_site_from_gitea already validates; a missing file is // the default config, same as at portal boot. let site = content::load_site_from_gitea(&repo_url, &branch).await?; // Optional: a fetch failure means the repo declares no needs. let needs_raw = portal::http::get_text(&format!("{}/needs.yaml?ref={branch}", content::gitea_raw_base(&repo_url)?), portal::http::GITEA_CAP).await.ok(); (questions, aggregates_map, site, needs_raw) } else { eprintln!( "usage: iris check (--repo [--branch main] [--subdir questions] | --path ) [--needs-tasks [--skim]] [--needs-sim ] [--access] [--needs-score [--min-accuracy 0.8]]" ); std::process::exit(2); }; if let Err(e) = site.validate() { eprintln!("FAIL: {e}"); std::process::exit(1); } let aggregates_map = content::with_builtin_aggregates(aggregates_map); match content::validate_questions(&questions, &aggregates_map) { Ok(()) => { // The runtime's own event desk: a repo announcing pages // (Question.event) should read portal_events somewhere, or // "post what happened" tasks pile up unseen. Warn, don't // fail - the runtime creates the records either way. if questions.values().any(|q| q.event.is_some()) && !content::bucket_is_read(&questions, content::EVENTS_BUCKET) { eprintln!( "WARN: pages carry `event:` but no kv resource reads bucket {:?} - add a desk so summaries get posted", content::EVENTS_BUCKET ); } println!( "OK: {} question(s), {} aggregate(s) valid", questions.len(), aggregates_map.len() ); // The access policy the site will compile from this content. // It must type-check against portal's schema: a rule the // schema cannot express would grant nothing at runtime. let policy = portal::access::Policy::from_content(&questions, &aggregates_map); if let Err(e) = policy.validate() { eprintln!("FAIL: access policy does not validate: {e}"); std::process::exit(1); } println!("OK: access policy, {} rule(s), validates", policy.rules.len()); if show_access { println!("\n{:<16} {:<44} {:<40} {}", "ACTION", "RESOURCE", "WHO", "WHEN"); let mut rows = policy.rules.clone(); rows.sort_by(|a, b| (&a.resource, &a.action, &a.who).cmp(&(&b.resource, &b.action, &b.who))); for r in rows { println!("{:<16} {:<44} {:<40} {}", r.action, r.resource, r.who, r.when); } println!(); } check_needs( needs_raw.as_deref(), &questions, &aggregates_map, tasks_out.as_deref(), skim, sim_out.as_deref(), answers_in.as_deref(), min_accuracy, ) } Err(e) => { eprintln!("FAIL: {e}"); std::process::exit(1); } } } /// The business-needs pass (see `crate::needs`): structure always, /// then the wording tasks written out and/or an engine's answers /// scored, when asked. A repo without `needs.yaml` skips all of it. fn check_needs( raw: Option<&str>, questions: &std::collections::HashMap, aggregates_map: &std::collections::HashMap, tasks_out: Option<&str>, skim: bool, sim_out: Option<&str>, answers_in: Option<&str>, min_accuracy: f64, ) -> anyhow::Result<()> { let Some(raw) = raw else { if tasks_out.is_some() || answers_in.is_some() { eprintln!("FAIL: --needs-tasks/--needs-score given, but the repo has no needs.yaml"); std::process::exit(1); } return Ok(()); }; let file = match needs::parse(raw) { Ok(file) => file, Err(e) => { eprintln!("FAIL: {e}"); std::process::exit(1); } }; let findings = needs::check(&file, questions, aggregates_map); let mut failed = 0; for finding in &findings { let level = match finding.level { needs::Level::Fail => { failed += 1; "FAIL" } needs::Level::Warn => "WARN", }; eprintln!("{level}: need {:?}: {}", finding.need, finding.message); } for need in &file.needs { if let Some(path) = needs::path_for(questions, need) { let steps: Vec = path .iter() .map(|s| format!("{} [{}]", s.page, s.alternative)) .collect(); println!("need {:?}: {} -> {:?}", need.id, steps.join(" -> "), need.lands_in); } } if failed == 0 { let planned = file.needs.iter().filter(|n| n.planned).count(); println!( "OK: {} need(s) structurally met, {} planned, {} persona(s)", file.needs.len() - planned, planned, file.personas.len() ); } // The wording pass still runs for the needs that do have a path - // a closed flow elsewhere is no reason to stop grading the rest. if let Some(out) = sim_out { let model = needs::sim_model(&file, questions, aggregates_map); std::fs::write(out, serde_json::to_string_pretty(&model)?) .map_err(|e| anyhow::anyhow!("writing {out}: {e}"))?; println!("wrote the simulation model to {out}"); } let tasks = needs::tasks(&file, questions, skim); if let Some(out) = tasks_out { let mut lines = String::new(); for task in &tasks { lines.push_str(&serde_json::to_string(task)?); lines.push('\n'); } std::fs::write(out, lines).map_err(|e| anyhow::anyhow!("writing {out}: {e}"))?; println!("wrote {} wording task(s) to {out}", tasks.len()); } if let Some(path) = answers_in { let raw = std::fs::read_to_string(path).map_err(|e| anyhow::anyhow!("reading {path}: {e}"))?; let mut answers = Vec::new(); for (n, line) in raw.lines().enumerate().filter(|(_, l)| !l.trim().is_empty()) { answers.push( serde_json::from_str::(line) .map_err(|e| anyhow::anyhow!("{path}:{}: {e}", n + 1))?, ); } let score = needs::score(&tasks, &answers, 0.6); for (id, picked, expected, confidence) in &score.misroutes { eprintln!( "MISROUTE: {id}: picked {picked:?}{}, the need is met by {expected:?}", confidence.map(|c| format!(" ({c:.2})")).unwrap_or_default() ); } for (id, confidence) in &score.hesitant { eprintln!("HESITANT: {id}: right, at {confidence:.2}"); } for id in &score.unanswered { eprintln!("UNANSWERED: {id}"); } println!( "wording: {}/{} routed right ({:.0}%)", score.correct, score.total, score.accuracy() * 100.0 ); if score.accuracy() < min_accuracy { eprintln!("FAIL: below --min-accuracy {min_accuracy}"); std::process::exit(1); } } if failed > 0 { eprintln!("FAIL: {failed} unmet business need finding(s)"); std::process::exit(1); } Ok(()) }