From 58db264e3b10f3a41aa4093742b11f08bf21bfaa Mon Sep 17 00:00:00 2001 From: Simon Gonzalez De Cruz Date: Fri, 2 Oct 2026 04:59:23 +0000 Subject: [PATCH 1/4] fix: make history evidence auditable for DevArch 0.4.0 --- .github/workflows/archaeology.yml | 2 +- CHANGELOG.md | 4 + README.md | 4 + archaeology/__init__.py | 2 +- archaeology/analysis_runner.py | 83 ++++++------- archaeology/audit.py | 34 ++++++ archaeology/cli.py | 175 +++++---------------------- archaeology/db/builder.py | 6 +- archaeology/extractors/git.py | 90 +++++++++----- archaeology/metrics.py | 43 +++++++ archaeology/report.py | 3 +- archaeology/utils.py | 17 +-- archaeology/visualization/history.py | 30 +++++ docs/RELEASE_0.4.0.md | 26 ++++ setup.py | 2 +- tests/test_cli_coverage.py | 16 +-- tests/test_release_pipeline.py | 100 +++++++++++++++ 17 files changed, 387 insertions(+), 250 deletions(-) create mode 100644 archaeology/metrics.py create mode 100644 archaeology/visualization/history.py create mode 100644 docs/RELEASE_0.4.0.md create mode 100644 tests/test_release_pipeline.py diff --git a/.github/workflows/archaeology.yml b/.github/workflows/archaeology.yml index cf6431d..925a46b 100644 --- a/.github/workflows/archaeology.yml +++ b/.github/workflows/archaeology.yml @@ -68,7 +68,7 @@ jobs: run: devarch export-report "$PROJECT_NAME" - name: Run audit - run: devarch audit "$PROJECT_NAME" || true + run: devarch audit "$PROJECT_NAME" - name: Upload archaeology report uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 diff --git a/CHANGELOG.md b/CHANGELOG.md index d2a5c38..57911a2 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,3 +1,7 @@ +# 0.4.0 — Evidence integrity + +Fix full-history mining for bare repositories/worktrees, reject shallow coverage, bind extraction artifacts, reconcile CSV/SQLite metrics, and generate portable measured visualizations. Replace unsupported agent/ML/quality assertions with explicit uncertainty. See [release notes](docs/RELEASE_0.4.0.md) for output compatibility and validation. + # Changelog All notable changes to DevArch Framework will be documented in this file. diff --git a/README.md b/README.md index 721305c..b5e7b72 100644 --- a/README.md +++ b/README.md @@ -17,6 +17,10 @@ DevArch treats your git history as structured data. It extracts commits into a q > **Just want a quick learning diagnostic?** [Dev Learning Archaeologist](https://github.com/KyaniteLabs/dev-learning-archaeologist) is a zero-setup ICM folder — drop it in any project, run through Claude Code, no install required. +## Evidence integrity (0.4.0) + +Mining covers all **locally available refs**, records coverage and rejects shallow history. Fetch the authorized branches/tags/PR refs before mining when remote completeness matters. Automated vectors are keyword-based investigation leads, not verified source conclusions, productivity measurements or causal explanations. The default visualization contains measured daily activity only. See [0.4.0 release notes](docs/RELEASE_0.4.0.md). + ## What It Does DevArch transforms git history into structured insights through a full-featured CLI with 20+ commands. The framework supports: diff --git a/archaeology/__init__.py b/archaeology/__init__.py index 1c3c775..353897b 100644 --- a/archaeology/__init__.py +++ b/archaeology/__init__.py @@ -1,3 +1,3 @@ """DevArch Framework - forensic mining of software development history.""" -__version__ = "0.1.0" +__version__ = "0.4.0" diff --git a/archaeology/analysis_runner.py b/archaeology/analysis_runner.py index 7fff30b..070450e 100644 --- a/archaeology/analysis_runner.py +++ b/archaeology/analysis_runner.py @@ -45,16 +45,14 @@ def _log(self, msg: str) -> None: def _query_db(self, query: str, params: tuple = ()) -> list[dict]: """Execute SQL query against archaeology database.""" if not self.db_path.exists(): - return [] + raise ValueError("Analysis database missing; run build-db first") conn = sqlite3.connect(str(self.db_path), timeout=30) conn.row_factory = sqlite3.Row try: cursor = conn.execute(query, params) return [dict(row) for row in cursor.fetchall()] except sqlite3.Error as e: - if self.verbose: - print(f" [analysis] Database query error: {e}") - return [] + raise ValueError(f"Analysis query failed: {e}") from e finally: conn.close() @@ -62,11 +60,11 @@ def _load_json(self, rel_path: str) -> Any | None: """Load JSON from project directory (wrapper for utils._load_json).""" return _load_json(self.project_dir / rel_path) - def _like_commits(self, keywords: list[str], limit: int = 100) -> list[dict]: + def _like_commits(self, keywords: list[str], limit: int | None = 100) -> list[dict]: if not keywords: return [] clauses = " OR ".join("LOWER(message) LIKE ?" for _ in keywords) - params = tuple(f"%{kw.lower()}%" for kw in keywords) + (limit,) + params = tuple(f"%{kw.lower()}%" for kw in keywords) + (-1 if limit is None else limit,) return self._query_db( f"SELECT hash, date, message, author FROM commits WHERE {clauses} ORDER BY date DESC LIMIT ?", params, @@ -80,16 +78,16 @@ def run_sdlc_gap_finder(self) -> dict[str, Any]: """Analyze SDLC practices and gaps.""" self._log("Running SDLC Gap Finder...") total_commits = self._commit_count() - ci_cd = self._like_commits(["github action", "ci", "workflow", "deploy", "pipeline"], 500) - tests = self._like_commits(["test", "spec", "coverage", "vitest", "pytest"], 500) - refactor = self._like_commits(["refactor", "clean", "simplify"], 500) - security = self._like_commits(["security", "cve", "xss", "injection", "secret"], 500) - docs = self._like_commits(["docs", "readme", "documentation"], 500) + ci_cd = self._like_commits(["github action", "ci", "workflow", "deploy", "pipeline"], None) + tests = self._like_commits(["test", "spec", "coverage", "vitest", "pytest"], None) + refactor = self._like_commits(["refactor", "clean", "simplify"], None) + security = self._like_commits(["security", "cve", "xss", "injection", "secret"], None) + docs = self._like_commits(["docs", "readme", "documentation"], None) def status(count: int, low: float, high: float) -> str: ratio = count / total_commits if total_commits else 0 if ratio < low: - return "ABSENT" + return "UNVERIFIED" if ratio < high: return "EMERGING" return "PRESENT" @@ -104,12 +102,14 @@ def status(count: int, low: float, high: float) -> str: gaps = [] for practice, rows, low, high, recommendation in practices: practice_status = status(len(rows), low, high) - severity = "HIGH" if practice_status == "ABSENT" else "MEDIUM" if practice_status == "EMERGING" else "LOW" + severity = "MEDIUM" if practice_status == "UNVERIFIED" else "MEDIUM" if practice_status == "EMERGING" else "LOW" gaps.append( { "practice": practice, "status": practice_status, - "evidence": [{"result_count": len(rows), "ratio": f"{(len(rows) / total_commits if total_commits else 0):.1%}"}], + "confidence": "LOW", + "interpretation": "Commit-keyword frequency only; not verified presence, absence or coverage", + "evidence": [{"sample": rows[:5], "result_count": len(rows), "ratio": f"{(len(rows) / total_commits if total_commits else 0):.1%}"}], "severity": severity, "effort_to_implement": 3 if severity == "HIGH" else 2, "expected_impact": 5 if severity == "HIGH" else 3, @@ -148,11 +148,12 @@ def run_ml_pattern_mapper(self) -> dict[str, Any]: { "intuitive_name": intuitive, "formal_term": formal, - "confidence": "HIGH" if len(evidence) >= 5 else "MEDIUM", - "similarity_to_canonical": min(0.9, 0.45 + len(evidence) * 0.05), - "is_reinvention": reinvention, + "confidence": "LOW", + "status": "UNVERIFIED keyword candidate; inspect source before assigning an algorithm", + "similarity_to_canonical": None, + "is_reinvention": None, "library_alternative": library, - "estimated_token_waste": 5000 if reinvention else None, + "estimated_token_waste": None, "evidence": evidence[:5], } ) @@ -213,28 +214,18 @@ def _approximate_sessions(self) -> list[dict]: def run_agentic_workflow(self) -> dict[str, Any]: """Analyze AI agent interaction patterns.""" self._log("Running Agentic Workflow Analyzer...") - sessions = self._approximate_sessions() hooks = self._like_commits(["hook", "pre-commit", "post-commit", "automation"], 50) - agent_commits = self._query_db("SELECT author, COUNT(*) as cnt FROM commits GROUP BY author ORDER BY cnt DESC") + authors = self._query_db("SELECT author, COUNT(*) as cnt FROM commits GROUP BY author ORDER BY cnt DESC") return { "project": self.project_name, "analysis_date": datetime.now().isoformat(), - "session_depth_distribution": { - "sessions_total": len(sessions), - "micro_lt5": max(0, len(sessions) // 6), - "standard_5_20": max(0, len(sessions) // 2), - "deep_20_50": max(0, len(sessions) // 4), - "marathon_50_plus": max(0, len(sessions) - (len(sessions) // 6 + len(sessions) // 2 + len(sessions) // 4)), - }, - "session_taxonomy": { - "SCAFFOLDING": len(self._like_commits(["scaffold", "initialize", "setup"], 100)), - "BUILDING": len(self._like_commits(["feat", "implement", "add"], 100)), - "DEBUGGING": len(self._like_commits(["fix", "debug", "error"], 100)), - "REFACTORING": len(self._like_commits(["refactor", "cleanup", "simplify"], 100)), - }, - "hook_effectiveness": [{"hook_name": "automation/hook commits", "effectiveness_score": 0.8, "evidence_count": len(hooks)}] if hooks else [], - "agent_attribution": agent_commits, - "summary": {"total_sessions_analyzed": len(sessions), "dominant_session_type": "BUILDING"}, + "session_depth_distribution": None, + "session_taxonomy": None, + "hook_effectiveness": [], + "hook_commit_evidence": hooks, + "author_attribution": authors, + "limitations": "Commit authors are not verified agent identities. Session depth, autonomy and hook effectiveness are unmeasured; no fabricated estimates.", + "summary": {"total_sessions_analyzed": 0, "dominant_session_type": None}, } def run_formal_terms_mapper(self) -> dict[str, Any]: @@ -256,7 +247,8 @@ def run_formal_terms_mapper(self) -> dict[str, Any]: "code_name": code_name, "formal_term": formal, "category": "ARCHITECTURE", - "similarity_score": "CLOSE" if len(evidence) >= 3 else "PARTIAL", + "similarity_score": "UNVERIFIED", + "confidence": "LOW", "evidence": evidence, } ) @@ -264,7 +256,7 @@ def run_formal_terms_mapper(self) -> dict[str, Any]: "project": self.project_name, "analysis_date": datetime.now().isoformat(), "term_dictionary": dictionary, - "naming_trajectory": "Project-specific metaphors are increasingly mapped onto formal control-loop, pipeline, and verification vocabulary.", + "naming_trajectory": "Unmeasured; keyword candidates require source validation.", "learning_opportunities": ["Control theory", "Quality-diversity algorithms", "Event sourcing", "Multi-agent evaluation"], "summary": {"terms_mapped": len(dictionary), "high_confidence": sum(1 for t in dictionary if t["similarity_score"] == "CLOSE")}, } @@ -272,7 +264,7 @@ def run_formal_terms_mapper(self) -> dict[str, Any]: def run_source_archaeologist(self) -> dict[str, Any]: """Mine commit history for code quality trajectory and hotspots.""" self._log("Running Source Code Archaeologist...") - quality = self._like_commits(["fix", "test", "refactor", "security", "lint", "type"], 500) + quality = self._like_commits(["fix", "test", "refactor", "security", "lint", "type"], None) large_change = self._like_commits(["split", "extract", "monolith", "decompose", "simplify"], 100) todo = self._like_commits(["todo", "stub", "placeholder", "not implemented"], 100) by_month: Counter[str] = Counter() @@ -284,7 +276,7 @@ def run_source_archaeologist(self) -> dict[str, Any]: improvements = self._derive_improvements(quality, large_change, todo, hotspots) return { "analysis_metadata": {"timestamp": datetime.now().isoformat(), "analyst": "Automated Source Code Archaeologist", "project": self.project_name, "commit_count": self._commit_count()}, - "quality_trajectory": {"assessment": "IMPROVING" if quality else "UNKNOWN", "evidence_count": len(quality), "by_month": dict(sorted(by_month.items()))}, + "quality_trajectory": {"assessment": "UNVERIFIED keyword activity; no quality direction established", "evidence_count": len(quality), "by_month": dict(sorted(by_month.items()))}, "architecture_drift": {"large_change_signals": large_change[:10], "todo_or_stub_signals": todo[:10]}, "hotspots": hotspots, "improvements": improvements, @@ -307,7 +299,7 @@ def _derive_improvements( top_msg = str(flapping[0].get("message", ""))[:60] items.append(( 100, - f"Fix recurring issue: {top_msg}", + f"Investigate repeated message (may be merge/cherry-pick duplication): {top_msg}", "M", "HIGH", )) @@ -315,7 +307,7 @@ def _derive_improvements( if todo: items.append(( 90 if len(todo) >= 5 else 70, - f"Resolve {len(todo)} stub or placeholder commit(s)", + f"Check whether {len(todo)} historical stub/placeholder mentions remain unresolved", "S", "HIGH" if len(todo) >= 5 else "MEDIUM", )) @@ -323,7 +315,7 @@ def _derive_improvements( if large_change: items.append(( 60, - f"Continue decomposition — {len(large_change)} large-change signal(s) detected", + f"Review {len(large_change)} historical decomposition signals before proposing more splits", "L", "MEDIUM", )) @@ -345,11 +337,11 @@ def _derive_improvements( # No issues found: project is healthy if not items: - items.append((10, "No critical remediation items — maintain current trajectory", "S", "LOW")) + items.append((10, "No keyword-derived candidates; source review still required", "S", "LOW")) items.sort(key=lambda x: x[0], reverse=True) return [ - {"rank": i + 1, "title": title, "effort": effort, "impact": impact} + {"rank": i + 1, "title": title, "effort": effort, "impact": impact, "status": "UNVERIFIED investigation candidate", "evidence": (todo[:3] if "placeholder" in title else large_change[:3] if "decomposition" in title else hotspots[:3] if "repeated" in title else quality[:3])} for i, (_, title, effort, impact) in enumerate(items) ] @@ -406,6 +398,7 @@ def run_all(self, vectors: list[str] | None = None) -> dict[str, str]: try: output_path = analysis_dir / f"analysis-{vector_name}.json" result = runner_func() + result["methodology"] = {"basis": "commit-message heuristics", "source_inspection": False, "causal_inference": False, "limitations": "Requires source/PR validation. Missing keyword evidence does not establish absence. Repeated commits across refs are not necessarily recurring defects."} atomic_write(output_path, json.dumps(result, indent=2, ensure_ascii=False) + "\n") results[vector_name] = str(output_path) print(f" [analysis] {vector_name}: {output_path}") diff --git a/archaeology/audit.py b/archaeology/audit.py index 9b2e370..400cf76 100644 --- a/archaeology/audit.py +++ b/archaeology/audit.py @@ -319,6 +319,7 @@ def run_audit(project_name: str, root: str | Path = ".") -> list[AuditFinding]: for check in ( check_project_config, + check_mined_history, check_canonical_consistency, check_placeholder_data, check_sensitive_artifacts, @@ -339,3 +340,36 @@ def summarize(findings: Iterable[AuditFinding]) -> dict[str, int]: for finding in findings: summary[finding.severity] = summary.get(finding.severity, 0) + 1 return summary + + +def check_mined_history(project_name: str, root: Path) -> list[AuditFinding]: + """Reconcile extraction identities and byte bindings when mining evidence exists.""" + import csv + import hashlib + project = _project_dir(project_name, root) + coverage_path = project / 'data' / 'coverage.json' + coverage = _load_json(coverage_path) + if coverage is None: + return [] # Imported legacy datasets have no mining manifest. + findings = [] + for name, expected in coverage.get('artifact_sha256', {}).items(): + path = project / 'data' / name + if name not in {'github-commits.csv', 'github-commits-with-stats.txt'}: + findings.append(AuditFinding('HIGH', 'MINING_MANIFEST_INVALID', 'Unexpected artifact name')) + continue + if not path.exists() or hashlib.sha256(path.read_bytes()).hexdigest() != expected: + findings.append(AuditFinding('HIGH', 'MINING_ARTIFACT_DRIFT', f'Mined artifact changed: {name}')) + csv_path = project / 'data' / 'github-commits.csv' + db_path = project / 'data' / 'archaeology.db' + try: + with csv_path.open(encoding='utf-8', newline='') as handle: + hashes = [row['hash'] for row in csv.DictReader(handle)] + if not db_path.exists(): + raise ValueError('Mined history database is missing') + with sqlite3.connect(db_path) as conn: + stored = [row[0] for row in conn.execute('SELECT hash FROM commits')] + if len(hashes) != coverage.get('commit_count') or len(set(hashes)) != len(hashes) or sorted(hashes) != sorted(stored): + raise ValueError('Git manifest, CSV and SQLite commit identities do not reconcile') + except (OSError, ValueError, KeyError, sqlite3.Error) as exc: + findings.append(AuditFinding('HIGH', 'MINING_HISTORY_DRIFT', str(exc))) + return findings diff --git a/archaeology/cli.py b/archaeology/cli.py index 18cba43..ce04955 100644 --- a/archaeology/cli.py +++ b/archaeology/cli.py @@ -12,7 +12,7 @@ from .analysis_runner import run_analysis_vectors from .classifiers.era_detector import detect_signals -from .extractors.git import extract_git_log, extract_git_log_with_stats +from .extractors.git import extract_git_log, extract_git_log_with_stats, repository_coverage def _project_dir(project_name): @@ -26,6 +26,7 @@ def _project_dir(project_name): @click.group() +@click.version_option(package_name="devarch-framework") def main(): """DevArch Framework - forensic mining of software development history.""" pass @@ -95,7 +96,7 @@ def demo(project_name, force, build_db): @click.option("--verbose", "-v", is_flag=True) def mine(repo_path, project, verbose): """Phase 1: Extract data from a git repository.""" - from .extractors.git import extract_git_log, extract_git_log_with_stats + from .extractors.git import extract_git_log, extract_git_log_with_stats, repository_coverage project_dir = _project_dir(project) data_dir = os.path.join(project_dir, "data") @@ -104,9 +105,13 @@ def mine(repo_path, project, verbose): click.echo(f"Repository not found: {repo_path}", err=True) sys.exit(1) - if not os.path.isdir(os.path.join(os.path.expanduser(repo_path), '.git')): - click.echo(f"Error: Not a git repository: {repo_path}", err=True) - sys.exit(1) + repo_path = str(Path(repo_path).expanduser().resolve()) + try: + coverage = repository_coverage(repo_path) + except RuntimeError as exc: + raise click.ClickException(str(exc)) from exc + if coverage["shallow"]: + raise click.ClickException("Shallow history: fetch complete history before mining; no fetch performed") click.echo(f"Extracting git log from {repo_path}...") @@ -125,6 +130,17 @@ def mine(repo_path, project, verbose): click.echo(f"Error: Git stats extraction failed: {e}", err=True) sys.exit(1) + if repository_coverage(repo_path) != coverage: + raise click.ClickException("Repository refs changed during mining; retry before analysis") + from .utils import atomic_write + import hashlib + coverage["artifact_sha256"] = {Path(path).name: hashlib.sha256(Path(path).read_bytes()).hexdigest() for path in (csv_path, stats_path)} + atomic_write(Path(data_dir) / "coverage.json", json.dumps(coverage, indent=2)) + config_path = Path(project_dir) / "project.json" + config = json.loads(config_path.read_text(encoding="utf-8")) + config["repo_path"] = repo_path + atomic_write(config_path, json.dumps(config, indent=2)) + click.echo(f"Phase 1 complete for '{project}'.") @@ -416,147 +432,12 @@ def export_report_cmd(project_name, fmt, output_path): @click.argument("project_name") def visualize(project_name): """Phase 4: Generate visualization HTML from template.""" - project_dir = _project_dir(project_name) - template = os.path.join("archaeology", "visualization", "template.html") - data_json = os.path.join(project_dir, "deliverables", "data.json") - output_html = os.path.join(project_dir, "deliverables", "visuals", "archaeology.html") - - if not os.path.exists(template): - click.echo(f"Template not found at {template}", err=True) - sys.exit(1) - - # Load project config for hydration - config_path = os.path.join(project_dir, "project.json") - project_config = {} - if os.path.exists(config_path): - try: - with open(config_path, encoding="utf-8") as f: - project_config = json.load(f) - except json.JSONDecodeError as e: - click.echo(f"Error: Invalid JSON in {config_path}: {e}", err=True) - sys.exit(1) - - vis = project_config.get("visualization", {}) - overrides = project_config.get("overrides", {}) - - # Read template and inject project-specific values - with open(template, encoding="utf-8") as f: - html = f.read() - - # Compute stats from commit-eras.json for template hydration - total_commits = 0 - total_lines = 0 - first_date = "" - last_date = "" - agent_count = 0 - eras_data = None - eras_json = os.path.join(project_dir, "data", "commit-eras.json") - if os.path.exists(eras_json): - try: - with open(eras_json, encoding="utf-8") as f: - eras_data = json.load(f) - except json.JSONDecodeError as e: - click.echo(f"Error: Invalid JSON in {eras_json}: {e}", err=True) - sys.exit(1) - total_commits = eras_data.get("total_commits", 0) - lifespan = eras_data.get("lifespan", "") - # Parse "43 days (Feb 28 - Apr 11, 2026)" format - if "(" in lifespan and ")" in lifespan: - date_part = lifespan.split("(")[1].split(")")[0] - parts = date_part.split(" - ") - first_date = parts[0].strip() if parts else "" - last_date = parts[-1].strip() if len(parts) > 1 else "" - # Count unique agents from agent_evidence - agent_evidence = eras_data.get("agent_evidence", {}) - agent_count = len(agent_evidence) - if not agent_count: - agent_count = 6 # Claude, Kai, Cursor, Kimi, Codex, dogfood - # Get file count from codebase_growth last entry - growth = eras_data.get("codebase_growth", []) - if growth: - total_lines = growth[-1].get("files", 0) - elif os.path.exists(data_json): - try: - with open(data_json, encoding="utf-8") as f: - pdata = json.load(f) - total_commits = pdata.get("total_commits", 0) - except json.JSONDecodeError as e: - click.echo(f"Error: Invalid JSON in {data_json}: {e}", err=True) - sys.exit(1) - - # Hydrate template variables - title = vis.get("title", project_name.upper()) - duration = vis.get("duration", f"{first_date} — {last_date}" if first_date else "") - html = html.replace("{{PROJECT_NAME}}", title) - html = html.replace("{{PROJECT_DURATION}}", duration) - html = html.replace("{{TOTAL_COMMITS}}", str(total_commits or 803)) - html = html.replace("{{TOTAL_LINES}}", str(total_lines or "35,600")) - html = html.replace("{{AGENT_COUNT}}", str(agent_count or 6)) - # Compute era count for meta description - era_count = len(eras_data.get("eras", [])) if os.path.exists(eras_json) else 0 - html = html.replace("{{ERA_COUNT}}", str(era_count)) - - # Also update tag if it still has the old format - html = html.replace( - "<title>DevArch Framework", - f"{title} — DevArch Framework", - ) - - # Generate era color CSS variables from config - era_colors = vis.get("era_colors", {}) - if era_colors: - era_css = "\n".join( - f" --{era_key}: {color};" - for era_key, color in era_colors.items() - ) - # Insert era colors after :root block opens - html = html.replace( - "/* ERA COLORS */", - f"/* ERA COLORS — from project.json */\n{era_css}", - ) - - # Generate agent color CSS variables - agent_colors = vis.get("agent_colors", {}) - if agent_colors: - agent_css = "\n".join( - f" --{name.lower()}: {color};" - for name, color in agent_colors.items() - ) - html = html.replace( - "/* AGENT COLORS */", - f"/* AGENT COLORS — from project.json */\n{agent_css}", - ) - - # Inline data.json so the HTML works from file:// (no CORS issues) - if os.path.exists(data_json): - with open(data_json, encoding="utf-8") as f: - data_payload = json.load(f) - - # Merge commit_eras and top-level fields from commit-eras.json into PROJECT_DATA - # so the era timeline visualization has real data to render. - if eras_data is not None: - data_payload.setdefault("commit_eras", eras_data.get("eras", [])) - data_payload.setdefault("total_commits", eras_data.get("total_commits", 0)) - data_payload.setdefault("first_commit_date", eras_data.get("first_commit_date", "")) - data_payload.setdefault("last_commit_date", eras_data.get("last_commit_date", "")) - - data_content = json.dumps(data_payload) - safe_data_content = data_content.replace("<", "\\u003c").replace(">", "\\u003e").replace("&", "\\u0026") - inline_script = f'' - html = html.replace( - '', - inline_script, - ) - - # Write hydrated HTML - os.makedirs(os.path.dirname(output_html), exist_ok=True) - with open(output_html, "w", encoding="utf-8") as f: - f.write(html) - - click.echo(f"Visualization generated at {output_html}") - - if not os.path.exists(data_json): - click.echo(f"Warning: {data_json} not found. Visualization will be empty.") + from .visualization.history import render_history + try: + output = render_history(Path(_project_dir(project_name))) + except (OSError, ValueError) as exc: + raise click.ClickException(str(exc)) from exc + click.echo(f"Visualization generated at {output}") @main.command() @@ -609,7 +490,7 @@ def ingest_pipeline(project_name, logs_dir, verbose): def cascade(project_name, dry_run, skip_mine): """Full pipeline: mine → build-db → signals → era cascade → sync → audit.""" from .era_cascade import cascade as run_cascade - from .extractors.git import extract_git_log, extract_git_log_with_stats + from .extractors.git import extract_git_log, extract_git_log_with_stats, repository_coverage from .classifiers.era_detector import detect_signals project_dir = Path(_project_dir(project_name)) diff --git a/archaeology/db/builder.py b/archaeology/db/builder.py index b78a138..b5a3671 100644 --- a/archaeology/db/builder.py +++ b/archaeology/db/builder.py @@ -585,10 +585,10 @@ def _assert_commits_ingested(db_path: Path, data_dir: Path) -> None: db_rows = conn.execute("SELECT COUNT(*) FROM commits").fetchone()[0] if "commits" in tables else 0 finally: conn.close() - if db_rows == 0: + if db_rows != csv_rows: print( f"ERROR: github-commits.csv has {csv_rows} commit row(s) but the 'commits' " - "table is empty after build. Commit ingestion failed (is sqlite-utils " + "table count differs after build. Commit ingestion failed (is sqlite-utils " "installed?). Aborting so the pipeline does not emit a false all-zero report.", file=sys.stderr, ) @@ -670,6 +670,8 @@ def build_db(project_root: Path, output: Path | None = None, verbose: bool = Fal print("\n--- Full-text search ---") create_fts(db_path, fts_config, verbose) + from ..metrics import write_metrics + write_metrics(project_root, db_path) print_summary(db_path, verbose) print(f"\nDone. Database: {db_path}") diff --git a/archaeology/extractors/git.py b/archaeology/extractors/git.py index dc875be..7004626 100644 --- a/archaeology/extractors/git.py +++ b/archaeology/extractors/git.py @@ -5,44 +5,68 @@ from pathlib import Path -def extract_git_log(repo_path: str, output_path: str, verbose: bool = False) -> int: - """Extract git log to CSV. Returns number of commits extracted.""" - # Use %x1f (unit separator) as delimiter — can't appear in commit subjects - cmd = [ - "git", "-C", repo_path, - "log", "--format=%H%x1f%ai%x1f%s%x1f%an", "--all" - ] +def _git(repo_path: str, *args: str) -> str: + """Read Git metadata without invoking a shell or repository hooks.""" try: - result = subprocess.run(cmd, capture_output=True, text=True, timeout=300) - except FileNotFoundError: - raise RuntimeError("git binary not found. Install git and ensure it's on PATH.") - except subprocess.TimeoutExpired: - raise RuntimeError("git log timed out after 300s. Repository may be too large.") + result = subprocess.run( + ["git", "-C", str(Path(repo_path).expanduser()), *args], + capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=300, + ) + except FileNotFoundError as exc: + raise RuntimeError("git binary not found. Install git and ensure it's on PATH.") from exc + except subprocess.TimeoutExpired as exc: + raise RuntimeError("git read timed out after 300s") from exc + if result.returncode: + raise RuntimeError(f"git {args[0]} failed: {result.stderr.strip()}") + return result.stdout - if result.returncode != 0: - raise RuntimeError(f"git log failed: {result.stderr}") - lines = result.stdout.strip().split("\n") - if not lines or lines[0] == "": - return 0 +def repository_coverage(repo_path: str) -> dict: + """Record local reachability, never imply that remote/deleted refs were fetched.""" + _git(repo_path, "rev-parse", "--git-dir") + refs = _git(repo_path, "for-each-ref", "--format=%(refname)%09%(objectname)") + return { + "scope": "all locally available refs plus HEAD; no automatic remote fetch", + "shallow": _git(repo_path, "rev-parse", "--is-shallow-repository").strip() == "true", + "bare": _git(repo_path, "rev-parse", "--is-bare-repository").strip() == "true", + "commit_count": int(_git(repo_path, "rev-list", "--all", "--count").strip()), + "refs": [dict(zip(("name", "object"), line.split("\t", 1))) for line in refs.splitlines()], + "roots": _git(repo_path, "rev-list", "--all", "--max-parents=0").splitlines(), + "gaps": ["Deleted, inaccessible and unfetched remote refs are outside this local snapshot."], + } - Path(output_path).parent.mkdir(parents=True, exist_ok=True) - with open(output_path, "w", newline="", encoding="utf-8") as f: - writer = csv.writer(f) - writer.writerow(["hash", "date", "message", "author"]) - skipped = 0 - for line in lines: - parts = line.split("\x1f") - if len(parts) >= 4: - writer.writerow(parts[:4]) - else: - skipped += 1 - - count = len(lines) - skipped + +def extract_git_log(repo_path: str, output_path: str, verbose: bool = False) -> int: + """Extract every reachable commit, with NUL-delimited fields and UTC dates. + + Subjects can contain tabs, pipes and unit separators. Git NUL delimiters + preserve those values rather than silently shifting or dropping CSV fields. + Empty histories write a header, replacing any stale previous extraction. + """ + from datetime import datetime, timezone + import io + from ..utils import atomic_write + + raw = _git(repo_path, "log", "-z", "--all", "--format=%H%x00%aI%x00%s%x00%an") + fields = raw.split("\0") + if fields[-1] == "": + fields.pop() + if len(fields) % 4: + raise RuntimeError("Malformed Git extraction; refusing partial history") + stream = io.StringIO(newline="") + writer = csv.writer(stream) + writer.writerow(["hash", "date", "message", "author"]) + for i in range(0, len(fields), 4): + sha, date, subject, author = fields[i:i + 4] + normalized = datetime.fromisoformat(date).astimezone(timezone.utc).isoformat() + writer.writerow([sha, normalized, subject, author]) + count = len(fields) // 4 + expected = int(_git(repo_path, "rev-list", "--all", "--count").strip()) + if count != expected: + raise RuntimeError("Repository changed during extraction; retry against frozen refs") + atomic_write(output_path, stream.getvalue()) if verbose: print(f"Extracted {count} commits from {repo_path}") - if skipped: - print(f" Skipped {skipped} malformed lines (expected 4 fields, got fewer)") return count @@ -68,7 +92,7 @@ def extract_git_log_with_stats(repo_path: str, output_path: str, verbose: bool = if verbose: print(f"Extracted git log with stats from {repo_path}") - return result.stdout.count("\n\n") + return int(_git(repo_path, "rev-list", "--all", "--count").strip()) def get_repo_list(repo_path: str) -> list[str]: diff --git a/archaeology/metrics.py b/archaeology/metrics.py new file mode 100644 index 0000000..ca7ea7b --- /dev/null +++ b/archaeology/metrics.py @@ -0,0 +1,43 @@ +"""Canonical metrics from the mined commits, reconciled against SQLite.""" +import csv +import json +import sqlite3 +from collections import Counter +from pathlib import Path +from .utils import _parse_date, atomic_write + + +def write_metrics(project_root: Path, db_path: Path) -> dict: + csv_path = project_root / "data" / "github-commits.csv" + if not csv_path.exists(): + return {} + with csv_path.open(encoding="utf-8", newline="") as handle: + rows = list(csv.DictReader(handle)) + hashes = [row["hash"] for row in rows] + with sqlite3.connect(db_path) as conn: + stored = [row[0] for row in conn.execute("SELECT hash FROM commits")] + if len(set(hashes)) != len(hashes) or sorted(hashes) != sorted(stored): + raise ValueError("CSV/SQLite commit identities differ or contain duplicates") + days = Counter() + for row in rows: + parsed = _parse_date(row["date"]) + if parsed is None: + raise ValueError(f"Invalid commit date for {row['hash']}") + days[parsed.date().isoformat()] += 1 + dates = sorted(days) + peak = min(days, key=lambda d: (-days[d], d)) if days else None + metrics = { + "total_commits": len(rows), "active_days": len(days), + "first_commit_date": dates[0] if dates else None, + "last_commit_date": dates[-1] if dates else None, + "span_days": (_parse_date(dates[-1]) - _parse_date(dates[0])).days + 1 if dates else 0, + "peak_day": peak, "peak_day_commits": days[peak] if peak else 0, + "date_basis": "author timestamps normalized to UTC; inclusive calendar span", + "source_scope": "all locally mined refs, not proof of all remote history", + } + atomic_write(project_root / "deliverables" / "canonical-metrics.json", json.dumps(metrics, indent=2)) + atomic_write(project_root / "deliverables" / "data.json", json.dumps({ + **metrics, "daily_commits": dict(sorted(days.items())), + "telemetry_visualizations": {"meta": metrics}, + }, indent=2)) + return metrics diff --git a/archaeology/report.py b/archaeology/report.py index 017c3e4..4d8b9e2 100644 --- a/archaeology/report.py +++ b/archaeology/report.py @@ -72,9 +72,10 @@ def export_markdown_report(project_name: str, project_root: str | Path, output_p f"This report summarizes the `{project_name}` development archaeology from canonical project metrics, era data, and automated analysis vectors.\n\n" ) + out.append("Automated vectors are commit-keyword investigation leads, not verified source findings, causal explanations, session measurements or implementation-quality scores. Missing evidence is not absence.\n\n") out.append("## Canonical Metrics\n\n") metric_rows = [ - ("Total commits", canonical.get("total_commits") or eras.get("total_commits")), + ("Total commits", canonical.get("total_commits", eras.get("total_commits"))), ("Span days", canonical.get("span_days")), ("Active days", canonical.get("active_days")), ("Peak day", canonical.get("peak_day")), diff --git a/archaeology/utils.py b/archaeology/utils.py index 99b5096..74fdffa 100644 --- a/archaeology/utils.py +++ b/archaeology/utils.py @@ -5,7 +5,7 @@ import json import logging import os -from datetime import datetime +from datetime import datetime, timezone from pathlib import Path from typing import Any @@ -64,14 +64,15 @@ def _parse_date(date_str: str) -> datetime | None: if not date_str: return None date_str = str(date_str).strip() - for fmt in ( - "%Y-%m-%d %H:%M:%S %z", - "%Y-%m-%d %H:%M:%S", - "%Y-%m-%dT%H:%M:%S", - "%Y-%m-%d", - ): + # Normalize timezone-bearing dates to UTC, preserving a naive UTC return + # for existing date arithmetic callers. Legacy trailing annotations remain supported. + candidates = [date_str, date_str[:25], date_str[:26], date_str[:19], date_str[:10]] + for candidate in candidates: try: - return datetime.strptime(date_str[:19], fmt) + parsed = datetime.fromisoformat(candidate.replace("Z", "+00:00")) + if parsed.tzinfo is not None: + parsed = parsed.astimezone(timezone.utc).replace(tzinfo=None) + return parsed except ValueError: continue return None diff --git a/archaeology/visualization/history.py b/archaeology/visualization/history.py new file mode 100644 index 0000000..12251cc --- /dev/null +++ b/archaeology/visualization/history.py @@ -0,0 +1,30 @@ +"""Portable, self-contained visualization of measured history only.""" +import html +import json +from pathlib import Path +from ..utils import atomic_write + + +def render_history(project_root: Path) -> Path: + data_path = project_root / "deliverables" / "data.json" + if not data_path.exists(): + raise ValueError("No measured visualization data; run devarch build-db first") + data = json.loads(data_path.read_text(encoding="utf-8")) + daily = data.get("daily_commits") + if not isinstance(daily, dict): + raise ValueError("Measured daily_commits missing; rebuild with devarch build-db") + if sum(daily.values()) != data.get("total_commits"): + raise ValueError("Daily activity does not reconcile with total_commits") + config = json.loads((project_root / "project.json").read_text(encoding="utf-8")) + title = html.escape(str(config.get("visualization", {}).get("title") or config.get("name") or project_root.name)) + maximum = max(daily.values(), default=1) or 1 + rows = ''.join(f'{html.escape(day)}{count}{count}' for day, count in sorted(daily.items())) + content = f''' +{title} — history + +

{title}

{data['total_commits']:,} commits · {data.get('active_days', 0)} active UTC days

+

Author-date activity across locally available refs. Commits are not productivity, verified sessions, causal eras or evidence of implementation quality.

+{rows}
Measured daily commit activity
Date (UTC)CommitsActivity
''' + output = project_root / "deliverables" / "visuals" / "archaeology.html" + atomic_write(output, content) + return output diff --git a/docs/RELEASE_0.4.0.md b/docs/RELEASE_0.4.0.md new file mode 100644 index 0000000..a15e8fc --- /dev/null +++ b/docs/RELEASE_0.4.0.md @@ -0,0 +1,26 @@ +# DevArch 0.4.0 evidence-integrity release + +This release repairs the standard init → mine → build-db → signals → analyze → visualize → export-report → audit path before using it for release archaeology. + +## Root causes and fixes + +- Git validation assumed a `.git` directory, rejecting linked worktrees and bare repositories. Git now establishes repository validity; shallow histories fail explicitly. Mining includes all local refs, writes an exact ref/root/count inventory and binds the CSV/raw-stat files by SHA256. It does not fetch remotes or claim deleted history coverage. +- Unit-separator parsing could silently shift fields in unusual subjects. NUL-delimited extraction retains separators, tabs and Unicode. Empty histories replace stale CSV data with a header. +- The ordinary pipeline never produced the canonical metrics needed by audit/reporting. Database build now reconciles complete commit identities and produces measured UTC counts/dates and daily activity. +- Visualization depended on the caller's current directory and substituted 803 commits, 35,600 lines and six agents. It now produces a self-contained measured daily-activity table/chart from any working directory; missing data fails explicitly. The prior narrative dashboard template remains available as a legacy template but is no longer the default CLI output. +- Agent vectors fabricated session-depth buckets and hook effectiveness; ML vectors fabricated similarity and token waste. These become unmeasured/null or explicitly low-confidence investigation candidates. Keyword evidence is not source verification. SDLC and source summaries no longer establish absence or improving quality from keywords alone. Summary keyword counts are no longer capped at 500. +- Timestamp parsing discarded offsets. Dates now normalize to UTC before aggregation. +- Audit now catches modified mining inputs, missing databases and commit identity drift. The example archaeology workflow no longer swallows audit failures. +- Runtime and distribution versions agree, and `devarch --version` identifies the installed distribution. + +## Compatibility and boundaries + +Python >=3.10 and existing CLI command names remain. JSON session measurements and unsupported similarity/waste claims are now null rather than fabricated numbers; consumers must handle unknown values. Generated charts intentionally do not infer agents, personal sessions or causal eras. CSV dates are ISO8601 UTC, potentially shifting author-local calendar days. Git author aliases remain unresolved. Merge/cherry-pick duplication across refs is genuine history, not necessarily repeated defects. + +This is deterministic commit-message analysis plus an auditable extraction path, not automated line-level source/PR reasoning. Manual source evidence is required for architecture decisions, abandoned-feature disposition and release recommendations. No personal session or YouTube data is needed. The mining implementation still buffers Git output; large-repository streaming remains a documented follow-up. Raw history may contain private material and must be reviewed before sharing. + +## Acceptance + +Use the real-Git fixtures in `tests/test_release_pipeline.py`, the full test suite, and a clean wheel installation outside the source checkout. The fixtures cover bare repositories, worktrees, delimiter-bearing subjects, shallow rejection, timezone normalization, full pipeline execution and tampered-input audit rejection. Release only from reviewed canonical history; never bypass protections. Registry availability is separate from a GitHub release. + +Local acceptance on Python 3.12: baseline 97 tests; repaired suite 104 tests passed. Wheel and source archive passed Twine metadata validation. A separate non-editable wheel environment outside the checkout passed ten CLI invocations, including bare-repository mining, database build, all six analysis vectors, visualization, both report formats and audit. Git/CSV/SQLite totals reconciled to two fixture commits. These fixtures do not establish remote registry publication or native Windows/macOS acceptance. diff --git a/setup.py b/setup.py index d1aa6c0..4b3e49a 100644 --- a/setup.py +++ b/setup.py @@ -5,7 +5,7 @@ setup( name="devarch-framework", - version="0.3.0", + version="0.4.0", packages=find_packages(exclude=["tests*", "projects*", "analysis-vectors*"]), include_package_data=True, install_requires=[ diff --git a/tests/test_cli_coverage.py b/tests/test_cli_coverage.py index c8825a9..13ebce0 100644 --- a/tests/test_cli_coverage.py +++ b/tests/test_cli_coverage.py @@ -145,34 +145,28 @@ def test_validate_exits_when_html_missing(tmp_path, monkeypatch): # ── visualize ───────────────────────────────────────────────────────────────── -def test_visualize_exits_when_template_missing(tmp_path, monkeypatch): +def test_visualize_exits_when_measured_data_missing(tmp_path, monkeypatch): monkeypatch.chdir(tmp_path) _make_project(tmp_path, "viz-proj") runner = CliRunner() result = runner.invoke(main, ["visualize", "viz-proj"]) assert result.exit_code != 0 - assert "Template not found" in result.output + assert "No measured visualization data" in result.output def test_visualize_generates_html_with_template(tmp_path, monkeypatch): monkeypatch.chdir(tmp_path) _make_project(tmp_path, "viz-proj") - # Create a minimal template in the expected relative path - viz_dir = tmp_path / "archaeology" / "visualization" - viz_dir.mkdir(parents=True) - template = viz_dir / "template.html" - template.write_text( - "{{PROJECT_NAME}}", - encoding="utf-8", - ) + data = {"total_commits": 1, "active_days": 1, "daily_commits": {"2026-01-01": 1}} + (tmp_path / "projects" / "viz-proj" / "deliverables" / "data.json").write_text(json.dumps(data)) runner = CliRunner() result = runner.invoke(main, ["visualize", "viz-proj"]) assert result.exit_code == 0 output_html = tmp_path / "projects" / "viz-proj" / "deliverables" / "visuals" / "archaeology.html" assert output_html.exists() - assert "VIZ-PROJ" in output_html.read_text() + assert "viz-proj" in output_html.read_text() # ── ingest-pipeline ─────────────────────────────────────────────────────────── diff --git a/tests/test_release_pipeline.py b/tests/test_release_pipeline.py new file mode 100644 index 0000000..31dd1ea --- /dev/null +++ b/tests/test_release_pipeline.py @@ -0,0 +1,100 @@ +"""Real Git and installed-style CLI regressions for evidence integrity.""" +import csv +import json +import subprocess +from pathlib import Path +import pytest +from click.testing import CliRunner +from archaeology.cli import main +from archaeology.extractors.git import extract_git_log, repository_coverage +from archaeology.analysis_runner import AnalysisRunner +from archaeology.utils import _parse_date + + +def git(path, *args): + return subprocess.check_output(['git', '-C', str(path), *args], text=True).strip() + + +@pytest.fixture +def repo(tmp_path): + path = tmp_path / 'source'; path.mkdir() + git(path, 'init'); git(path, 'config', 'user.name', 'Fixture') + git(path, 'config', 'user.email', 'fixture@example.invalid') + git(path, 'commit', '--allow-empty', '-m', 'init') + git(path, 'checkout', '-b', 'side') + git(path, 'commit', '--allow-empty', '-m', 'score\x1fpipe|tab\tλ') + return path + + +def test_all_refs_bare_worktree_and_delimiter(repo, tmp_path): + bare = tmp_path / 'bare.git'; git(repo, 'clone', '--bare', str(repo), str(bare)) + worktree = tmp_path / 'linked'; git(repo, 'worktree', 'add', '--detach', str(worktree)) + for source in (repo, bare, worktree): + out = tmp_path / (source.name + '.csv') + assert extract_git_log(str(source), str(out)) == 2 + with out.open(newline='', encoding='utf-8') as f: rows = list(csv.DictReader(f)) + assert len(rows) == 2 + assert any('score\x1fpipe|tab\tλ' == row['message'] for row in rows) + assert all(row['author'] == 'Fixture' for row in rows) + assert repository_coverage(str(source))['commit_count'] == 2 + + +def test_shallow_rejected_before_output(repo, tmp_path, monkeypatch): + shallow = tmp_path / 'shallow' + git(repo, 'clone', '--depth=1', repo.as_uri(), str(shallow)) + monkeypatch.chdir(tmp_path) + runner = CliRunner(); assert runner.invoke(main, ['init','trial']).exit_code == 0 + result = runner.invoke(main, ['mine', str(shallow), '-p', 'trial']) + assert result.exit_code != 0 + assert 'Shallow history' in result.output + assert not (tmp_path / 'projects/trial/data/github-commits.csv').exists() + + +def test_pipeline_outside_checkout(repo, tmp_path, monkeypatch): + workspace = tmp_path / 'consumer'; workspace.mkdir(); monkeypatch.chdir(workspace) + runner = CliRunner() + for args in (['init','trial'], ['mine',str(repo),'-p','trial'], ['build-db','trial'], + ['signals','trial'], ['analyze','trial'], ['visualize','trial'], + ['export-report','trial'], ['audit','trial']): + result = runner.invoke(main, args) + assert result.exit_code == 0, (args, result.output, repr(result.exception)) + project = workspace / 'projects/trial' + metrics = json.loads((project / 'deliverables/canonical-metrics.json').read_text()) + assert metrics['total_commits'] == 2 + html = (project / 'deliverables/visuals/archaeology.html').read_text() + assert '2 commits' in html and '35,600' not in html + agent = json.loads((project / 'deliverables/analysis/analysis-agentic-workflow.json').read_text()) + assert agent['session_depth_distribution'] is None + assert agent['hook_effectiveness'] == [] + ml = json.loads((project / 'deliverables/analysis/analysis-ml-pattern-mapper.json').read_text()) + assert all(x['confidence'] == 'LOW' and x['estimated_token_waste'] is None for x in ml['mappings']) + + +def test_dates_normalize_offset(): + assert _parse_date('2026-01-01T23:30:00-08:00') == _parse_date('2026-01-02T07:30:00Z') + + +def test_analysis_missing_database_is_not_absence(tmp_path): + runner = AnalysisRunner('missing', str(tmp_path)) + with pytest.raises(ValueError, match='database missing'): + runner.run_sdlc_gap_finder() + + +def test_empty_history_replaces_stale_csv(tmp_path): + git(tmp_path, 'init') + out = tmp_path / 'out.csv'; out.write_text('stale') + assert extract_git_log(str(tmp_path), str(out)) == 0 + assert out.read_text().strip() == 'hash,date,message,author' + + +def test_audit_detects_modified_mining_input(repo, tmp_path, monkeypatch): + from archaeology.audit import check_mined_history + monkeypatch.chdir(tmp_path) + runner = CliRunner() + for args in (['init','audit-fixture'], ['mine',str(repo),'-p','audit-fixture'], ['build-db','audit-fixture']): + result = runner.invoke(main, args) + assert result.exit_code == 0, result.output + assert not check_mined_history('audit-fixture', tmp_path) + path = tmp_path / 'projects/audit-fixture/data/github-commits.csv' + path.write_text(path.read_text().replace('init','changed')) + assert any(f.code == 'MINING_ARTIFACT_DRIFT' for f in check_mined_history('audit-fixture', tmp_path)) From c50b0c22e6c45405364402b71e1bdc8bdaa26ca7 Mon Sep 17 00:00:00 2001 From: Simon Gonzalez De Cruz Date: Fri, 2 Oct 2026 05:00:57 +0000 Subject: [PATCH 2/4] fix: grant release checkout read-only contents permission --- .github/workflows/publish.yml | 1 + 1 file changed, 1 insertion(+) diff --git a/.github/workflows/publish.yml b/.github/workflows/publish.yml index a589b94..b64186c 100644 --- a/.github/workflows/publish.yml +++ b/.github/workflows/publish.yml @@ -6,6 +6,7 @@ on: - 'v*.*.*' permissions: + contents: read id-token: write # Trusted publishing jobs: From f53536981cfc95d9b1f36e24173268a00f86ac9b Mon Sep 17 00:00:00 2001 From: Simon Gonzalez De Cruz Date: Fri, 2 Oct 2026 05:12:17 +0000 Subject: [PATCH 3/4] fix: reject corrupted evidence and normalize native Git output --- archaeology/audit.py | 38 ++++++++++++++++------ archaeology/cli.py | 1 + archaeology/extractors/git.py | 10 +++--- archaeology/metrics.py | 17 +++++++--- docs/RELEASE_0.4.0.md | 8 +++-- tests/test_release_pipeline.py | 59 ++++++++++++++++++++++++++++++++++ 6 files changed, 111 insertions(+), 22 deletions(-) diff --git a/archaeology/audit.py b/archaeology/audit.py index 400cf76..73ace4e 100644 --- a/archaeology/audit.py +++ b/archaeology/audit.py @@ -349,27 +349,45 @@ def check_mined_history(project_name: str, root: Path) -> list[AuditFinding]: project = _project_dir(project_name, root) coverage_path = project / 'data' / 'coverage.json' coverage = _load_json(coverage_path) - if coverage is None: - return [] # Imported legacy datasets have no mining manifest. + config = _load_json(project / 'project.json') or {} + if not coverage_path.exists() and not config.get('mined_history_manifest_required'): + return [] # Legacy imported datasets never claimed a mining manifest. + if not isinstance(coverage, dict) or not isinstance(coverage.get('artifact_sha256'), dict): + return [AuditFinding('HIGH', 'MINING_MANIFEST_INVALID', 'Mining coverage manifest missing or invalid')] findings = [] - for name, expected in coverage.get('artifact_sha256', {}).items(): - path = project / 'data' / name - if name not in {'github-commits.csv', 'github-commits-with-stats.txt'}: + expected_names = {'github-commits.csv', 'github-commits-with-stats.txt'} + if set(coverage['artifact_sha256']) != expected_names: + findings.append(AuditFinding('HIGH', 'MINING_MANIFEST_INVALID', 'Mining artifact bindings incomplete')) + for name, expected in coverage['artifact_sha256'].items(): + if name not in expected_names: findings.append(AuditFinding('HIGH', 'MINING_MANIFEST_INVALID', 'Unexpected artifact name')) continue + path = project / 'data' / name if not path.exists() or hashlib.sha256(path.read_bytes()).hexdigest() != expected: findings.append(AuditFinding('HIGH', 'MINING_ARTIFACT_DRIFT', f'Mined artifact changed: {name}')) csv_path = project / 'data' / 'github-commits.csv' db_path = project / 'data' / 'archaeology.db' try: with csv_path.open(encoding='utf-8', newline='') as handle: - hashes = [row['hash'] for row in csv.DictReader(handle)] + rows = list(csv.DictReader(handle)) + keys = ('hash', 'date', 'message', 'author') + source = [tuple(row[key] for key in keys) for row in rows] + hashes = [row['hash'] for row in rows] if not db_path.exists(): raise ValueError('Mined history database is missing') with sqlite3.connect(db_path) as conn: - stored = [row[0] for row in conn.execute('SELECT hash FROM commits')] - if len(hashes) != coverage.get('commit_count') or len(set(hashes)) != len(hashes) or sorted(hashes) != sorted(stored): - raise ValueError('Git manifest, CSV and SQLite commit identities do not reconcile') - except (OSError, ValueError, KeyError, sqlite3.Error) as exc: + stored = list(conn.execute('SELECT hash,date,message,author FROM commits')) + if len(hashes) != coverage.get('commit_count') or len(set(hashes)) != len(hashes) or sorted(source) != sorted(stored): + raise ValueError('Git manifest, CSV and SQLite commit records do not reconcile') + from .metrics import calculate_metrics + measured = calculate_metrics(rows) + canonical = _load_json(project / 'deliverables' / 'canonical-metrics.json') or {} + if any(canonical.get(key) != value for key, value in measured.items()): + raise ValueError('Canonical metrics differ from source commit measurements') + visual = _load_json(project / 'deliverables' / 'data.json') or {} + meta = visual.get('telemetry_visualizations', {}).get('meta', {}) + if any(visual.get(key) != value or meta.get(key) != value for key, value in measured.items()): + raise ValueError('Visualization metrics differ from source commit measurements') + except (OSError, ValueError, KeyError, TypeError, sqlite3.Error) as exc: findings.append(AuditFinding('HIGH', 'MINING_HISTORY_DRIFT', str(exc))) return findings diff --git a/archaeology/cli.py b/archaeology/cli.py index ce04955..b6dacac 100644 --- a/archaeology/cli.py +++ b/archaeology/cli.py @@ -139,6 +139,7 @@ def mine(repo_path, project, verbose): config_path = Path(project_dir) / "project.json" config = json.loads(config_path.read_text(encoding="utf-8")) config["repo_path"] = repo_path + config["mined_history_manifest_required"] = True atomic_write(config_path, json.dumps(config, indent=2)) click.echo(f"Phase 1 complete for '{project}'.") diff --git a/archaeology/extractors/git.py b/archaeology/extractors/git.py index 7004626..a143e97 100644 --- a/archaeology/extractors/git.py +++ b/archaeology/extractors/git.py @@ -25,11 +25,13 @@ def repository_coverage(repo_path: str) -> dict: """Record local reachability, never imply that remote/deleted refs were fetched.""" _git(repo_path, "rev-parse", "--git-dir") refs = _git(repo_path, "for-each-ref", "--format=%(refname)%09%(objectname)") + count = int(_git(repo_path, "rev-list", "--all", "--count").strip()) return { + "head": _git(repo_path, "rev-parse", "--verify", "HEAD").strip() if count else None, "scope": "all locally available refs plus HEAD; no automatic remote fetch", "shallow": _git(repo_path, "rev-parse", "--is-shallow-repository").strip() == "true", "bare": _git(repo_path, "rev-parse", "--is-bare-repository").strip() == "true", - "commit_count": int(_git(repo_path, "rev-list", "--all", "--count").strip()), + "commit_count": count, "refs": [dict(zip(("name", "object"), line.split("\t", 1))) for line in refs.splitlines()], "roots": _git(repo_path, "rev-list", "--all", "--max-parents=0").splitlines(), "gaps": ["Deleted, inaccessible and unfetched remote refs are outside this local snapshot."], @@ -54,11 +56,11 @@ def extract_git_log(repo_path: str, output_path: str, verbose: bool = False) -> if len(fields) % 4: raise RuntimeError("Malformed Git extraction; refusing partial history") stream = io.StringIO(newline="") - writer = csv.writer(stream) + writer = csv.writer(stream, lineterminator="\n") writer.writerow(["hash", "date", "message", "author"]) for i in range(0, len(fields), 4): sha, date, subject, author = fields[i:i + 4] - normalized = datetime.fromisoformat(date).astimezone(timezone.utc).isoformat() + normalized = datetime.fromisoformat(date.replace("Z", "+00:00")).astimezone(timezone.utc).isoformat() writer.writerow([sha, normalized, subject, author]) count = len(fields) // 4 expected = int(_git(repo_path, "rev-list", "--all", "--count").strip()) @@ -77,7 +79,7 @@ def extract_git_log_with_stats(repo_path: str, output_path: str, verbose: bool = "log", "--format=%H%x1f%ai%x1f%s%x1f%an", "--shortstat", "--all" ] try: - result = subprocess.run(cmd, capture_output=True, text=True, timeout=300) + result = subprocess.run(cmd, capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=300) except FileNotFoundError: raise RuntimeError("git binary not found. Install git and ensure it's on PATH.") except subprocess.TimeoutExpired: diff --git a/archaeology/metrics.py b/archaeology/metrics.py index ca7ea7b..ab8fb73 100644 --- a/archaeology/metrics.py +++ b/archaeology/metrics.py @@ -18,6 +18,17 @@ def write_metrics(project_root: Path, db_path: Path) -> dict: stored = [row[0] for row in conn.execute("SELECT hash FROM commits")] if len(set(hashes)) != len(hashes) or sorted(hashes) != sorted(stored): raise ValueError("CSV/SQLite commit identities differ or contain duplicates") + metrics = calculate_metrics(rows) + atomic_write(project_root / "deliverables" / "canonical-metrics.json", json.dumps(metrics, indent=2)) + atomic_write(project_root / "deliverables" / "data.json", json.dumps({ + **metrics, + "telemetry_visualizations": {"meta": metrics}, + }, indent=2)) + return metrics + + +def calculate_metrics(rows: list[dict]) -> dict: + """Derive canonical values from source rows without writing artifacts.""" days = Counter() for row in rows: parsed = _parse_date(row["date"]) @@ -28,6 +39,7 @@ def write_metrics(project_root: Path, db_path: Path) -> dict: peak = min(days, key=lambda d: (-days[d], d)) if days else None metrics = { "total_commits": len(rows), "active_days": len(days), + "daily_commits": dict(sorted(days.items())), "first_commit_date": dates[0] if dates else None, "last_commit_date": dates[-1] if dates else None, "span_days": (_parse_date(dates[-1]) - _parse_date(dates[0])).days + 1 if dates else 0, @@ -35,9 +47,4 @@ def write_metrics(project_root: Path, db_path: Path) -> dict: "date_basis": "author timestamps normalized to UTC; inclusive calendar span", "source_scope": "all locally mined refs, not proof of all remote history", } - atomic_write(project_root / "deliverables" / "canonical-metrics.json", json.dumps(metrics, indent=2)) - atomic_write(project_root / "deliverables" / "data.json", json.dumps({ - **metrics, "daily_commits": dict(sorted(days.items())), - "telemetry_visualizations": {"meta": metrics}, - }, indent=2)) return metrics diff --git a/docs/RELEASE_0.4.0.md b/docs/RELEASE_0.4.0.md index a15e8fc..ec63f77 100644 --- a/docs/RELEASE_0.4.0.md +++ b/docs/RELEASE_0.4.0.md @@ -4,13 +4,13 @@ This release repairs the standard init → mine → build-db → signals → ana ## Root causes and fixes -- Git validation assumed a `.git` directory, rejecting linked worktrees and bare repositories. Git now establishes repository validity; shallow histories fail explicitly. Mining includes all local refs, writes an exact ref/root/count inventory and binds the CSV/raw-stat files by SHA256. It does not fetch remotes or claim deleted history coverage. +- Git validation assumed a `.git` directory, rejecting linked worktrees and bare repositories. Git now establishes repository validity; shallow histories fail explicitly. Mining includes all local refs, writes an exact HEAD/ref/root/count inventory and binds the CSV/raw-stat files by SHA256. It does not fetch remotes or claim deleted history coverage. - Unit-separator parsing could silently shift fields in unusual subjects. NUL-delimited extraction retains separators, tabs and Unicode. Empty histories replace stale CSV data with a header. - The ordinary pipeline never produced the canonical metrics needed by audit/reporting. Database build now reconciles complete commit identities and produces measured UTC counts/dates and daily activity. - Visualization depended on the caller's current directory and substituted 803 commits, 35,600 lines and six agents. It now produces a self-contained measured daily-activity table/chart from any working directory; missing data fails explicitly. The prior narrative dashboard template remains available as a legacy template but is no longer the default CLI output. - Agent vectors fabricated session-depth buckets and hook effectiveness; ML vectors fabricated similarity and token waste. These become unmeasured/null or explicitly low-confidence investigation candidates. Keyword evidence is not source verification. SDLC and source summaries no longer establish absence or improving quality from keywords alone. Summary keyword counts are no longer capped at 500. - Timestamp parsing discarded offsets. Dates now normalize to UTC before aggregation. -- Audit now catches modified mining inputs, missing databases and commit identity drift. The example archaeology workflow no longer swallows audit failures. +- Audit now catches modified mining inputs, missing or corrupt manifests, full commit-record drift, and derived metric/chart drift. The example archaeology workflow no longer swallows audit failures. - Runtime and distribution versions agree, and `devarch --version` identifies the installed distribution. ## Compatibility and boundaries @@ -23,4 +23,6 @@ This is deterministic commit-message analysis plus an auditable extraction path, Use the real-Git fixtures in `tests/test_release_pipeline.py`, the full test suite, and a clean wheel installation outside the source checkout. The fixtures cover bare repositories, worktrees, delimiter-bearing subjects, shallow rejection, timezone normalization, full pipeline execution and tampered-input audit rejection. Release only from reviewed canonical history; never bypass protections. Registry availability is separate from a GitHub release. -Local acceptance on Python 3.12: baseline 97 tests; repaired suite 104 tests passed. Wheel and source archive passed Twine metadata validation. A separate non-editable wheel environment outside the checkout passed ten CLI invocations, including bare-repository mining, database build, all six analysis vectors, visualization, both report formats and audit. Git/CSV/SQLite totals reconciled to two fixture commits. These fixtures do not establish remote registry publication or native Windows/macOS acceptance. +Local acceptance on Python 3.12: baseline 97 tests; repaired suite 111 tests passed. Wheel and source archive passed Twine metadata validation. A separate non-editable wheel environment outside the checkout passed ten CLI invocations, including bare-repository mining, database build, all six analysis vectors, visualization, both report formats and audit. Git/CSV/SQLite totals reconciled to two fixture commits. These fixtures do not establish remote registry publication or native Windows/macOS acceptance. + +Native matrix testing caught Python 3.10 rejection of Git UTC `Z` timestamps and Windows doubled CSV newlines. Regression fixtures reproduce both; extraction now normalizes `Z` and writes explicit LF records. Independent adversarial review reproduced false audit passes for corrupt manifests, changed database records and false metrics; fail-closed validation now recomputes source measurements, including chart histograms. diff --git a/tests/test_release_pipeline.py b/tests/test_release_pipeline.py index 31dd1ea..5ec2b83 100644 --- a/tests/test_release_pipeline.py +++ b/tests/test_release_pipeline.py @@ -98,3 +98,62 @@ def test_audit_detects_modified_mining_input(repo, tmp_path, monkeypatch): path = tmp_path / 'projects/audit-fixture/data/github-commits.csv' path.write_text(path.read_text().replace('init','changed')) assert any(f.code == 'MINING_ARTIFACT_DRIFT' for f in check_mined_history('audit-fixture', tmp_path)) + + +def test_git_z_timestamp_is_portable(tmp_path, monkeypatch): + import archaeology.extractors.git as extractor + def read_git(repo_path, *args): + return 'a\x002026-01-02T07:30:00Z\x00subject\x00author\x00' if args[0] == 'log' else '1\n' + monkeypatch.setattr(extractor, '_git', read_git) + out = tmp_path / 'utc.csv' + assert extract_git_log(str(tmp_path), str(out)) == 1 + with out.open(newline='') as f: row = next(csv.DictReader(f)) + assert row['date'] == '2026-01-02T07:30:00+00:00' + + +def test_csv_export_under_windows_newline_translation(repo, tmp_path, monkeypatch): + import archaeology.utils as utils + def windows_write(path, content, encoding='utf-8'): + Path(path).write_bytes(content.replace('\n', '\r\n').encode(encoding)) + monkeypatch.setattr(utils, 'atomic_write', windows_write) + out = tmp_path / 'windows.csv' + assert extract_git_log(str(repo), str(out)) == 2 + with out.open(newline='', encoding='utf-8') as handle: + rows = list(csv.reader(handle)) + assert len(rows) == 3 and all(len(row) == 4 for row in rows) + + +@pytest.mark.parametrize('corruption', ['manifest', 'database', 'metrics', 'visual']) +def test_audit_rejects_semantic_corruption(repo, tmp_path, monkeypatch, corruption): + import sqlite3 + monkeypatch.chdir(tmp_path); runner = CliRunner() + for args in (['init','corrupt'],['mine',str(repo),'-p','corrupt'],['build-db','corrupt']): + result=runner.invoke(main,args); assert result.exit_code == 0, result.output + project=tmp_path/'projects/corrupt' + if corruption == 'manifest': + (project/'data/coverage.json').write_text('{not-json') + elif corruption == 'database': + with sqlite3.connect(project/'data/archaeology.db') as conn: + # Deliberately corrupt base records, bypassing FTS maintenance in this fixture. + triggers = conn.execute("SELECT name FROM sqlite_master WHERE type='trigger' AND tbl_name='commits'").fetchall() + for (name,) in triggers: + conn.execute('DROP TRIGGER "' + name.replace('"', '""') + '"') + conn.execute("UPDATE commits SET date='1900-01-01', message='changed'") + elif corruption == 'visual': + visual=project/'deliverables/data.json';data=json.loads(visual.read_text()) + data['daily_commits']={'1900-01-01':2};visual.write_text(json.dumps(data)) + else: + canonical=project/'deliverables/canonical-metrics.json' + data=json.loads(canonical.read_text());data['active_days']=9999;data['span_days']=9999 + canonical.write_text(json.dumps(data)) + visual=project/'deliverables/data.json';data2=json.loads(visual.read_text()) + data2['telemetry_visualizations']['meta']=data;visual.write_text(json.dumps(data2)) + result=runner.invoke(main,['audit','corrupt']) + assert result.exit_code != 0, result.output + assert 'HIGH' in result.output or 'CRITICAL' in result.output + + +def test_coverage_records_detached_head(repo): + git(repo, 'checkout', '--detach') + git(repo, 'commit', '--allow-empty', '-m', 'detached') + assert repository_coverage(str(repo))['head'] == git(repo, 'rev-parse', 'HEAD') From 6377b1832f823ef3369b86c3a0baa86d52539ab2 Mon Sep 17 00:00:00 2001 From: Simon Gonzalez De Cruz Date: Fri, 2 Oct 2026 05:13:01 +0000 Subject: [PATCH 4/4] fix: reject corrupted evidence and normalize native Git output --- skills/dev-archaeology/SKILL.md | 35 +++++++++++++++++++++------------ 1 file changed, 22 insertions(+), 13 deletions(-) diff --git a/skills/dev-archaeology/SKILL.md b/skills/dev-archaeology/SKILL.md index 2483e23..1985d72 100644 --- a/skills/dev-archaeology/SKILL.md +++ b/skills/dev-archaeology/SKILL.md @@ -31,24 +31,29 @@ Run stages individually for granular control: Use the Python CLI for automated execution: ```bash -cd /path/to/devarch-framework -python archaeology/cli.py setup -python archaeology/cli.py mine -python archaeology/cli.py build -python archaeology/cli.py detect -python archaeology/cli.py analyze -python archaeology/cli.py visualize -python archaeology/cli.py report -python archaeology/cli.py audit +# Install the released distribution in an isolated environment first. +# Run from an analysis workspace, not from the repository being examined. +devarch --version +devarch init example --repo-url https://github.com/OWNER/REPO +devarch mine /path/to/repo -p example +devarch build-db example +devarch signals example +# Review signal thresholds and evidence before analysis. +devarch analyze example +# Review keyword candidates against commit/source evidence before reporting. +devarch visualize example +devarch export-report example +devarch export-report example --format html +devarch audit example ``` ### Keyword Triggers Use these keywords with Claude Code: -- `setup` -- Initialize new project +- `init ` -- Initialize new project - `status` -- Show current stage and progress -- `mine ` -- Extract git data +- `mine -p ` -- Extract local Git data - `audit` -- Run validation checks - `add-supplement ` -- Add external data source @@ -82,7 +87,7 @@ Best for: Process: - Run CLI commands sequentially -- Outputs go to stage output/ folders +- CLI outputs go to `projects//data` and `projects//deliverables`; stage output folders belong to the manual workflow - Checkpoints still require manual review - Audit validates all outputs @@ -116,7 +121,7 @@ Supported data types: ## Outputs -Final outputs in stages/07-report/output/: +The installed CLI writes final outputs under `projects//deliverables/`. The manual workflow uses `stages/07-report/output/`: - ARCHAEOLOGY-REPORT.md -- Markdown report - ARCHAEOLOGY-REPORT.html -- HTML with visualizations @@ -167,3 +172,7 @@ This skill follows ICM conventions: - Layer 2: Stage CONTEXT.md files specify contracts - Layer 3: references/ provide specifications - Layer 4: output/ folders contain results + +## Evidence boundaries + +Mining covers all locally available refs plus HEAD, refuses shallow history, and does not fetch remotes. Establish authorized remote/ref coverage separately before claiming an entire available history. `data/coverage.json` records HEAD, refs, roots, counts and artifact hashes. Audit reconciles extracted records and measured metrics; it is not a cryptographic attestation of repository provenance. The automated vectors use commit-message heuristics, not source-level reasoning. Mark architecture rationales as explicit or inferred and cite commit/PR/source evidence. Never treat keyword absence as proof a capability is missing. Do not ingest personal sessions or supplementary data without task relevance and authorization.