Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion .github/workflows/archaeology.yml
Original file line number Diff line number Diff line change
Expand Up @@ -68,7 +68,7 @@ jobs:
run: devarch export-report "$PROJECT_NAME"

- name: Run audit
run: devarch audit "$PROJECT_NAME" || true
run: devarch audit "$PROJECT_NAME"

- name: Upload archaeology report
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
Expand Down
1 change: 1 addition & 0 deletions .github/workflows/publish.yml
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,7 @@ on:
- 'v*.*.*'

permissions:
contents: read
id-token: write # Trusted publishing

jobs:
Expand Down
4 changes: 4 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
@@ -1,3 +1,7 @@
# 0.4.0 — Evidence integrity

Fix full-history mining for bare repositories/worktrees, reject shallow coverage, bind extraction artifacts, reconcile CSV/SQLite metrics, and generate portable measured visualizations. Replace unsupported agent/ML/quality assertions with explicit uncertainty. See [release notes](docs/RELEASE_0.4.0.md) for output compatibility and validation.

# Changelog

All notable changes to DevArch Framework will be documented in this file.
Expand Down
4 changes: 4 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -17,6 +17,10 @@ DevArch treats your git history as structured data. It extracts commits into a q

> **Just want a quick learning diagnostic?** [Dev Learning Archaeologist](https://github.com/KyaniteLabs/dev-learning-archaeologist) is a zero-setup ICM folder — drop it in any project, run through Claude Code, no install required.

## Evidence integrity (0.4.0)

Mining covers all **locally available refs**, records coverage and rejects shallow history. Fetch the authorized branches/tags/PR refs before mining when remote completeness matters. Automated vectors are keyword-based investigation leads, not verified source conclusions, productivity measurements or causal explanations. The default visualization contains measured daily activity only. See [0.4.0 release notes](docs/RELEASE_0.4.0.md).

## What It Does

DevArch transforms git history into structured insights through a full-featured CLI with 20+ commands. The framework supports:
Expand Down
2 changes: 1 addition & 1 deletion archaeology/__init__.py
Original file line number Diff line number Diff line change
@@ -1,3 +1,3 @@
"""DevArch Framework - forensic mining of software development history."""

__version__ = "0.1.0"
__version__ = "0.4.0"
83 changes: 38 additions & 45 deletions archaeology/analysis_runner.py
Original file line number Diff line number Diff line change
Expand Up @@ -45,28 +45,26 @@ def _log(self, msg: str) -> None:
def _query_db(self, query: str, params: tuple = ()) -> list[dict]:
"""Execute SQL query against archaeology database."""
if not self.db_path.exists():
return []
raise ValueError("Analysis database missing; run build-db first")
conn = sqlite3.connect(str(self.db_path), timeout=30)
conn.row_factory = sqlite3.Row
try:
cursor = conn.execute(query, params)
return [dict(row) for row in cursor.fetchall()]
except sqlite3.Error as e:
if self.verbose:
print(f" [analysis] Database query error: {e}")
return []
raise ValueError(f"Analysis query failed: {e}") from e
finally:
conn.close()

def _load_json(self, rel_path: str) -> Any | None:
"""Load JSON from project directory (wrapper for utils._load_json)."""
return _load_json(self.project_dir / rel_path)

def _like_commits(self, keywords: list[str], limit: int = 100) -> list[dict]:
def _like_commits(self, keywords: list[str], limit: int | None = 100) -> list[dict]:
if not keywords:
return []
clauses = " OR ".join("LOWER(message) LIKE ?" for _ in keywords)
params = tuple(f"%{kw.lower()}%" for kw in keywords) + (limit,)
params = tuple(f"%{kw.lower()}%" for kw in keywords) + (-1 if limit is None else limit,)
return self._query_db(
f"SELECT hash, date, message, author FROM commits WHERE {clauses} ORDER BY date DESC LIMIT ?",
params,
Expand All @@ -80,16 +78,16 @@ def run_sdlc_gap_finder(self) -> dict[str, Any]:
"""Analyze SDLC practices and gaps."""
self._log("Running SDLC Gap Finder...")
total_commits = self._commit_count()
ci_cd = self._like_commits(["github action", "ci", "workflow", "deploy", "pipeline"], 500)
tests = self._like_commits(["test", "spec", "coverage", "vitest", "pytest"], 500)
refactor = self._like_commits(["refactor", "clean", "simplify"], 500)
security = self._like_commits(["security", "cve", "xss", "injection", "secret"], 500)
docs = self._like_commits(["docs", "readme", "documentation"], 500)
ci_cd = self._like_commits(["github action", "ci", "workflow", "deploy", "pipeline"], None)
tests = self._like_commits(["test", "spec", "coverage", "vitest", "pytest"], None)
refactor = self._like_commits(["refactor", "clean", "simplify"], None)
security = self._like_commits(["security", "cve", "xss", "injection", "secret"], None)
docs = self._like_commits(["docs", "readme", "documentation"], None)

def status(count: int, low: float, high: float) -> str:
ratio = count / total_commits if total_commits else 0
if ratio < low:
return "ABSENT"
return "UNVERIFIED"
if ratio < high:
return "EMERGING"
return "PRESENT"
Expand All @@ -104,12 +102,14 @@ def status(count: int, low: float, high: float) -> str:
gaps = []
for practice, rows, low, high, recommendation in practices:
practice_status = status(len(rows), low, high)
severity = "HIGH" if practice_status == "ABSENT" else "MEDIUM" if practice_status == "EMERGING" else "LOW"
severity = "MEDIUM" if practice_status == "UNVERIFIED" else "MEDIUM" if practice_status == "EMERGING" else "LOW"
gaps.append(
{
"practice": practice,
"status": practice_status,
"evidence": [{"result_count": len(rows), "ratio": f"{(len(rows) / total_commits if total_commits else 0):.1%}"}],
"confidence": "LOW",
"interpretation": "Commit-keyword frequency only; not verified presence, absence or coverage",
"evidence": [{"sample": rows[:5], "result_count": len(rows), "ratio": f"{(len(rows) / total_commits if total_commits else 0):.1%}"}],
"severity": severity,
"effort_to_implement": 3 if severity == "HIGH" else 2,
"expected_impact": 5 if severity == "HIGH" else 3,
Expand Down Expand Up @@ -148,11 +148,12 @@ def run_ml_pattern_mapper(self) -> dict[str, Any]:
{
"intuitive_name": intuitive,
"formal_term": formal,
"confidence": "HIGH" if len(evidence) >= 5 else "MEDIUM",
"similarity_to_canonical": min(0.9, 0.45 + len(evidence) * 0.05),
"is_reinvention": reinvention,
"confidence": "LOW",
"status": "UNVERIFIED keyword candidate; inspect source before assigning an algorithm",
"similarity_to_canonical": None,
"is_reinvention": None,
"library_alternative": library,
"estimated_token_waste": 5000 if reinvention else None,
"estimated_token_waste": None,
"evidence": evidence[:5],
}
)
Expand Down Expand Up @@ -213,28 +214,18 @@ def _approximate_sessions(self) -> list[dict]:
def run_agentic_workflow(self) -> dict[str, Any]:
"""Analyze AI agent interaction patterns."""
self._log("Running Agentic Workflow Analyzer...")
sessions = self._approximate_sessions()
hooks = self._like_commits(["hook", "pre-commit", "post-commit", "automation"], 50)
agent_commits = self._query_db("SELECT author, COUNT(*) as cnt FROM commits GROUP BY author ORDER BY cnt DESC")
authors = self._query_db("SELECT author, COUNT(*) as cnt FROM commits GROUP BY author ORDER BY cnt DESC")
return {
"project": self.project_name,
"analysis_date": datetime.now().isoformat(),
"session_depth_distribution": {
"sessions_total": len(sessions),
"micro_lt5": max(0, len(sessions) // 6),
"standard_5_20": max(0, len(sessions) // 2),
"deep_20_50": max(0, len(sessions) // 4),
"marathon_50_plus": max(0, len(sessions) - (len(sessions) // 6 + len(sessions) // 2 + len(sessions) // 4)),
},
"session_taxonomy": {
"SCAFFOLDING": len(self._like_commits(["scaffold", "initialize", "setup"], 100)),
"BUILDING": len(self._like_commits(["feat", "implement", "add"], 100)),
"DEBUGGING": len(self._like_commits(["fix", "debug", "error"], 100)),
"REFACTORING": len(self._like_commits(["refactor", "cleanup", "simplify"], 100)),
},
"hook_effectiveness": [{"hook_name": "automation/hook commits", "effectiveness_score": 0.8, "evidence_count": len(hooks)}] if hooks else [],
"agent_attribution": agent_commits,
"summary": {"total_sessions_analyzed": len(sessions), "dominant_session_type": "BUILDING"},
"session_depth_distribution": None,
"session_taxonomy": None,
"hook_effectiveness": [],
"hook_commit_evidence": hooks,
"author_attribution": authors,
"limitations": "Commit authors are not verified agent identities. Session depth, autonomy and hook effectiveness are unmeasured; no fabricated estimates.",
"summary": {"total_sessions_analyzed": 0, "dominant_session_type": None},
}

def run_formal_terms_mapper(self) -> dict[str, Any]:
Expand All @@ -256,23 +247,24 @@ def run_formal_terms_mapper(self) -> dict[str, Any]:
"code_name": code_name,
"formal_term": formal,
"category": "ARCHITECTURE",
"similarity_score": "CLOSE" if len(evidence) >= 3 else "PARTIAL",
"similarity_score": "UNVERIFIED",
"confidence": "LOW",
"evidence": evidence,
}
)
return {
"project": self.project_name,
"analysis_date": datetime.now().isoformat(),
"term_dictionary": dictionary,
"naming_trajectory": "Project-specific metaphors are increasingly mapped onto formal control-loop, pipeline, and verification vocabulary.",
"naming_trajectory": "Unmeasured; keyword candidates require source validation.",
"learning_opportunities": ["Control theory", "Quality-diversity algorithms", "Event sourcing", "Multi-agent evaluation"],
"summary": {"terms_mapped": len(dictionary), "high_confidence": sum(1 for t in dictionary if t["similarity_score"] == "CLOSE")},
}

def run_source_archaeologist(self) -> dict[str, Any]:
"""Mine commit history for code quality trajectory and hotspots."""
self._log("Running Source Code Archaeologist...")
quality = self._like_commits(["fix", "test", "refactor", "security", "lint", "type"], 500)
quality = self._like_commits(["fix", "test", "refactor", "security", "lint", "type"], None)
large_change = self._like_commits(["split", "extract", "monolith", "decompose", "simplify"], 100)
todo = self._like_commits(["todo", "stub", "placeholder", "not implemented"], 100)
by_month: Counter[str] = Counter()
Expand All @@ -284,7 +276,7 @@ def run_source_archaeologist(self) -> dict[str, Any]:
improvements = self._derive_improvements(quality, large_change, todo, hotspots)
return {
"analysis_metadata": {"timestamp": datetime.now().isoformat(), "analyst": "Automated Source Code Archaeologist", "project": self.project_name, "commit_count": self._commit_count()},
"quality_trajectory": {"assessment": "IMPROVING" if quality else "UNKNOWN", "evidence_count": len(quality), "by_month": dict(sorted(by_month.items()))},
"quality_trajectory": {"assessment": "UNVERIFIED keyword activity; no quality direction established", "evidence_count": len(quality), "by_month": dict(sorted(by_month.items()))},
"architecture_drift": {"large_change_signals": large_change[:10], "todo_or_stub_signals": todo[:10]},
"hotspots": hotspots,
"improvements": improvements,
Expand All @@ -307,23 +299,23 @@ def _derive_improvements(
top_msg = str(flapping[0].get("message", ""))[:60]
items.append((
100,
f"Fix recurring issue: {top_msg}",
f"Investigate repeated message (may be merge/cherry-pick duplication): {top_msg}",
"M", "HIGH",
))

# Unresolved stubs / TODOs
if todo:
items.append((
90 if len(todo) >= 5 else 70,
f"Resolve {len(todo)} stub or placeholder commit(s)",
f"Check whether {len(todo)} historical stub/placeholder mentions remain unresolved",
"S", "HIGH" if len(todo) >= 5 else "MEDIUM",
))

# Decomposition momentum: carry it through
if large_change:
items.append((
60,
f"Continue decomposition — {len(large_change)} large-change signal(s) detected",
f"Review {len(large_change)} historical decomposition signals before proposing more splits",
"L", "MEDIUM",
))

Expand All @@ -345,11 +337,11 @@ def _derive_improvements(

# No issues found: project is healthy
if not items:
items.append((10, "No critical remediation items — maintain current trajectory", "S", "LOW"))
items.append((10, "No keyword-derived candidates; source review still required", "S", "LOW"))

items.sort(key=lambda x: x[0], reverse=True)
return [
{"rank": i + 1, "title": title, "effort": effort, "impact": impact}
{"rank": i + 1, "title": title, "effort": effort, "impact": impact, "status": "UNVERIFIED investigation candidate", "evidence": (todo[:3] if "placeholder" in title else large_change[:3] if "decomposition" in title else hotspots[:3] if "repeated" in title else quality[:3])}
for i, (_, title, effort, impact) in enumerate(items)
]

Expand Down Expand Up @@ -406,6 +398,7 @@ def run_all(self, vectors: list[str] | None = None) -> dict[str, str]:
try:
output_path = analysis_dir / f"analysis-{vector_name}.json"
result = runner_func()
result["methodology"] = {"basis": "commit-message heuristics", "source_inspection": False, "causal_inference": False, "limitations": "Requires source/PR validation. Missing keyword evidence does not establish absence. Repeated commits across refs are not necessarily recurring defects."}
atomic_write(output_path, json.dumps(result, indent=2, ensure_ascii=False) + "\n")
results[vector_name] = str(output_path)
print(f" [analysis] {vector_name}: {output_path}")
Expand Down
52 changes: 52 additions & 0 deletions archaeology/audit.py
Original file line number Diff line number Diff line change
Expand Up @@ -319,6 +319,7 @@ def run_audit(project_name: str, root: str | Path = ".") -> list[AuditFinding]:

for check in (
check_project_config,
check_mined_history,
check_canonical_consistency,
check_placeholder_data,
check_sensitive_artifacts,
Expand All @@ -339,3 +340,54 @@ def summarize(findings: Iterable[AuditFinding]) -> dict[str, int]:
for finding in findings:
summary[finding.severity] = summary.get(finding.severity, 0) + 1
return summary


def check_mined_history(project_name: str, root: Path) -> list[AuditFinding]:
"""Reconcile extraction identities and byte bindings when mining evidence exists."""
import csv
import hashlib
project = _project_dir(project_name, root)
coverage_path = project / 'data' / 'coverage.json'
coverage = _load_json(coverage_path)
config = _load_json(project / 'project.json') or {}
if not coverage_path.exists() and not config.get('mined_history_manifest_required'):
return [] # Legacy imported datasets never claimed a mining manifest.
if not isinstance(coverage, dict) or not isinstance(coverage.get('artifact_sha256'), dict):
return [AuditFinding('HIGH', 'MINING_MANIFEST_INVALID', 'Mining coverage manifest missing or invalid')]
findings = []
expected_names = {'github-commits.csv', 'github-commits-with-stats.txt'}
if set(coverage['artifact_sha256']) != expected_names:
findings.append(AuditFinding('HIGH', 'MINING_MANIFEST_INVALID', 'Mining artifact bindings incomplete'))
for name, expected in coverage['artifact_sha256'].items():
if name not in expected_names:
findings.append(AuditFinding('HIGH', 'MINING_MANIFEST_INVALID', 'Unexpected artifact name'))
continue
path = project / 'data' / name
if not path.exists() or hashlib.sha256(path.read_bytes()).hexdigest() != expected:
findings.append(AuditFinding('HIGH', 'MINING_ARTIFACT_DRIFT', f'Mined artifact changed: {name}'))
csv_path = project / 'data' / 'github-commits.csv'
db_path = project / 'data' / 'archaeology.db'
try:
with csv_path.open(encoding='utf-8', newline='') as handle:
rows = list(csv.DictReader(handle))
keys = ('hash', 'date', 'message', 'author')
source = [tuple(row[key] for key in keys) for row in rows]
hashes = [row['hash'] for row in rows]
if not db_path.exists():
raise ValueError('Mined history database is missing')
with sqlite3.connect(db_path) as conn:
stored = list(conn.execute('SELECT hash,date,message,author FROM commits'))
if len(hashes) != coverage.get('commit_count') or len(set(hashes)) != len(hashes) or sorted(source) != sorted(stored):
raise ValueError('Git manifest, CSV and SQLite commit records do not reconcile')
from .metrics import calculate_metrics
measured = calculate_metrics(rows)
canonical = _load_json(project / 'deliverables' / 'canonical-metrics.json') or {}
if any(canonical.get(key) != value for key, value in measured.items()):
raise ValueError('Canonical metrics differ from source commit measurements')
visual = _load_json(project / 'deliverables' / 'data.json') or {}
meta = visual.get('telemetry_visualizations', {}).get('meta', {})
if any(visual.get(key) != value or meta.get(key) != value for key, value in measured.items()):
raise ValueError('Visualization metrics differ from source commit measurements')
except (OSError, ValueError, KeyError, TypeError, sqlite3.Error) as exc:
findings.append(AuditFinding('HIGH', 'MINING_HISTORY_DRIFT', str(exc)))
return findings
Loading
Loading