Skip to content

Commit e1c4cc3

Browse files
committed
Add catalog and job status commands for DataHerb Explorer
Adds `dataherb catalog build|validate|lint|serve` and `dataherb status emit|check`, backed by a new dataherb.catalog package (config, stores for git/s3/http/local, metadata resolution, JSON Schemas, the dataherb.status/v1 job status spec, metadata quality lint). `dataherb create` now scaffolds a v2 dataherb.json from the data files (with --no-input for scripts), and `dataherb validate` adds a schema check and a metadata quality score. `dataherb upload` uses the current folder instead of the import-time cwd. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_016Zz2NMq7okdD9v71q6vtDq
1 parent ef808f5 commit e1c4cc3

31 files changed

Lines changed: 3069 additions & 142 deletions

‎MANIFEST.in‎

Lines changed: 1 addition & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -1,2 +1,3 @@
11
include README.md
22
graft dataherb/serve/mkdocs_template
3+
graft dataherb/catalog/schemas

‎README.md‎

Lines changed: 21 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -56,10 +56,12 @@ dataherb download covid19_eu_data
5656

5757
We provide a template for dataset creation.
5858

59-
Within a dataset folder where the data files are located, use the following command line tool to create the metadata template.
59+
Within a dataset folder where the data files are located, use the following command line tool to create the metadata. Columns, types and row counts are inferred from csv, tsv, parquet and json files.
6060

6161
```bash
62-
dataherb create
62+
dataherb create # interactive
63+
dataherb create --no-input # infer only
64+
dataherb validate # check it
6365
```
6466

6567
### Upload dataset to remote
@@ -70,6 +72,23 @@ Within the dataset folder, run
7072
dataherb upload
7173
```
7274

75+
### Catalog website and job status
76+
77+
Build a static catalog and explorer website ([DataHerb Explorer](https://github.com/DataHerb/dataherb-explorer)) from a `dataherb.config.yml`:
78+
79+
```bash
80+
dataherb catalog validate # check config and catalog entries
81+
dataherb catalog lint # metadata quality per dataset
82+
dataherb catalog build # write the site to dist/
83+
```
84+
85+
Report job runs so the catalog can show freshness and failures:
86+
87+
```bash
88+
dataherb status emit --target s3://bucket/_dataherb/status/ --job-id my-crawler --status success --expected-interval P1D
89+
dataherb status check # exit 1 if any job is failing, stuck or stale
90+
```
91+
7392
### UI for all the datasets in a flora
7493

7594

‎dataherb/catalog/__init__.py‎

Lines changed: 5 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,5 @@
1+
"""Static data catalog tooling: build a DataHerb Explorer site from a YAML
2+
config, validate and lint catalog entries, and read or write job status files.
3+
4+
The site template lives in https://github.com/DataHerb/dataherb-explorer.
5+
"""

‎dataherb/catalog/build.py‎

Lines changed: 151 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,151 @@
1+
"""`dataherb catalog build`: config + catalog entries + job status -> a static site in dist/."""
2+
3+
from __future__ import annotations
4+
5+
import shutil
6+
from dataclasses import dataclass
7+
from pathlib import Path
8+
9+
from dataherb.version import __version__
10+
from .resolve import BuildIssue, build_catalog, load_entries
11+
from .config import Config
12+
from .lint import lint_dataset
13+
from .status import HEALTH_ORDER, collect
14+
from .stores import LocalStore, make_stores
15+
from .util import dump_json, iso, log, utcnow
16+
from .validate import errors
17+
18+
19+
@dataclass
20+
class BuildSummary:
21+
datasets: int
22+
jobs: int
23+
errors: int
24+
warnings: int
25+
out: Path
26+
27+
28+
def validate_inputs(cfg: Config) -> list[BuildIssue]:
29+
issues = [
30+
BuildIssue("error", None, f"dataherb.config.yml {e}")
31+
for e in errors("config", cfg.data)
32+
]
33+
entries, load_issues = load_entries(cfg)
34+
issues.extend(load_issues)
35+
for e in entries:
36+
doc = {k: v for k, v in e.items() if k != "_file"}
37+
for msg in errors("catalog-entry", doc):
38+
issues.append(BuildIssue("error", e.get("id"), f"{e['_file']} {msg}"))
39+
if e.get("inline"):
40+
for msg in errors("dataset", doc):
41+
issues.append(BuildIssue("warning", e.get("id"), f"{e['_file']} {msg}"))
42+
return issues
43+
44+
45+
def link_jobs(datasets: list[dict], jobs: list[dict]) -> None:
46+
"""Attach job health to datasets, via status_job in metadata or datasets[] in status files."""
47+
by_dataset: dict[str, set[str]] = {}
48+
for j in jobs:
49+
for d in (j.get("state") or {}).get("datasets") or []:
50+
if d.get("id"):
51+
by_dataset.setdefault(str(d["id"]), set()).add(j["id"])
52+
job_by_id = {j["id"]: j for j in jobs}
53+
for d in datasets:
54+
ids = list(
55+
dict.fromkeys(
56+
[*d.get("status_jobs", []), *sorted(by_dataset.get(d["id"], ()))]
57+
)
58+
)
59+
d["status_jobs"] = ids
60+
healths = [job_by_id[i]["health"] for i in ids if i in job_by_id]
61+
d["health"] = min(healths, key=HEALTH_ORDER.index) if healths else None
62+
updated = []
63+
for i in ids:
64+
jb = job_by_id.get(i)
65+
if not jb:
66+
continue
67+
updated.append(jb.get("last_success_at"))
68+
for ds in (jb.get("state") or {}).get("datasets") or []:
69+
if str(ds.get("id")) == d["id"] and ds.get("data_updated_at"):
70+
updated.append(ds["data_updated_at"])
71+
d["updated_at"] = max((u for u in updated if u), default=None)
72+
73+
74+
def build(
75+
cfg: Config, out: Path, site_dir: Path | None = None, strict: bool = False
76+
) -> BuildSummary:
77+
started = utcnow()
78+
site_dir = site_dir or cfg.root / "site"
79+
if not (site_dir / "index.html").is_file():
80+
raise FileNotFoundError(
81+
f"no site template at {site_dir}; build from a fork of "
82+
"https://github.com/DataHerb/dataherb-explorer or pass --site"
83+
)
84+
issues = validate_inputs(cfg)
85+
86+
log.info("resolving catalog")
87+
cat = build_catalog(cfg)
88+
issues.extend(cat.issues)
89+
90+
log.info("collecting job status")
91+
jobs, status_issues = collect(cfg, now=started)
92+
issues.extend(BuildIssue(**i) for i in status_issues)
93+
94+
link_jobs(cat.datasets, jobs)
95+
known = {j["id"] for j in jobs}
96+
for d in cat.datasets:
97+
d["quality"] = lint_dataset(d, known_jobs=known)
98+
99+
# Assemble the site.
100+
if out.exists():
101+
shutil.rmtree(out)
102+
shutil.copytree(site_dir, out, ignore=shutil.ignore_patterns("data", ".DS_Store"))
103+
for store in make_stores(cfg.stores, cfg.root).values():
104+
if isinstance(store, LocalStore):
105+
store.publish(out)
106+
if (
107+
cfg.explorer.get("duckdb", {}).get("mode") == "vendored"
108+
and not (out / "vendor" / "duckdb").exists()
109+
):
110+
issues.append(
111+
BuildIssue(
112+
"error",
113+
None,
114+
"explorer.duckdb.mode is 'vendored' but site/vendor/duckdb is missing; run `npm ci && npm run vendor`",
115+
)
116+
)
117+
118+
visible = [d for d in cat.datasets if not d.get("hidden")]
119+
n_err = sum(1 for i in issues if i.level == "error")
120+
n_warn = sum(1 for i in issues if i.level == "warning")
121+
generated = iso(utcnow())
122+
dump_json(
123+
{**cfg.public(), "generated_at": generated, "version": __version__},
124+
out / "data" / "config.json",
125+
)
126+
dump_json(
127+
{"generated_at": generated, "datasets": visible}, out / "data" / "catalog.json"
128+
)
129+
dump_json({"generated_at": generated, "jobs": jobs}, out / "data" / "status.json")
130+
dump_json(
131+
{
132+
"generated_at": generated,
133+
"duration_seconds": round((utcnow() - started).total_seconds(), 2),
134+
"version": __version__,
135+
"datasets": len(visible),
136+
"jobs": len(jobs),
137+
"errors": n_err,
138+
"warnings": n_warn,
139+
"issues": [i.as_dict() for i in issues],
140+
},
141+
out / "data" / "build.json",
142+
pretty=True,
143+
)
144+
(out / ".nojekyll").write_text("")
145+
if strict and n_err:
146+
raise SystemExit(
147+
f"build finished with {n_err} error(s); see {out / 'data' / 'build.json'}"
148+
)
149+
return BuildSummary(
150+
datasets=len(visible), jobs=len(jobs), errors=n_err, warnings=n_warn, out=out
151+
)

‎dataherb/catalog/config.py‎

Lines changed: 119 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,119 @@
1+
"""Load dataherb.config.yml and fill defaults."""
2+
3+
from __future__ import annotations
4+
5+
import copy
6+
from dataclasses import dataclass
7+
from pathlib import Path
8+
from typing import Any
9+
10+
from .util import load_file
11+
12+
DEFAULTS: dict[str, Any] = {
13+
"site": {
14+
"title": "DataHerb Explorer",
15+
"description": "",
16+
"logo": None,
17+
"accent": "#2f7d4f",
18+
"links": [],
19+
"repository": None,
20+
},
21+
"stores": {
22+
"github": {
23+
"type": "git",
24+
"raw_url_template": "https://raw.githubusercontent.com/{repo}/{ref}/{path}",
25+
"web_url_template": "https://github.com/{repo}",
26+
"default_ref": "HEAD",
27+
"token_env": "GITHUB_TOKEN",
28+
},
29+
"local": {"type": "local", "path": "demo"},
30+
},
31+
"catalog": {"dirs": ["catalog"], "discover": []},
32+
"status": {"sources": [], "live": True, "stale_grace": 0.5, "history": 30},
33+
"explorer": {
34+
"enabled": True,
35+
"duckdb": {
36+
"mode": "cdn",
37+
"version": "1.29.0",
38+
"cdn_base": "https://cdn.jsdelivr.net/npm/@duckdb/duckdb-wasm@{version}",
39+
},
40+
"preview_rows": 200,
41+
"max_browser_bytes": 500_000_000,
42+
},
43+
"snippets": [],
44+
}
45+
46+
47+
def _merge(base: Any, override: Any) -> Any:
48+
if isinstance(base, dict) and isinstance(override, dict):
49+
out = dict(base)
50+
for k, v in override.items():
51+
out[k] = _merge(base.get(k), v) if k in base else v
52+
return out
53+
return copy.deepcopy(override) if override is not None else copy.deepcopy(base)
54+
55+
56+
@dataclass
57+
class Config:
58+
data: dict
59+
root: Path
60+
61+
@property
62+
def site(self) -> dict:
63+
return self.data["site"]
64+
65+
@property
66+
def stores(self) -> dict:
67+
return self.data["stores"]
68+
69+
@property
70+
def catalog(self) -> dict:
71+
return self.data["catalog"]
72+
73+
@property
74+
def status(self) -> dict:
75+
return self.data["status"]
76+
77+
@property
78+
def explorer(self) -> dict:
79+
return self.data["explorer"]
80+
81+
def path(self, p: str) -> Path:
82+
q = Path(p)
83+
return q if q.is_absolute() else self.root / q
84+
85+
def public(self) -> dict:
86+
"""The subset of the config the browser gets. Never includes store credentials."""
87+
stores = {}
88+
for name, s in self.stores.items():
89+
stores[name] = {
90+
k: v
91+
for k, v in s.items()
92+
if k in ("type", "web_url_template", "public_base_url")
93+
}
94+
return {
95+
"site": self.site,
96+
"stores": stores,
97+
"status": {k: self.status.get(k) for k in ("live", "stale_grace")},
98+
"explorer": self.explorer,
99+
"snippets": self.data.get("snippets", []),
100+
}
101+
102+
103+
def load_config(path: str | Path = "dataherb.config.yml") -> Config:
104+
path = Path(path).resolve()
105+
raw = load_file(path) or {}
106+
if not isinstance(raw, dict):
107+
raise ValueError(f"{path}: expected a mapping at the top level")
108+
# Stores are replaced wholesale when given, so a fork can drop the demo stores.
109+
data = _merge(DEFAULTS, {k: v for k, v in raw.items() if k != "stores"})
110+
data["stores"] = raw.get("stores") or copy.deepcopy(DEFAULTS["stores"])
111+
for name, store in data["stores"].items():
112+
if not isinstance(store, dict) or "type" not in store:
113+
raise ValueError(f"store '{name}' needs a type (git, s3, http, local)")
114+
return Config(data=data, root=path.parent)
115+
116+
117+
def load_config_defaults(root: str | Path = ".") -> Config:
118+
"""A Config with only the defaults, for checking a single dataset outside a catalog."""
119+
return Config(data=copy.deepcopy(DEFAULTS), root=Path(root).resolve())

0 commit comments

Comments
 (0)