Skip to content

Commit cdd2e76

Browse files
committed
Read Markdown catalog entries; catalog add writes them
catalog/<id>.md holds the entry fields in YAML front matter and free text in the body, which becomes the dataset's documentation (before the dataset's own). .md files without front matter, like a README, are ignored. `dataherb catalog add` writes .md by default (--format yml for the old style) and keeps the body when overwriting with --force. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Ggg4AutCcRdK8c57NMxSqp
1 parent 818151f commit cdd2e76

8 files changed

Lines changed: 145 additions & 17 deletions

File tree

‎dataherb/catalog/add.py‎

Lines changed: 25 additions & 6 deletions
Original file line numberDiff line numberDiff line change
@@ -1,5 +1,8 @@
11
"""Add git repositories to the catalog as entries in catalog/.
22
3+
Entries are Markdown files: the fields go in the YAML front matter and the
4+
body is free text shown on the dataset page.
5+
36
A repo that already carries metadata (dataherb.json, dataherb.yml or the
47
legacy .dataherb/metadata.yml) gets a short pointer entry; the builder reads
58
the metadata on every build. A repo without metadata is cloned and scanned
@@ -22,7 +25,14 @@
2225
from .infer import scaffold
2326
from .resolve import METADATA_CANDIDATES, load_entries
2427
from .stores import GitStore, StoreError, git_key, make_stores
25-
from .util import FetchError, env_token, http_get, slugify
28+
from .util import (
29+
FetchError,
30+
env_token,
31+
front_matter_document,
32+
http_get,
33+
slugify,
34+
split_front_matter,
35+
)
2636

2737

2838
@dataclass
@@ -107,6 +117,7 @@ def add_repos(
107117
clone_url_template: str | None = None,
108118
force: bool = False,
109119
dry_run: bool = False,
120+
fmt: str = "md",
110121
) -> list[Added]:
111122
stores = make_stores(cfg.stores, cfg.root)
112123
git_stores = {n: s for n, s in stores.items() if isinstance(s, GitStore)}
@@ -154,7 +165,7 @@ def add_repos(
154165
continue
155166
did = str(existing["id"]) if existing else default_id(repo, strip_prefix)
156167
target = (
157-
cfg.path(existing["_file"]) if existing else out_dir / f"{did}.yml"
168+
cfg.path(existing["_file"]) if existing else out_dir / f"{did}.{fmt}"
158169
)
159170
if not existing and (did in known_ids or target.exists()) and not force:
160171
results.append(
@@ -205,16 +216,24 @@ def add_repos(
205216

206217
if not dry_run:
207218
target.parent.mkdir(parents=True, exist_ok=True)
208-
target.write_text(
209-
yaml.safe_dump(entry, sort_keys=False, allow_unicode=True),
210-
encoding="utf-8",
211-
)
219+
if target.suffix == ".md":
220+
text = front_matter_document(entry, _keep_body(target) if force else "")
221+
else:
222+
text = yaml.safe_dump(entry, sort_keys=False, allow_unicode=True)
223+
target.write_text(text, encoding="utf-8")
212224
known_ids.add(did)
213225
known_repos[repo.lower()] = {"id": did, "_file": str(target)}
214226
results.append(Added(repo, did, target, kind, note))
215227
return results
216228

217229

230+
def _keep_body(path: Path) -> str:
231+
"""The Markdown body of an entry that is being overwritten, so --force keeps hand-written notes."""
232+
if not path.exists():
233+
return ""
234+
return split_front_matter(path.read_text(encoding="utf-8"))[1]
235+
236+
218237
def org_token(cfg: Config, store_name: str | None) -> str | None:
219238
for name, conf in (cfg.stores or {}).items():
220239
if conf.get("type") == "git" and (store_name in (None, name)):

‎dataherb/catalog/build.py‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -33,7 +33,7 @@ def validate_inputs(cfg: Config) -> list[BuildIssue]:
3333
entries, load_issues = load_entries(cfg)
3434
issues.extend(load_issues)
3535
for e in entries:
36-
doc = {k: v for k, v in e.items() if k != "_file"}
36+
doc = {k: v for k, v in e.items() if not k.startswith("_")}
3737
for msg in errors("catalog-entry", doc):
3838
issues.append(BuildIssue("error", e.get("id"), f"{e['_file']} {msg}"))
3939
if e.get("inline"):

‎dataherb/catalog/resolve.py‎

Lines changed: 28 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -20,7 +20,15 @@
2020
join,
2121
make_stores,
2222
)
23-
from .util import FetchError, iso, load_file, load_structured, log, utcnow
23+
from .util import (
24+
FetchError,
25+
iso,
26+
load_file,
27+
load_structured,
28+
log,
29+
split_front_matter,
30+
utcnow,
31+
)
2432

2533
METADATA_CANDIDATES = (
2634
"dataherb.json",
@@ -29,6 +37,10 @@
2937
".dataherb/metadata.yml",
3038
)
3139

40+
# Catalog entry files. Markdown entries keep the fields in YAML front matter
41+
# and free text (shown on the dataset page) in the body.
42+
ENTRY_SUFFIXES = (".md", ".yml", ".yaml", ".json")
43+
3244
# Keys in a catalog entry that say where the dataset is, rather than describe it.
3345
LOCATOR_KEYS = {
3446
"store",
@@ -40,6 +52,7 @@
4052
"inline",
4153
"hidden",
4254
"_file",
55+
"_body",
4356
}
4457

4558
FORMAT_BY_EXT = {
@@ -109,10 +122,17 @@ def load_entries(cfg: Config) -> tuple[list[dict], list[BuildIssue]]:
109122
)
110123
continue
111124
for p in sorted(folder.rglob("*")):
112-
if p.suffix not in (".yml", ".yaml", ".json") or p.name.startswith("_"):
125+
if p.suffix not in ENTRY_SUFFIXES or p.name.startswith("_"):
113126
continue
127+
body = ""
114128
try:
115-
data = load_file(p)
129+
if p.suffix == ".md":
130+
data, body = split_front_matter(p.read_text(encoding="utf-8"))
131+
if data is None: # a README or notes, not an entry
132+
continue
133+
data = data or {}
134+
else:
135+
data = load_file(p)
116136
except Exception as e:
117137
issues.append(
118138
BuildIssue(
@@ -134,6 +154,8 @@ def load_entries(cfg: Config) -> tuple[list[dict], list[BuildIssue]]:
134154
continue
135155
item = dict(item)
136156
item["_file"] = str(p.relative_to(cfg.root)).replace("\\", "/")
157+
if body.strip():
158+
item["_body"] = body.strip()
137159
if "id" not in item:
138160
item["id"] = p.stem
139161
entries.append(item)
@@ -320,7 +342,9 @@ def normalize(
320342
"id": str(merged.get("id")),
321343
"name": merged.get("name") or merged.get("title") or str(merged.get("id")),
322344
"description": merged.get("description") or "",
323-
"documentation": merged.get("documentation") or "",
345+
"documentation": "\n\n".join(
346+
x for x in (entry.get("_body"), merged.get("documentation")) if x
347+
),
324348
"tags": sorted({str(t) for t in _as_list(merged.get("tags")) if t}),
325349
"domain": merged.get("domain"),
326350
"owner": _owner(merged.get("owner")),

‎dataherb/catalog/util.py‎

Lines changed: 18 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -28,6 +28,24 @@ def load_structured(text: str | bytes, name: str = "") -> Any:
2828
return yaml.safe_load(text)
2929

3030

31+
_FRONT_MATTER = re.compile(r"\A---[ \t]*\r?\n(.*?)^(?:---|\.\.\.)[ \t]*(?:\r?\n|\Z)", re.S | re.M)
32+
33+
34+
def split_front_matter(text: str) -> tuple[Any, str]:
35+
"""Split Markdown into (YAML front matter, body). Front matter is None when absent."""
36+
text = text.lstrip("\ufeff")
37+
m = _FRONT_MATTER.match(text)
38+
if not m:
39+
return None, text
40+
return yaml.safe_load(m.group(1)), text[m.end() :]
41+
42+
43+
def front_matter_document(data: dict, body: str = "") -> str:
44+
"""Markdown with data as YAML front matter."""
45+
head = yaml.safe_dump(data, sort_keys=False, allow_unicode=True)
46+
return f"---\n{head}---\n" + (f"\n{body.strip()}\n" if body.strip() else "")
47+
48+
3149
def load_file(path: Path) -> Any:
3250
return load_structured(path.read_text(encoding="utf-8"), path.name)
3351

‎dataherb/cmd/catalog.py‎

Lines changed: 16 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -143,7 +143,19 @@ def lint(config_path, min_score):
143143
help="GitHub API, for --org (GitHub Enterprise: https://HOST/api/v3).",
144144
)
145145
@click.option("--include-archived", is_flag=True, help="With --org: keep archived repos.")
146-
@click.option("--force", is_flag=True, help="Overwrite entries that already exist.")
146+
@click.option(
147+
"--format",
148+
"fmt",
149+
type=click.Choice(["md", "yml"]),
150+
default="md",
151+
show_default=True,
152+
help="Markdown with YAML front matter, or plain YAML.",
153+
)
154+
@click.option(
155+
"--force",
156+
is_flag=True,
157+
help="Overwrite entries that already exist (a Markdown body is kept).",
158+
)
147159
@click.option("--dry-run", is_flag=True, help="Show what would be written.")
148160
def add_to_catalog(
149161
config_path,
@@ -156,10 +168,11 @@ def add_to_catalog(
156168
tags,
157169
api_url,
158170
include_archived,
171+
fmt,
159172
force,
160173
dry_run,
161174
):
162-
"""Add git repos (owner/name) to the catalog, one file per repo in catalog/.
175+
"""Add git repos (owner/name) to the catalog, one Markdown file per repo in catalog/.
163176
164177
Repos with a dataherb.json/.yml get a pointer entry. Repos without one
165178
are cloned and scanned, and get an inline entry with the inferred files
@@ -192,6 +205,7 @@ def add_to_catalog(
192205
tags=tags,
193206
force=force,
194207
dry_run=dry_run,
208+
fmt=fmt,
195209
)
196210
colors = {"pointer": "green", "inline": "green", "skipped": "yellow"}
197211
for r in results:

‎docs/changelog.md‎

Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -5,6 +5,8 @@
55
Added:
66

77
- `dataherb catalog build | validate | lint | serve`: build a static DataHerb Explorer catalog site from `dataherb.config.yml` (git, S3, HTTP and local sources, S3 discovery, metadata quality scores).
8+
- `dataherb catalog add`: list git repos in the catalog (`--org ORG --match PREFIX` for a whole org). Pointer entries for repos with metadata, inline entries inferred from a clone otherwise.
9+
- Catalog entries can be Markdown files (`catalog/<id>.md`): fields in YAML front matter, the body is shown on the dataset page above the dataset's own documentation. `.yml`/`.json` entries still work.
810
- `dataherb status emit | check`: write and check job status files (`dataherb.status/v1`).
911
- DataHerb v2 metadata (owner, tags, license, classification, update frequency, status job, related datasets) with JSON Schemas in `dataherb/catalog/schemas`.
1012
- Optional extras: `dataherb[s3]` (boto3), `dataherb[infer]` (duckdb).

‎tests/catalog/test_add.py‎

Lines changed: 27 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -7,6 +7,7 @@
77
from dataherb.catalog.add import add_repos, default_id, list_org_repos
88
from dataherb.catalog.config import load_config
99
from dataherb.catalog.resolve import build_catalog
10+
from dataherb.catalog.util import split_front_matter
1011
from dataherb.command import dataherb
1112

1213

@@ -50,13 +51,14 @@ def test_add_pointer_inline_and_skip(project, tmp_path):
5051
"acme/raw": "inline",
5152
"acme/gone": "skipped",
5253
}
53-
meta = yaml.safe_load((project / "catalog" / "meta.yml").read_text())
54+
meta, body = split_front_matter((project / "catalog" / "meta.md").read_text())
5455
assert meta == {"id": "meta", "repo": "acme/meta", "tags": ["new"]}
55-
raw = yaml.safe_load((project / "catalog" / "raw.yml").read_text())
56+
assert body == ""
57+
raw, _ = split_front_matter((project / "catalog" / "raw.md").read_text())
5658
assert raw["inline"] is True and raw["repo"] == "acme/raw"
5759
r0 = raw["datapackage"]["resources"][0]
5860
assert r0["path"] == "data/x.csv" and r0["rows"] == 2
59-
assert not (project / "catalog" / "gone.yml").exists()
61+
assert not (project / "catalog" / "gone.md").exists()
6062

6163
ds = {d["id"]: d for d in build_catalog(load_config(project / "dataherb.config.yml")).datasets}
6264
assert ds["meta"]["name"] == "Meta"
@@ -94,4 +96,25 @@ def test_cli_dry_run(project):
9496
)
9597
assert res.exit_code == 0, res.output
9698
assert "pointer" in res.output and "would add 1 of 1" in res.output
97-
assert not (project / "catalog" / "meta.yml").exists()
99+
assert not (project / "catalog" / "meta.md").exists()
100+
101+
102+
def test_force_keeps_markdown_body(project):
103+
meta_dir = project / "git" / "acme" / "meta" / "HEAD"
104+
meta_dir.mkdir(parents=True)
105+
(meta_dir / "dataherb.json").write_text("{}")
106+
entry = project / "catalog" / "meta.md"
107+
entry.write_text("---\nid: meta\nrepo: acme/meta\n---\n\nHand-written notes.\n")
108+
cfg = load_config(project / "dataherb.config.yml")
109+
[r] = add_repos(cfg, ["acme/meta"], tags=("x",), force=True)
110+
assert r.kind == "pointer"
111+
data, body = split_front_matter(entry.read_text())
112+
assert data["tags"] == ["x"] and body.strip() == "Hand-written notes."
113+
114+
115+
def test_yml_format(project):
116+
meta_dir = project / "git" / "acme" / "meta" / "HEAD"
117+
meta_dir.mkdir(parents=True)
118+
(meta_dir / "dataherb.json").write_text("{}")
119+
add_repos(load_config(project / "dataherb.config.yml"), ["acme/meta"], fmt="yml")
120+
assert yaml.safe_load((project / "catalog" / "meta.yml").read_text())["repo"] == "acme/meta"

‎tests/catalog/test_catalog.py‎

Lines changed: 28 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -42,3 +42,31 @@ def test_discovered_and_catalog_entries_merge(project):
4242
res = build_catalog(load_config(project / "dataherb.config.yml"))
4343
# regions is both discovered (datasets/regions/dataherb.yml) and listed: one record.
4444
assert [d["id"] for d in res.datasets].count("regions") == 1
45+
46+
47+
def test_markdown_entries(project):
48+
cat = project / "catalog"
49+
(cat / "orders.yml").unlink()
50+
(cat / "orders.md").write_text(
51+
"---\nid: orders\nrepo: acme/orders\nref: main\ntags: [sales]\n---\n\n"
52+
"## Caveats\n\nRefunds arrive a day late.\n"
53+
)
54+
(cat / "README.md").write_text("# About this folder\n") # no front matter: not an entry
55+
(cat / "notes.md").write_text("---\nid: notes-ds\ninline: true\nname: Notes\n---\n")
56+
res = build_catalog(load_config(project / "dataherb.config.yml"))
57+
ds = by_id(res)
58+
assert "README" not in ds
59+
assert ds["orders"]["documentation"].startswith("## Caveats")
60+
assert ds["orders"]["tags"] == ["sales"]
61+
assert ds["orders"]["catalog_file"] == "catalog/orders.md"
62+
assert ds["orders"]["resources"][0]["url"].endswith("/data/orders.csv")
63+
assert ds["notes-ds"]["documentation"] == ""
64+
65+
66+
def test_split_front_matter():
67+
from dataherb.catalog.util import split_front_matter
68+
69+
assert split_front_matter("---\na: 1\n---\nbody\n") == ({"a": 1}, "body\n")
70+
assert split_front_matter("---\n---\nx\n") == (None, "x\n") # empty front matter
71+
assert split_front_matter("# title\n---\n") == (None, "# title\n---\n")
72+
assert split_front_matter("---\na: 1\n...\n") == ({"a": 1}, "")

0 commit comments

Comments
 (0)