mirror of
https://github.com/codician-team/growmos.git
synced 2026-08-18 06:47:17 +02:00
Close the loop with no manual step: gold-set and review packets
growmos next now also hands out gold packets (agent writes the eval reference answer, provenance recorded) and periodic review packets (verify one node's edges against sources); eval auto-extends the scorer alias map; headless ingest does the same. Doctor shows who reviewed. Repo URL → codician-team/growmos.
This commit is contained in:
@@ -0,0 +1,11 @@
|
||||
You are building the gold set used to score the extraction pipeline. Read the document carefully and produce the reference answer: the entities that are genuinely central to what this document is about (with their types) and the relations between them as (source, target) pairs. Use the current extraction below only as a starting point — add what it missed, remove what is not actually stated or is merely incidental, and correct types. Precision matters more than recall: do not include an entity you cannot point to in the text.
|
||||
|
||||
<document source="{source_ref}">
|
||||
{text}
|
||||
</document>
|
||||
|
||||
<current_extraction>
|
||||
{extraction}
|
||||
</current_extraction>
|
||||
|
||||
Allowed entity types: {entity_types}
|
||||
@@ -0,0 +1,11 @@
|
||||
This is the periodic comprehension check on the knowledge graph (the playbook's "read a random node each day"). Below is one node with its profile, edges and provenance, followed by excerpts of the sources those edges cite. Verify each edge against the sources: is it stated there, with that direction and meaning? Is the description/profile faithful? Are there obviously missing central relations?
|
||||
|
||||
<node>
|
||||
{card}
|
||||
</node>
|
||||
|
||||
<sources>
|
||||
{excerpts}
|
||||
</sources>
|
||||
|
||||
Report ok=true only if every edge is supported. List concrete issues (edge id + what is wrong). List fixes as growmos commands you would run (e.g. growmos link "A" "pred" "B" --source docs/x.md), or apply them yourself before reporting.
|
||||
@@ -1,8 +1,8 @@
|
||||
{"added": "2026-08-17T11:49:36Z", "extracted_at": "2026-08-17T11:50:35Z", "id": "src_15e0af4a3dd5", "kind": "file", "ref": "docs/evaluation.md", "schema_version": 1, "sha256": "6889be62d360fcc781748d83f147e995f95fc8304155ad291890a9ff9b6e0dba", "stats": {"entities": 5, "relations": 5}, "status": "extracted", "title": "evaluation.md", "updated": "2026-08-17T11:49:36Z"}
|
||||
{"added": "2026-08-17T11:49:36Z", "extracted_at": "2026-08-17T11:50:35Z", "id": "src_15e0af4a3dd5", "kind": "file", "ref": "docs/evaluation.md", "schema_version": 1, "sha256": "1eb6e959059e66ecdbcc35b93ab524635607f157254f763ef030a58b797cd227", "stats": {"entities": 5, "relations": 5}, "status": "pending", "title": "evaluation.md", "updated": "2026-08-17T11:58:38Z"}
|
||||
{"added": "2026-08-17T11:49:36Z", "extracted_at": "2026-08-17T11:50:35Z", "id": "src_3fd317a2cb2b", "kind": "file", "ref": "docs/file-format.md", "schema_version": 1, "sha256": "9f93981a6778d7a991c735ff0b2e8b1d3a51bd1d11cd1bd908dc3b401ce33c04", "stats": {"entities": 7, "relations": 6}, "status": "extracted", "title": "file-format.md", "updated": "2026-08-17T11:49:36Z"}
|
||||
{"added": "2026-08-17T11:49:36Z", "extracted_at": "2026-08-17T11:50:35Z", "id": "src_5dd15b5b20d6", "kind": "file", "ref": "METHODOLOGY.md", "schema_version": 1, "sha256": "7ad5c4305900d6f954fb124e41692573267f3ccb49e363786c4c136fb7e3ba1e", "stats": {"entities": 13, "relations": 13}, "status": "extracted", "title": "METHODOLOGY.md", "updated": "2026-08-17T11:49:36Z"}
|
||||
{"added": "2026-08-17T11:49:36Z", "extracted_at": "2026-08-17T11:50:35Z", "id": "src_644a9dc71c36", "kind": "file", "ref": "docs/headless.md", "schema_version": 1, "sha256": "5db7953e9111f7722f740e80220d3ad099e667e5d848e6e47c02047f722eb983", "stats": {"entities": 5, "relations": 4}, "status": "extracted", "title": "headless.md", "updated": "2026-08-17T11:49:36Z"}
|
||||
{"added": "2026-08-17T11:51:11Z", "id": "src_7a4afa70f300", "kind": "note", "ref": "session:2026-08-17 build", "sha256": "7a4afa70f3001f4f653daa0f7af9b9219e8ea734ce4e7604c707c1437906efe2", "status": "note", "title": "session:2026-08-17 build", "updated": "2026-08-17T11:51:11Z"}
|
||||
{"added": "2026-08-17T11:49:36Z", "extracted_at": "2026-08-17T11:50:35Z", "id": "src_a384388adcb2", "kind": "file", "ref": "docs/agents.md", "schema_version": 1, "sha256": "1db8b7ef7f859f0d26caa4f2b53b52d581510ea8a02ed5b2fb8beed9fe972ad1", "stats": {"entities": 9, "relations": 8}, "status": "extracted", "title": "agents.md", "updated": "2026-08-17T11:49:36Z"}
|
||||
{"added": "2026-08-17T11:49:36Z", "extracted_at": "2026-08-17T11:50:35Z", "id": "src_b33563055168", "kind": "file", "ref": "README.md", "schema_version": 1, "sha256": "def95aadb9d9764769dfac2f285b9edbeb29147acfadfb3c97accaf8f733d153", "stats": {"entities": 14, "relations": 13}, "status": "extracted", "title": "README.md", "updated": "2026-08-17T11:49:36Z"}
|
||||
{"added": "2026-08-17T11:49:36Z", "extracted_at": "2026-08-17T11:50:35Z", "id": "src_b33563055168", "kind": "file", "ref": "README.md", "schema_version": 1, "sha256": "eb3d831021dd979a6f2a888a37d10caefdacf30dd3d3961b73e83c1c6b8a01c7", "stats": {"entities": 14, "relations": 13}, "status": "pending", "title": "README.md", "updated": "2026-08-17T11:58:38Z"}
|
||||
{"added": "2026-08-17T11:49:36Z", "extracted_at": "2026-08-17T11:50:35Z", "id": "src_eca12c0a30e2", "kind": "file", "ref": "CONTRIBUTING.md", "schema_version": 1, "sha256": "08808623d8d898f9185fc82b35383d89e1b1c0e1455423b376365f9e01e979a8", "stats": {"entities": 4, "relations": 3}, "status": "extracted", "title": "CONTRIBUTING.md", "updated": "2026-08-17T11:49:36Z"}
|
||||
|
||||
+1
-1
@@ -18,7 +18,7 @@ Thanks for helping the organism grow. growmos is MIT-licensed and maintained by
|
||||
## Dev loop
|
||||
|
||||
```bash
|
||||
git clone https://github.com/codician/growmos && cd growmos
|
||||
git clone https://github.com/codician-team/growmos && cd growmos
|
||||
pip install -e . # or: uv tool install -e .
|
||||
python -m unittest discover -s tests -v
|
||||
./examples/apollo/run_demo.sh /tmp/apollo # rebuild the playbook corpus end-to-end
|
||||
|
||||
@@ -41,9 +41,12 @@ growmos makes that a **living organism inside your repo**:
|
||||
clusters provisional entities using descriptions ("Edwin Aldrin" → "Buzz Aldrin").
|
||||
- **It answers with citations.** `growmos query` serializes the k-hop subgraph around a
|
||||
question; the answer must cite edge ids; `growmos check` fact-checks claims against edges.
|
||||
- **It measures itself.** Gold sets + `growmos eval` (P/R/F1, raw and resolved), a 10-item
|
||||
`growmos doctor` readiness checklist, health signals (components, density, compression
|
||||
ratio), and `growmos sample` for the daily human comprehension check.
|
||||
- **It measures itself — with no manual step.** `growmos next` also hands out *gold-set* packets
|
||||
(the agent writes the reference answer from the source document) and periodic *review*
|
||||
packets (verify one node's edges against its sources), so `growmos eval` (P/R/F1, raw and
|
||||
resolved), the 10-item `growmos doctor` checklist and the health signals (components, density,
|
||||
compression) all stay green on autopilot. Every gold file records who reviewed it
|
||||
(`agent` / `human`) — humans can overrule at any time, but never have to.
|
||||
- **It is agent-native.** No API key needed: the CLI does the deterministic work, and hands the
|
||||
*judgment* work (extraction, resolution, summarization) to whatever agent you already run as
|
||||
a **task packet** — prompt + JSON shape + the exact `growmos apply …` command. Optional
|
||||
@@ -75,7 +78,7 @@ Manually, the loop is:
|
||||
growmos next # packet: prompt + shape + apply command
|
||||
# … agent produces the JSON …
|
||||
growmos apply extraction out.json --source src_ab12 --chunk 0
|
||||
growmos next # → resolution packet, then profile packets, then "up to date"
|
||||
growmos next # → resolution → profiles → gold set → review → "up to date"
|
||||
growmos query "what depends on the Store and who decided that?"
|
||||
growmos remember "Scheduler" --type COMPONENT --desc "Schedules jobs; depends on Store."
|
||||
growmos link "Scheduler" "depends on" "Store"
|
||||
|
||||
+19
-10
@@ -3,15 +3,21 @@
|
||||
> Change the extraction prompt → rerun the scorer → watch F1 move. A team that ships without
|
||||
> this loop cannot tell whether a prompt change improved or degraded quality.
|
||||
|
||||
## 1. Build a gold set
|
||||
## 1. Build a gold set (automatic by default)
|
||||
|
||||
Pick at least two representative documents. Pre-fill from the current extraction, then
|
||||
**hand-correct** (remove wrong entities, add missed ones — the whole point is human judgment):
|
||||
`growmos next` hands out a **gold packet** for the richest extracted documents until `gold_min`
|
||||
(default 2) files exist: the agent reads the document and writes the reference answer, and the file
|
||||
records `_reviewed_by: agent`. Nothing manual is required. If you want a human pass, pre-fill and
|
||||
edit — the file then records your name:
|
||||
|
||||
```bash
|
||||
growmos gold-template docs/apollo-11.md
|
||||
growmos gold-template docs/apollo-11.md # pre-fill from current extraction
|
||||
$EDITOR .growmos/eval/gold/apollo-11.json
|
||||
# or: growmos apply gold my.json --source docs/apollo-11.md --reviewer human
|
||||
```
|
||||
Caveat worth knowing: an agent-written gold set shares the blind spots of the agent that wrote it;
|
||||
it still catches regressions (prompt drift, resolver over-merging) reliably, which is what the loop
|
||||
is for.
|
||||
|
||||
Gold format:
|
||||
```json
|
||||
@@ -23,11 +29,12 @@ Relations are scored on (source, target) pairs, direction-agnostic, ignoring pre
|
||||
an upper bound on relation recall that catches structural errors (missing/wrong connections),
|
||||
which matter more than wording.
|
||||
|
||||
## 2. Keep the scorer alias map current
|
||||
## 2. Scorer alias map (automatic)
|
||||
|
||||
If the resolver picks a canonical form the gold set doesn't use ("Neil Alden Armstrong" vs
|
||||
"Neil Armstrong"), resolved recall drops — a scoring artifact, not a resolver bug.
|
||||
`growmos eval` lists such canonicals; add them to `.growmos/eval/aliases.json`:
|
||||
"Neil Armstrong"), resolved recall would drop — a scoring artifact, not a resolver bug.
|
||||
`growmos eval` detects these (the raw mention matched a gold name) and **extends
|
||||
`.growmos/eval/aliases.json` itself**, then rescores. You can also add entries by hand:
|
||||
```json
|
||||
{"Neil Alden Armstrong": "Neil Armstrong", "John F. Kennedy Space Center": "Kennedy Space Center"}
|
||||
```
|
||||
@@ -60,7 +67,9 @@ the change alters what the graph means, so rows produced under different prompts
|
||||
`pending_sources`, `provisional`, `stale_profiles`. Extraction rate per document is in each
|
||||
source's `stats`. Wire them into CI (`growmos integrate ci`).
|
||||
|
||||
## 6. The human sample
|
||||
## 6. The periodic review (automatic)
|
||||
|
||||
`growmos sample` prints a random (degree-weighted) node with its profile, edges, provenance and
|
||||
sources. Read it against the sources. `growmos doctor` turns red if nobody has done this in 7 days.
|
||||
Every `review_days` (default 7) `growmos next` hands out a **review packet**: one random
|
||||
(degree-weighted) node with its edges, provenance and source excerpts; the agent verifies each edge,
|
||||
fixes what's wrong, and reports ok/issues (recorded in the journal). `growmos sample` remains
|
||||
available for a human pass; `growmos doctor` shows who reviewed last and when.
|
||||
|
||||
+1
-1
@@ -22,7 +22,7 @@ dependencies = []
|
||||
|
||||
[project.urls]
|
||||
Homepage = "https://codician.com"
|
||||
Repository = "https://github.com/codician/growmos"
|
||||
Repository = "https://github.com/codician-team/growmos"
|
||||
|
||||
[project.scripts]
|
||||
growmos = "growmos.cli:main"
|
||||
|
||||
+104
-4
@@ -226,9 +226,38 @@ def _next_packet(st: Store) -> Optional[Dict[str, Any]]:
|
||||
if deg >= min_deg and st.profile_is_stale(eid):
|
||||
text, meta = summarize_packet(st, eid)
|
||||
return {"kind": "profile", "text": text, "meta": meta}
|
||||
# ground truth: keep at least `gold_min` gold files (agent-reviewed; humans may overwrite)
|
||||
from .prompts import gold_packet, review_packet
|
||||
gold_min = int(st.config.get("gold_min", 2))
|
||||
have = {read_json(f, {}).get("source") for f in st.p("eval", "gold").glob("*.json")}
|
||||
if len(have) < gold_min:
|
||||
cands = [r for r in st.sources.values() if r.get("status") == "extracted" and r.get("kind", "file") == "file"
|
||||
and r["ref"] not in have]
|
||||
cands.sort(key=lambda r: -(r.get("stats") or {}).get("entities", 0))
|
||||
if cands:
|
||||
text, meta = gold_packet(st, cands[0]["id"])
|
||||
return {"kind": "gold", "text": text, "meta": meta}
|
||||
# comprehension check: one node every `review_days`
|
||||
if st.entities and _days_since(st.state.get("last_sample")) is None or \
|
||||
(st.entities and _days_since(st.state.get("last_sample")) >= int(st.config.get("review_days", 7))):
|
||||
eid = g.sample(random.Random(len(st.relations)))
|
||||
if eid:
|
||||
text, meta = review_packet(st, eid)
|
||||
return {"kind": "review", "text": text, "meta": meta}
|
||||
return None
|
||||
|
||||
|
||||
def _days_since(ts: Optional[str]) -> Optional[int]:
|
||||
if not ts:
|
||||
return None
|
||||
from datetime import datetime, timezone
|
||||
try:
|
||||
dt = datetime.fromisoformat(ts.replace("Z", "+00:00"))
|
||||
except ValueError:
|
||||
return None
|
||||
return (datetime.now(timezone.utc) - dt).days
|
||||
|
||||
|
||||
def _remember_packet(st: Store, etype: str, names: List[str]) -> None:
|
||||
"""Cache the names handed out in a resolution packet so `apply` can enforce
|
||||
'every input name appears in exactly one cluster' against the real input."""
|
||||
@@ -317,8 +346,51 @@ def cmd_apply(args: argparse.Namespace) -> int:
|
||||
f"{prof['time_range']['start']}→{prof['time_range']['end']})")
|
||||
for p in problems:
|
||||
print(f" ! {p}")
|
||||
elif kind == "gold":
|
||||
source = args.source or (payload.get("source") if isinstance(payload, dict) else None)
|
||||
if not source:
|
||||
raise StoreError("--source <src_id> is required")
|
||||
if source not in st.sources:
|
||||
match = [s for s, r in st.sources.items() if r.get("ref") == source]
|
||||
if not match:
|
||||
raise StoreError(f"unknown source '{source}'")
|
||||
source = match[0]
|
||||
ref = st.sources[source]["ref"]
|
||||
ents = [{"name": e["name"], "type": str(e.get("type", "")).upper()} for e in payload.get("entities", [])
|
||||
if isinstance(e, dict) and isinstance(e.get("name"), str) and e["name"].strip()]
|
||||
pairs = [{"source": r["source"], "target": r["target"]} for r in payload.get("relations", [])
|
||||
if isinstance(r, dict) and isinstance(r.get("source"), str) and isinstance(r.get("target"), str)]
|
||||
out = st.p("eval", "gold", Path(ref).stem + ".json")
|
||||
write_json(out, {"source": ref, "entities": ents, "relations": pairs,
|
||||
"_reviewed_by": args.reviewer or "agent", "_ts": now_iso()})
|
||||
st.record_run("gold", {"source": source, "reviewer": args.reviewer or "agent"})
|
||||
st.save()
|
||||
print(f"✓ gold set written for {ref}: {len(ents)} entities, {len(pairs)} relation pairs "
|
||||
f"(reviewed by {args.reviewer or 'agent'}; humans may edit {out.relative_to(st.root)})")
|
||||
elif kind == "review":
|
||||
eid = args.entity or (payload.get("entity") if isinstance(payload, dict) else None)
|
||||
if not eid or eid not in st.entities:
|
||||
r = st.resolve_name(eid or "")
|
||||
if not r:
|
||||
raise StoreError(f"unknown entity '{eid}'")
|
||||
eid = r
|
||||
ok = bool(payload.get("ok"))
|
||||
issues = [i for i in payload.get("issues", []) if isinstance(i, str)]
|
||||
fixes = [f for f in payload.get("fixes", []) if isinstance(f, str)]
|
||||
st.state["last_sample"] = now_iso()
|
||||
st.state["last_sample_entity"] = eid
|
||||
st.state["last_sample_ok"] = ok
|
||||
st.record_run("review", {"entity": eid, "ok": ok, "issues": len(issues), "reviewer": args.reviewer or "agent"})
|
||||
note = f"Review of {st.entities[eid]['name']} ({eid}): {'ok' if ok else 'issues found'}."
|
||||
if issues:
|
||||
note += "\n" + "\n".join(f"- {i}" for i in issues)
|
||||
if fixes:
|
||||
note += "\nFixes:\n" + "\n".join(f"- {f}" for f in fixes)
|
||||
st.journal(note, author=args.reviewer or "agent-review")
|
||||
st.save()
|
||||
print(f"✓ review recorded for {st.entities[eid]['name']}: {'ok' if ok else f'{len(issues)} issue(s)'}")
|
||||
else:
|
||||
raise StoreError("kind must be extraction | resolution | profile")
|
||||
raise StoreError("kind must be extraction | resolution | profile | gold | review")
|
||||
return 0
|
||||
|
||||
|
||||
@@ -679,6 +751,33 @@ def cmd_ingest(args: argparse.Namespace) -> int:
|
||||
st.set_profile(eid, prof)
|
||||
st.save()
|
||||
print(f"profiled {st.entities[eid]['name']}")
|
||||
# gold + review, so the loop is complete without a human
|
||||
from .prompts import gold_packet, review_packet
|
||||
gold_min = int(st.config.get("gold_min", 2))
|
||||
have = {read_json(f, {}).get("source") for f in st.p("eval", "gold").glob("*.json")}
|
||||
cands = [r for r in st.sources.values() if r.get("status") == "extracted" and r.get("kind", "file") == "file"
|
||||
and r["ref"] not in have]
|
||||
cands.sort(key=lambda r: -(r.get("stats") or {}).get("entities", 0))
|
||||
for rec in cands[: max(0, gold_min - len(have))]:
|
||||
pkt, meta = gold_packet(st, rec["id"])
|
||||
data = call_structured(pkt.split("--- prompt ---\n", 1)[-1], S.GOLD_SCHEMA, stage="reason", config=st.config)
|
||||
write_json(st.p("eval", "gold", Path(rec["ref"]).stem + ".json"),
|
||||
{"source": rec["ref"], "entities": data.get("entities", []), "relations": data.get("relations", []),
|
||||
"_reviewed_by": "agent (headless)", "_ts": now_iso()})
|
||||
print(f"gold written for {rec['ref']}")
|
||||
d = _days_since(st.state.get("last_sample"))
|
||||
if st.entities and (d is None or d >= int(st.config.get("review_days", 7))):
|
||||
eid = g.sample(random.Random(len(st.relations)))
|
||||
pkt, meta = review_packet(st, eid)
|
||||
data = call_structured(pkt.split("--- prompt ---\n", 1)[-1], S.REVIEW_SCHEMA, stage="reason", config=st.config)
|
||||
st.state["last_sample"] = now_iso()
|
||||
st.state["last_sample_entity"] = eid
|
||||
st.state["last_sample_ok"] = bool(data.get("ok"))
|
||||
st.journal(f"Headless review of {st.entities[eid]['name']}: {'ok' if data.get('ok') else 'issues'}\n"
|
||||
+ "\n".join(f"- {i}" for i in data.get("issues", [])), author="agent-review")
|
||||
print(f"reviewed {st.entities[eid]['name']}: {'ok' if data.get('ok') else 'issues found'}")
|
||||
from .evaluate import Evaluator
|
||||
Evaluator(st).evaluate()
|
||||
except ProviderError as e:
|
||||
st.save()
|
||||
print(f"provider error: {e}", file=sys.stderr)
|
||||
@@ -702,7 +801,7 @@ def build_parser() -> argparse.ArgumentParser:
|
||||
p = argparse.ArgumentParser(
|
||||
prog="growmos",
|
||||
description="growmos — a living knowledge graph that grows with your repo (Codician, MIT).",
|
||||
epilog="Agent loop: growmos context → work → growmos remember/link/journal → growmos next → apply. Docs: https://github.com/codician/growmos",
|
||||
epilog="Agent loop: growmos context → work → growmos remember/link/journal → growmos next → apply. Docs: https://github.com/codician-team/growmos",
|
||||
)
|
||||
p.add_argument("--root", help="repository root (default: auto-detect via .growmos/ or .git/)")
|
||||
p.add_argument("--version", action="version", version=f"growmos {__version__}")
|
||||
@@ -744,13 +843,14 @@ def build_parser() -> argparse.ArgumentParser:
|
||||
sp.set_defaults(fn=cmd_next)
|
||||
|
||||
sp = sub.add_parser("apply", help="ingest a completed packet's JSON")
|
||||
sp.add_argument("kind", choices=["extraction", "resolution", "profile"])
|
||||
sp.add_argument("kind", choices=["extraction", "resolution", "profile", "gold", "review"])
|
||||
sp.add_argument("file", help="JSON file or - for stdin")
|
||||
sp.add_argument("--source", help="source id/ref (extraction)")
|
||||
sp.add_argument("--chunk", type=int, help="chunk index (extraction)")
|
||||
sp.add_argument("--partial", action="store_true", help="more chunks follow; keep source pending")
|
||||
sp.add_argument("--type", help="entity type (resolution)")
|
||||
sp.add_argument("--entity", help="entity id or name (profile)")
|
||||
sp.add_argument("--entity", help="entity id or name (profile/review)")
|
||||
sp.add_argument("--reviewer", help="who reviewed (gold/review): agent (default) or human")
|
||||
sp.set_defaults(fn=cmd_apply)
|
||||
|
||||
sp = sub.add_parser("extract", help="print the extraction packet for a source")
|
||||
|
||||
+34
-6
@@ -90,6 +90,29 @@ class Evaluator:
|
||||
"missed_entities": sorted(gold_ents - raw_names)[:10],
|
||||
"extra_entities": sorted(raw_names - gold_ents)[:10],
|
||||
})
|
||||
# Auto-extend the scorer alias map: a canonical form whose raw mention matched a gold
|
||||
# name is a scoring artifact, not a resolver bug — record canonical -> gold name.
|
||||
auto_added = {}
|
||||
if unrecognized:
|
||||
amap = read_json(st.p("eval", "aliases.json"), {}) or {}
|
||||
gold_names = {}
|
||||
for gf in self.gold_files():
|
||||
for e in (read_json(gf, {}) or {}).get("entities", []):
|
||||
if e.get("name"):
|
||||
gold_names[norm_name(e["name"])] = e["name"]
|
||||
for cname in unrecognized:
|
||||
eid = st.resolve_name(cname)
|
||||
if not eid:
|
||||
continue
|
||||
for (alias, _t), aeid in st.aliases.items():
|
||||
if aeid == eid and alias in gold_names and cname not in amap:
|
||||
amap[cname] = gold_names[alias]
|
||||
auto_added[cname] = gold_names[alias]
|
||||
break
|
||||
if auto_added:
|
||||
write_json(st.p("eval", "aliases.json"), amap)
|
||||
self.alias_map = {norm_name(k): v for k, v in amap.items()}
|
||||
return self.evaluate() # rescore once with the extended map
|
||||
report = {"ts": now_iso(), "docs": rows, "unrecognized_canonicals": sorted(unrecognized),
|
||||
"schema_version": st.schema_version}
|
||||
write_json(st.p("eval", "last_report.json"), report)
|
||||
@@ -137,10 +160,12 @@ def doctor(store: Store) -> List[Dict[str, str]]:
|
||||
checks.append({"item": name, "status": "ok" if ok else ("warn" if ok is None else "fail"),
|
||||
"detail": detail, "fix": fix})
|
||||
|
||||
gold_n = len(list(st.p("eval", "gold").glob("*.json")))
|
||||
gold_files = list(st.p("eval", "gold").glob("*.json"))
|
||||
gold_n = len(gold_files)
|
||||
reviewers = sorted({str((read_json(f, {}) or {}).get("_reviewed_by", "human")) for f in gold_files})
|
||||
add("Gold set", gold_n >= 2 if gold_n else False,
|
||||
f"{gold_n} gold file(s)" if gold_n else "no gold files → prompt changes are blind",
|
||||
"add ≥2 hand-labelled files in .growmos/eval/gold/ (growmos gold-template <source>)")
|
||||
(f"{gold_n} gold file(s), reviewed by: {', '.join(reviewers)}") if gold_n else "no gold files → prompt changes are blind",
|
||||
"growmos next hands out gold packets until gold_min is met (or hand-write .growmos/eval/gold/*.json)")
|
||||
amap = read_json(st.p("eval", "aliases.json"), None)
|
||||
last = read_json(st.p("eval", "last_report.json"), {}) or {}
|
||||
unrec = last.get("unrecognized_canonicals") or []
|
||||
@@ -175,9 +200,12 @@ def doctor(store: Store) -> List[Dict[str, str]]:
|
||||
days = (datetime.now(timezone.utc) - dt).days
|
||||
except ValueError:
|
||||
days = None
|
||||
add("Human sample", (days is not None and days <= 7) if ls else False,
|
||||
f"last sample {days} day(s) ago" if days is not None else "no one has sampled the graph yet",
|
||||
"growmos sample — read one node's profile and check its edges against the sources")
|
||||
who = "human" if st.state.get("last_sample_entity") is None else ("agent" if any(
|
||||
r.get("kind") == "review" for r in st.state.get("runs", [])[-20:]) else "human")
|
||||
add("Human sample", (days is not None and days <= int(st.config.get("review_days", 7))) if ls else False,
|
||||
f"last review {days} day(s) ago ({who}; ok={st.state.get('last_sample_ok', '?')})" if days is not None
|
||||
else "no one has reviewed a node yet",
|
||||
"growmos next hands out a review packet every review_days (or growmos sample for a human pass)")
|
||||
return checks
|
||||
|
||||
|
||||
|
||||
+54
-1
@@ -24,7 +24,7 @@ from .store import Store
|
||||
from .util import approx_tokens, truncate
|
||||
|
||||
TEMPLATE_DIR = Path(__file__).parent / "templates" / "prompts"
|
||||
PROMPT_NAMES = ["extract", "resolve", "summarize", "query", "check"]
|
||||
PROMPT_NAMES = ["extract", "resolve", "summarize", "query", "check", "gold", "review"]
|
||||
|
||||
|
||||
def install_default_prompts(dest: Path, force: bool = False) -> List[str]:
|
||||
@@ -120,6 +120,10 @@ def _shape(kind: str, entity_types: Optional[List[str]] = None) -> str:
|
||||
if kind == "profile":
|
||||
return ('{"summary": "2-3 paragraphs", "key_facts": ["3-5 atomic facts traceable to sources"],\n'
|
||||
' "time_range": {"start": "YYYY|unknown", "end": "YYYY|ongoing|unknown"}}')
|
||||
if kind == "gold":
|
||||
return '{"entities": [{"name": "…", "type": "…"}], "relations": [{"source": "<name>", "target": "<name>"}]}'
|
||||
if kind == "review":
|
||||
return '{"ok": true, "issues": ["<edge id>: what is wrong"], "fixes": ["growmos link … / growmos remember … you ran or recommend"]}'
|
||||
return "{}"
|
||||
|
||||
|
||||
@@ -305,3 +309,52 @@ def check_packet(store: Store, content: str, hops: int = 2) -> Tuple[str, Dict[s
|
||||
prompt = render(load_prompt(store, "check"), graph=triples or "(empty)", content=content)
|
||||
meta = {"seeds": seeds, "nodes": len(nodes), "edges": total, "tokens": approx_tokens(prompt)}
|
||||
return prompt, meta
|
||||
|
||||
|
||||
def gold_packet(store: Store, source_id: str) -> Tuple[str, Dict[str, Any]]:
|
||||
"""Ask the agent to produce (or correct) the gold answer for a source — the eval loop's ground truth."""
|
||||
rec = store.sources[source_id]
|
||||
text = store.source_text(source_id)
|
||||
ents = [{"name": m["name"], "type": m["type"]} for m in store.mentions() if m.get("source") == source_id]
|
||||
pairs = [{"source": store.entities[r["source"]]["name"], "target": store.entities[r["target"]]["name"]}
|
||||
for r in store.relations.values() if source_id in r.get("sources", [])
|
||||
and r["source"] in store.entities and r["target"] in store.entities]
|
||||
prompt = render(load_prompt(store, "gold"), source_ref=rec["ref"], text=truncate(text, 20000),
|
||||
extraction=json.dumps({"entities": ents, "relations": pairs}, ensure_ascii=False, indent=1),
|
||||
entity_types=", ".join(store.entity_types))
|
||||
stem = Path(rec["ref"]).stem or source_id
|
||||
out_file = f".growmos/cache/gold_{source_id}.json"
|
||||
apply_cmd = f"growmos apply gold {out_file} --source {source_id}"
|
||||
header = _packet_header(f"gold set · {rec['ref']}", apply_cmd, S.GOLD_SCHEMA, out_file, kind="gold")
|
||||
meta = {"source": source_id, "stem": stem, "apply": apply_cmd, "out_file": out_file, "tokens": approx_tokens(prompt)}
|
||||
return header + prompt, meta
|
||||
|
||||
|
||||
def review_packet(store: Store, eid: str, max_excerpt_chars: int = 1500) -> Tuple[str, Dict[str, Any]]:
|
||||
"""The comprehension check: one node, its edges, and the sources those edges cite."""
|
||||
g = Graph(store)
|
||||
card = g.entity_card(eid)
|
||||
ent = store.entities[eid]
|
||||
sids: List[str] = []
|
||||
for rid in g.out.get(eid, []) + g.inc.get(eid, []):
|
||||
for sid in store.relations[rid].get("sources", []):
|
||||
if sid not in sids:
|
||||
sids.append(sid)
|
||||
for sid in ent.get("sources", []):
|
||||
if sid not in sids:
|
||||
sids.append(sid)
|
||||
excerpts = []
|
||||
for sid in sids[:8]:
|
||||
rec = store.sources.get(sid, {})
|
||||
try:
|
||||
text = store.source_text(sid)
|
||||
except Exception:
|
||||
text = ""
|
||||
body = _excerpt_around(text, ent["name"], max_excerpt_chars) if text else "(note/session source — no text)"
|
||||
excerpts.append(f"[{rec.get('ref', sid)}]\n{body}")
|
||||
prompt = render(load_prompt(store, "review"), card=card, excerpts="\n\n".join(excerpts) or "(none)")
|
||||
out_file = f".growmos/cache/review_{eid.replace('/', '__')}.json"
|
||||
apply_cmd = f"growmos apply review {out_file} --entity \"{eid}\""
|
||||
header = _packet_header(f"review · {ent['name']}", apply_cmd, S.REVIEW_SCHEMA, out_file, kind="review")
|
||||
meta = {"entity": eid, "apply": apply_cmd, "out_file": out_file, "tokens": approx_tokens(prompt)}
|
||||
return header + prompt, meta
|
||||
|
||||
@@ -129,6 +129,31 @@ PROFILE_SCHEMA: Dict[str, Any] = {
|
||||
"additionalProperties": False,
|
||||
}
|
||||
|
||||
GOLD_SCHEMA: Dict[str, Any] = {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"entities": {"type": "array", "items": {"type": "object", "properties": {
|
||||
"name": {"type": "string"}, "type": {"type": "string"}},
|
||||
"required": ["name", "type"], "additionalProperties": False}},
|
||||
"relations": {"type": "array", "items": {"type": "object", "properties": {
|
||||
"source": {"type": "string"}, "target": {"type": "string"}},
|
||||
"required": ["source", "target"], "additionalProperties": False}},
|
||||
},
|
||||
"required": ["entities", "relations"],
|
||||
"additionalProperties": False,
|
||||
}
|
||||
|
||||
REVIEW_SCHEMA: Dict[str, Any] = {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"ok": {"type": "boolean"},
|
||||
"issues": {"type": "array", "items": {"type": "string"}},
|
||||
"fixes": {"type": "array", "items": {"type": "string"}},
|
||||
},
|
||||
"required": ["ok", "issues", "fixes"],
|
||||
"additionalProperties": False,
|
||||
}
|
||||
|
||||
ANSWER_SCHEMA: Dict[str, Any] = {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
|
||||
@@ -90,6 +90,8 @@ class Store:
|
||||
"chunk_chars": 6000,
|
||||
"profile_min_degree": 3,
|
||||
"resolve_batch_size": 80,
|
||||
"gold_min": 2,
|
||||
"review_days": 7,
|
||||
"provider": {"name": "", "extract_model": "", "reason_model": ""},
|
||||
}
|
||||
st.schema = {
|
||||
|
||||
@@ -42,6 +42,13 @@ no invented facts.
|
||||
**Query** — answer from the graph only, cite edges, flag gaps. On a private corpus only the
|
||||
grounded answer works at all.
|
||||
|
||||
**Gold packet** — you are writing the reference answer for the eval loop: read the document, keep
|
||||
only genuinely central entities and (source,target) pairs, add what extraction missed, remove what
|
||||
is not stated. Precision first. Humans may later overwrite the file; say so is fine.
|
||||
|
||||
**Review packet** — the periodic comprehension check: verify every edge of one node against the
|
||||
cited sources; fix what is wrong with `growmos link`/`remember`/`journal`, then report ok/issues.
|
||||
|
||||
## Applying results
|
||||
|
||||
Write the JSON to the `out_file` named in the packet header (or pipe on stdin) and run the
|
||||
@@ -51,6 +58,8 @@ printed command, e.g.
|
||||
growmos apply extraction .growmos/cache/extract_src_xxx_0.json --source src_xxx --chunk 0
|
||||
growmos apply resolution .growmos/cache/resolve_person_0.json --type PERSON
|
||||
growmos apply profile .growmos/cache/profile_x.json --entity "component/graph-store"
|
||||
growmos apply gold .growmos/cache/gold_src_xxx.json --source src_xxx
|
||||
growmos apply review .growmos/cache/review_x.json --entity "component/graph-store"
|
||||
```
|
||||
|
||||
The CLI validates against the schema, drops dangling relations, gives unmatched names a
|
||||
|
||||
@@ -14,8 +14,9 @@ memory you read at the start of work and write to as you develop. Zero-config co
|
||||
- `growmos link "<A>" "<predicate>" "<B>"` (short verb phrase predicates: "depends on", "replaces")
|
||||
- `growmos journal "<what changed and why>"`
|
||||
4. **Feed the organism** — run `growmos next`. It hands you a *task packet* (extraction / resolution /
|
||||
profile) with the exact prompt, the JSON schema, and the `growmos apply …` command. Do the judgment
|
||||
work yourself, write the JSON, apply it. Repeat until `growmos next` says the graph is up to date.
|
||||
profile / gold set / review) with the exact prompt, the JSON shape, and the `growmos apply …` command.
|
||||
Do the judgment work yourself, write the JSON, apply it. Repeat until `growmos next` says the graph is
|
||||
up to date — that loop covers everything, including the evaluation gold set and the periodic node review.
|
||||
Never invent facts not in the source; every relation must connect two extracted entities.
|
||||
5. **Before claiming facts about the repo in a summary/report** — `growmos check "<claim text>"` grounds
|
||||
your claims against edges with provenance (evaluator–optimizer loop).
|
||||
@@ -23,5 +24,5 @@ memory you read at the start of work and write to as you develop. Zero-config co
|
||||
|
||||
Store files are plain JSONL under `.growmos/` — commit them with your code. Do not hand-edit
|
||||
`entities.jsonl`/`relations.jsonl` (use the CLI); prompts in `.growmos/prompts/` are yours to tune.
|
||||
More: `growmos --help`, docs at https://github.com/codician/growmos.
|
||||
More: `growmos --help`, docs at https://github.com/codician-team/growmos.
|
||||
<!-- growmos:end -->
|
||||
|
||||
@@ -0,0 +1,11 @@
|
||||
You are building the gold set used to score the extraction pipeline. Read the document carefully and produce the reference answer: the entities that are genuinely central to what this document is about (with their types) and the relations between them as (source, target) pairs. Use the current extraction below only as a starting point — add what it missed, remove what is not actually stated or is merely incidental, and correct types. Precision matters more than recall: do not include an entity you cannot point to in the text.
|
||||
|
||||
<document source="{source_ref}">
|
||||
{text}
|
||||
</document>
|
||||
|
||||
<current_extraction>
|
||||
{extraction}
|
||||
</current_extraction>
|
||||
|
||||
Allowed entity types: {entity_types}
|
||||
@@ -0,0 +1,11 @@
|
||||
This is the periodic comprehension check on the knowledge graph (the playbook's "read a random node each day"). Below is one node with its profile, edges and provenance, followed by excerpts of the sources those edges cite. Verify each edge against the sources: is it stated there, with that direction and meaning? Is the description/profile faithful? Are there obviously missing central relations?
|
||||
|
||||
<node>
|
||||
{card}
|
||||
</node>
|
||||
|
||||
<sources>
|
||||
{excerpts}
|
||||
</sources>
|
||||
|
||||
Report ok=true only if every edge is supported. List concrete issues (edge id + what is wrong). List fixes as growmos commands you would run (e.g. growmos link "A" "pred" "B" --source docs/x.md), or apply them yourself before reporting.
|
||||
@@ -138,6 +138,48 @@ class ApolloFixture(unittest.TestCase):
|
||||
self.assertIn("growmos apply extraction", pkt)
|
||||
self.assertEqual(meta["chunks"], 1)
|
||||
|
||||
def test_next_hands_out_gold_then_review_then_done(self):
|
||||
import copy
|
||||
tmp2 = Path(tempfile.mkdtemp(prefix="growmos-next-"))
|
||||
shutil.copytree(self.tmp, tmp2 / "r")
|
||||
root = tmp2 / "r"
|
||||
st = Store(root).load()
|
||||
for f in st.p("eval", "gold").glob("*.json"):
|
||||
f.unlink()
|
||||
for eid, deg in Graph(st).hubs(50):
|
||||
if deg >= 3:
|
||||
st.set_profile(eid, {"summary": "s", "key_facts": ["k"], "time_range": {"start": "u", "end": "u"}})
|
||||
st.state["last_sample"] = None
|
||||
st.save()
|
||||
out = run_cli("--root", str(root), "next")
|
||||
self.assertIn("gold set", out)
|
||||
for _ in range(2):
|
||||
meta = json.loads(run_cli("--root", str(root), "next", "--json"))["meta"]
|
||||
run_cli("--root", str(root), "apply", "gold", "-", "--source", meta["source"],
|
||||
stdin=json.dumps({"entities": [{"name": "Apollo 11", "type": "EVENT"}], "relations": []}))
|
||||
out = run_cli("--root", str(root), "next")
|
||||
self.assertIn("review", out)
|
||||
meta = json.loads(run_cli("--root", str(root), "next", "--json"))["meta"]
|
||||
run_cli("--root", str(root), "apply", "review", "-", "--entity", meta["entity"],
|
||||
stdin=json.dumps({"ok": True, "issues": [], "fixes": []}))
|
||||
out = run_cli("--root", str(root), "next")
|
||||
self.assertIn("up to date", out)
|
||||
checks = {c["item"]: c["status"] for c in doctor(Store(root).load())}
|
||||
self.assertEqual(checks["Gold set"], "ok")
|
||||
self.assertEqual(checks["Human sample"], "ok")
|
||||
shutil.rmtree(tmp2, ignore_errors=True)
|
||||
|
||||
def test_eval_auto_extends_alias_map(self):
|
||||
tmp2 = Path(tempfile.mkdtemp(prefix="growmos-alias-"))
|
||||
shutil.copytree(self.tmp, tmp2 / "r")
|
||||
st = Store(tmp2 / "r").load()
|
||||
(st.p("eval", "aliases.json")).write_text("{}", encoding="utf-8")
|
||||
rep = Evaluator(st).evaluate()
|
||||
amap = read_json(st.p("eval", "aliases.json"))
|
||||
self.assertEqual(amap.get("Neil Alden Armstrong"), "Neil Armstrong")
|
||||
self.assertEqual(rep["unrecognized_canonicals"], [])
|
||||
shutil.rmtree(tmp2, ignore_errors=True)
|
||||
|
||||
def test_export_formats(self):
|
||||
from growmos.export import EXPORTERS
|
||||
for name, fn in EXPORTERS.items():
|
||||
|
||||
Reference in New Issue
Block a user