Close the loop with no manual step: gold-set and review packets

growmos next now also hands out gold packets (agent writes the eval reference
answer, provenance recorded) and periodic review packets (verify one node's
edges against sources); eval auto-extends the scorer alias map; headless ingest
does the same. Doctor shows who reviewed. Repo URL → codician-team/growmos.
This commit is contained in:
codician-dev
2026-08-17 14:58:47 +03:00
parent 988358ad1a
commit 39f38fb4d1
17 changed files with 348 additions and 32 deletions
+11
View File
@@ -0,0 +1,11 @@
You are building the gold set used to score the extraction pipeline. Read the document carefully and produce the reference answer: the entities that are genuinely central to what this document is about (with their types) and the relations between them as (source, target) pairs. Use the current extraction below only as a starting point — add what it missed, remove what is not actually stated or is merely incidental, and correct types. Precision matters more than recall: do not include an entity you cannot point to in the text.
<document source="{source_ref}">
{text}
</document>
<current_extraction>
{extraction}
</current_extraction>
Allowed entity types: {entity_types}
+11
View File
@@ -0,0 +1,11 @@
This is the periodic comprehension check on the knowledge graph (the playbook's "read a random node each day"). Below is one node with its profile, edges and provenance, followed by excerpts of the sources those edges cite. Verify each edge against the sources: is it stated there, with that direction and meaning? Is the description/profile faithful? Are there obviously missing central relations?
<node>
{card}
</node>
<sources>
{excerpts}
</sources>
Report ok=true only if every edge is supported. List concrete issues (edge id + what is wrong). List fixes as growmos commands you would run (e.g. growmos link "A" "pred" "B" --source docs/x.md), or apply them yourself before reporting.
+2 -2
View File
@@ -1,8 +1,8 @@
{"added": "2026-08-17T11:49:36Z", "extracted_at": "2026-08-17T11:50:35Z", "id": "src_15e0af4a3dd5", "kind": "file", "ref": "docs/evaluation.md", "schema_version": 1, "sha256": "6889be62d360fcc781748d83f147e995f95fc8304155ad291890a9ff9b6e0dba", "stats": {"entities": 5, "relations": 5}, "status": "extracted", "title": "evaluation.md", "updated": "2026-08-17T11:49:36Z"}
{"added": "2026-08-17T11:49:36Z", "extracted_at": "2026-08-17T11:50:35Z", "id": "src_15e0af4a3dd5", "kind": "file", "ref": "docs/evaluation.md", "schema_version": 1, "sha256": "1eb6e959059e66ecdbcc35b93ab524635607f157254f763ef030a58b797cd227", "stats": {"entities": 5, "relations": 5}, "status": "pending", "title": "evaluation.md", "updated": "2026-08-17T11:58:38Z"}
{"added": "2026-08-17T11:49:36Z", "extracted_at": "2026-08-17T11:50:35Z", "id": "src_3fd317a2cb2b", "kind": "file", "ref": "docs/file-format.md", "schema_version": 1, "sha256": "9f93981a6778d7a991c735ff0b2e8b1d3a51bd1d11cd1bd908dc3b401ce33c04", "stats": {"entities": 7, "relations": 6}, "status": "extracted", "title": "file-format.md", "updated": "2026-08-17T11:49:36Z"}
{"added": "2026-08-17T11:49:36Z", "extracted_at": "2026-08-17T11:50:35Z", "id": "src_5dd15b5b20d6", "kind": "file", "ref": "METHODOLOGY.md", "schema_version": 1, "sha256": "7ad5c4305900d6f954fb124e41692573267f3ccb49e363786c4c136fb7e3ba1e", "stats": {"entities": 13, "relations": 13}, "status": "extracted", "title": "METHODOLOGY.md", "updated": "2026-08-17T11:49:36Z"}
{"added": "2026-08-17T11:49:36Z", "extracted_at": "2026-08-17T11:50:35Z", "id": "src_644a9dc71c36", "kind": "file", "ref": "docs/headless.md", "schema_version": 1, "sha256": "5db7953e9111f7722f740e80220d3ad099e667e5d848e6e47c02047f722eb983", "stats": {"entities": 5, "relations": 4}, "status": "extracted", "title": "headless.md", "updated": "2026-08-17T11:49:36Z"}
{"added": "2026-08-17T11:51:11Z", "id": "src_7a4afa70f300", "kind": "note", "ref": "session:2026-08-17 build", "sha256": "7a4afa70f3001f4f653daa0f7af9b9219e8ea734ce4e7604c707c1437906efe2", "status": "note", "title": "session:2026-08-17 build", "updated": "2026-08-17T11:51:11Z"}
{"added": "2026-08-17T11:49:36Z", "extracted_at": "2026-08-17T11:50:35Z", "id": "src_a384388adcb2", "kind": "file", "ref": "docs/agents.md", "schema_version": 1, "sha256": "1db8b7ef7f859f0d26caa4f2b53b52d581510ea8a02ed5b2fb8beed9fe972ad1", "stats": {"entities": 9, "relations": 8}, "status": "extracted", "title": "agents.md", "updated": "2026-08-17T11:49:36Z"}
{"added": "2026-08-17T11:49:36Z", "extracted_at": "2026-08-17T11:50:35Z", "id": "src_b33563055168", "kind": "file", "ref": "README.md", "schema_version": 1, "sha256": "def95aadb9d9764769dfac2f285b9edbeb29147acfadfb3c97accaf8f733d153", "stats": {"entities": 14, "relations": 13}, "status": "extracted", "title": "README.md", "updated": "2026-08-17T11:49:36Z"}
{"added": "2026-08-17T11:49:36Z", "extracted_at": "2026-08-17T11:50:35Z", "id": "src_b33563055168", "kind": "file", "ref": "README.md", "schema_version": 1, "sha256": "eb3d831021dd979a6f2a888a37d10caefdacf30dd3d3961b73e83c1c6b8a01c7", "stats": {"entities": 14, "relations": 13}, "status": "pending", "title": "README.md", "updated": "2026-08-17T11:58:38Z"}
{"added": "2026-08-17T11:49:36Z", "extracted_at": "2026-08-17T11:50:35Z", "id": "src_eca12c0a30e2", "kind": "file", "ref": "CONTRIBUTING.md", "schema_version": 1, "sha256": "08808623d8d898f9185fc82b35383d89e1b1c0e1455423b376365f9e01e979a8", "stats": {"entities": 4, "relations": 3}, "status": "extracted", "title": "CONTRIBUTING.md", "updated": "2026-08-17T11:49:36Z"}
+1 -1
View File
@@ -18,7 +18,7 @@ Thanks for helping the organism grow. growmos is MIT-licensed and maintained by
## Dev loop
```bash
git clone https://github.com/codician/growmos && cd growmos
git clone https://github.com/codician-team/growmos && cd growmos
pip install -e . # or: uv tool install -e .
python -m unittest discover -s tests -v
./examples/apollo/run_demo.sh /tmp/apollo # rebuild the playbook corpus end-to-end
+7 -4
View File
@@ -41,9 +41,12 @@ growmos makes that a **living organism inside your repo**:
clusters provisional entities using descriptions ("Edwin Aldrin" → "Buzz Aldrin").
- **It answers with citations.** `growmos query` serializes the k-hop subgraph around a
question; the answer must cite edge ids; `growmos check` fact-checks claims against edges.
- **It measures itself.** Gold sets + `growmos eval` (P/R/F1, raw and resolved), a 10-item
`growmos doctor` readiness checklist, health signals (components, density, compression
ratio), and `growmos sample` for the daily human comprehension check.
- **It measures itself — with no manual step.** `growmos next` also hands out *gold-set* packets
(the agent writes the reference answer from the source document) and periodic *review*
packets (verify one node's edges against its sources), so `growmos eval` (P/R/F1, raw and
resolved), the 10-item `growmos doctor` checklist and the health signals (components, density,
compression) all stay green on autopilot. Every gold file records who reviewed it
(`agent` / `human`) — humans can overrule at any time, but never have to.
- **It is agent-native.** No API key needed: the CLI does the deterministic work, and hands the
*judgment* work (extraction, resolution, summarization) to whatever agent you already run as
a **task packet** — prompt + JSON shape + the exact `growmos apply …` command. Optional
@@ -75,7 +78,7 @@ Manually, the loop is:
growmos next # packet: prompt + shape + apply command
# … agent produces the JSON …
growmos apply extraction out.json --source src_ab12 --chunk 0
growmos next # → resolution packet, then profile packets, then "up to date"
growmos next # → resolution → profiles → gold set → review → "up to date"
growmos query "what depends on the Store and who decided that?"
growmos remember "Scheduler" --type COMPONENT --desc "Schedules jobs; depends on Store."
growmos link "Scheduler" "depends on" "Store"
+19 -10
View File
@@ -3,15 +3,21 @@
> Change the extraction prompt → rerun the scorer → watch F1 move. A team that ships without
> this loop cannot tell whether a prompt change improved or degraded quality.
## 1. Build a gold set
## 1. Build a gold set (automatic by default)
Pick at least two representative documents. Pre-fill from the current extraction, then
**hand-correct** (remove wrong entities, add missed ones — the whole point is human judgment):
`growmos next` hands out a **gold packet** for the richest extracted documents until `gold_min`
(default 2) files exist: the agent reads the document and writes the reference answer, and the file
records `_reviewed_by: agent`. Nothing manual is required. If you want a human pass, pre-fill and
edit — the file then records your name:
```bash
growmos gold-template docs/apollo-11.md
growmos gold-template docs/apollo-11.md # pre-fill from current extraction
$EDITOR .growmos/eval/gold/apollo-11.json
# or: growmos apply gold my.json --source docs/apollo-11.md --reviewer human
```
Caveat worth knowing: an agent-written gold set shares the blind spots of the agent that wrote it;
it still catches regressions (prompt drift, resolver over-merging) reliably, which is what the loop
is for.
Gold format:
```json
@@ -23,11 +29,12 @@ Relations are scored on (source, target) pairs, direction-agnostic, ignoring pre
an upper bound on relation recall that catches structural errors (missing/wrong connections),
which matter more than wording.
## 2. Keep the scorer alias map current
## 2. Scorer alias map (automatic)
If the resolver picks a canonical form the gold set doesn't use ("Neil Alden Armstrong" vs
"Neil Armstrong"), resolved recall drops — a scoring artifact, not a resolver bug.
`growmos eval` lists such canonicals; add them to `.growmos/eval/aliases.json`:
"Neil Armstrong"), resolved recall would drop — a scoring artifact, not a resolver bug.
`growmos eval` detects these (the raw mention matched a gold name) and **extends
`.growmos/eval/aliases.json` itself**, then rescores. You can also add entries by hand:
```json
{"Neil Alden Armstrong": "Neil Armstrong", "John F. Kennedy Space Center": "Kennedy Space Center"}
```
@@ -60,7 +67,9 @@ the change alters what the graph means, so rows produced under different prompts
`pending_sources`, `provisional`, `stale_profiles`. Extraction rate per document is in each
source's `stats`. Wire them into CI (`growmos integrate ci`).
## 6. The human sample
## 6. The periodic review (automatic)
`growmos sample` prints a random (degree-weighted) node with its profile, edges, provenance and
sources. Read it against the sources. `growmos doctor` turns red if nobody has done this in 7 days.
Every `review_days` (default 7) `growmos next` hands out a **review packet**: one random
(degree-weighted) node with its edges, provenance and source excerpts; the agent verifies each edge,
fixes what's wrong, and reports ok/issues (recorded in the journal). `growmos sample` remains
available for a human pass; `growmos doctor` shows who reviewed last and when.
+1 -1
View File
@@ -22,7 +22,7 @@ dependencies = []
[project.urls]
Homepage = "https://codician.com"
Repository = "https://github.com/codician/growmos"
Repository = "https://github.com/codician-team/growmos"
[project.scripts]
growmos = "growmos.cli:main"
+104 -4
View File
@@ -226,9 +226,38 @@ def _next_packet(st: Store) -> Optional[Dict[str, Any]]:
if deg >= min_deg and st.profile_is_stale(eid):
text, meta = summarize_packet(st, eid)
return {"kind": "profile", "text": text, "meta": meta}
# ground truth: keep at least `gold_min` gold files (agent-reviewed; humans may overwrite)
from .prompts import gold_packet, review_packet
gold_min = int(st.config.get("gold_min", 2))
have = {read_json(f, {}).get("source") for f in st.p("eval", "gold").glob("*.json")}
if len(have) < gold_min:
cands = [r for r in st.sources.values() if r.get("status") == "extracted" and r.get("kind", "file") == "file"
and r["ref"] not in have]
cands.sort(key=lambda r: -(r.get("stats") or {}).get("entities", 0))
if cands:
text, meta = gold_packet(st, cands[0]["id"])
return {"kind": "gold", "text": text, "meta": meta}
# comprehension check: one node every `review_days`
if st.entities and _days_since(st.state.get("last_sample")) is None or \
(st.entities and _days_since(st.state.get("last_sample")) >= int(st.config.get("review_days", 7))):
eid = g.sample(random.Random(len(st.relations)))
if eid:
text, meta = review_packet(st, eid)
return {"kind": "review", "text": text, "meta": meta}
return None
def _days_since(ts: Optional[str]) -> Optional[int]:
if not ts:
return None
from datetime import datetime, timezone
try:
dt = datetime.fromisoformat(ts.replace("Z", "+00:00"))
except ValueError:
return None
return (datetime.now(timezone.utc) - dt).days
def _remember_packet(st: Store, etype: str, names: List[str]) -> None:
"""Cache the names handed out in a resolution packet so `apply` can enforce
'every input name appears in exactly one cluster' against the real input."""
@@ -317,8 +346,51 @@ def cmd_apply(args: argparse.Namespace) -> int:
f"{prof['time_range']['start']}{prof['time_range']['end']})")
for p in problems:
print(f" ! {p}")
elif kind == "gold":
source = args.source or (payload.get("source") if isinstance(payload, dict) else None)
if not source:
raise StoreError("--source <src_id> is required")
if source not in st.sources:
match = [s for s, r in st.sources.items() if r.get("ref") == source]
if not match:
raise StoreError(f"unknown source '{source}'")
source = match[0]
ref = st.sources[source]["ref"]
ents = [{"name": e["name"], "type": str(e.get("type", "")).upper()} for e in payload.get("entities", [])
if isinstance(e, dict) and isinstance(e.get("name"), str) and e["name"].strip()]
pairs = [{"source": r["source"], "target": r["target"]} for r in payload.get("relations", [])
if isinstance(r, dict) and isinstance(r.get("source"), str) and isinstance(r.get("target"), str)]
out = st.p("eval", "gold", Path(ref).stem + ".json")
write_json(out, {"source": ref, "entities": ents, "relations": pairs,
"_reviewed_by": args.reviewer or "agent", "_ts": now_iso()})
st.record_run("gold", {"source": source, "reviewer": args.reviewer or "agent"})
st.save()
print(f"✓ gold set written for {ref}: {len(ents)} entities, {len(pairs)} relation pairs "
f"(reviewed by {args.reviewer or 'agent'}; humans may edit {out.relative_to(st.root)})")
elif kind == "review":
eid = args.entity or (payload.get("entity") if isinstance(payload, dict) else None)
if not eid or eid not in st.entities:
r = st.resolve_name(eid or "")
if not r:
raise StoreError(f"unknown entity '{eid}'")
eid = r
ok = bool(payload.get("ok"))
issues = [i for i in payload.get("issues", []) if isinstance(i, str)]
fixes = [f for f in payload.get("fixes", []) if isinstance(f, str)]
st.state["last_sample"] = now_iso()
st.state["last_sample_entity"] = eid
st.state["last_sample_ok"] = ok
st.record_run("review", {"entity": eid, "ok": ok, "issues": len(issues), "reviewer": args.reviewer or "agent"})
note = f"Review of {st.entities[eid]['name']} ({eid}): {'ok' if ok else 'issues found'}."
if issues:
note += "\n" + "\n".join(f"- {i}" for i in issues)
if fixes:
note += "\nFixes:\n" + "\n".join(f"- {f}" for f in fixes)
st.journal(note, author=args.reviewer or "agent-review")
st.save()
print(f"✓ review recorded for {st.entities[eid]['name']}: {'ok' if ok else f'{len(issues)} issue(s)'}")
else:
raise StoreError("kind must be extraction | resolution | profile")
raise StoreError("kind must be extraction | resolution | profile | gold | review")
return 0
@@ -679,6 +751,33 @@ def cmd_ingest(args: argparse.Namespace) -> int:
st.set_profile(eid, prof)
st.save()
print(f"profiled {st.entities[eid]['name']}")
# gold + review, so the loop is complete without a human
from .prompts import gold_packet, review_packet
gold_min = int(st.config.get("gold_min", 2))
have = {read_json(f, {}).get("source") for f in st.p("eval", "gold").glob("*.json")}
cands = [r for r in st.sources.values() if r.get("status") == "extracted" and r.get("kind", "file") == "file"
and r["ref"] not in have]
cands.sort(key=lambda r: -(r.get("stats") or {}).get("entities", 0))
for rec in cands[: max(0, gold_min - len(have))]:
pkt, meta = gold_packet(st, rec["id"])
data = call_structured(pkt.split("--- prompt ---\n", 1)[-1], S.GOLD_SCHEMA, stage="reason", config=st.config)
write_json(st.p("eval", "gold", Path(rec["ref"]).stem + ".json"),
{"source": rec["ref"], "entities": data.get("entities", []), "relations": data.get("relations", []),
"_reviewed_by": "agent (headless)", "_ts": now_iso()})
print(f"gold written for {rec['ref']}")
d = _days_since(st.state.get("last_sample"))
if st.entities and (d is None or d >= int(st.config.get("review_days", 7))):
eid = g.sample(random.Random(len(st.relations)))
pkt, meta = review_packet(st, eid)
data = call_structured(pkt.split("--- prompt ---\n", 1)[-1], S.REVIEW_SCHEMA, stage="reason", config=st.config)
st.state["last_sample"] = now_iso()
st.state["last_sample_entity"] = eid
st.state["last_sample_ok"] = bool(data.get("ok"))
st.journal(f"Headless review of {st.entities[eid]['name']}: {'ok' if data.get('ok') else 'issues'}\n"
+ "\n".join(f"- {i}" for i in data.get("issues", [])), author="agent-review")
print(f"reviewed {st.entities[eid]['name']}: {'ok' if data.get('ok') else 'issues found'}")
from .evaluate import Evaluator
Evaluator(st).evaluate()
except ProviderError as e:
st.save()
print(f"provider error: {e}", file=sys.stderr)
@@ -702,7 +801,7 @@ def build_parser() -> argparse.ArgumentParser:
p = argparse.ArgumentParser(
prog="growmos",
description="growmos — a living knowledge graph that grows with your repo (Codician, MIT).",
epilog="Agent loop: growmos context → work → growmos remember/link/journal → growmos next → apply. Docs: https://github.com/codician/growmos",
epilog="Agent loop: growmos context → work → growmos remember/link/journal → growmos next → apply. Docs: https://github.com/codician-team/growmos",
)
p.add_argument("--root", help="repository root (default: auto-detect via .growmos/ or .git/)")
p.add_argument("--version", action="version", version=f"growmos {__version__}")
@@ -744,13 +843,14 @@ def build_parser() -> argparse.ArgumentParser:
sp.set_defaults(fn=cmd_next)
sp = sub.add_parser("apply", help="ingest a completed packet's JSON")
sp.add_argument("kind", choices=["extraction", "resolution", "profile"])
sp.add_argument("kind", choices=["extraction", "resolution", "profile", "gold", "review"])
sp.add_argument("file", help="JSON file or - for stdin")
sp.add_argument("--source", help="source id/ref (extraction)")
sp.add_argument("--chunk", type=int, help="chunk index (extraction)")
sp.add_argument("--partial", action="store_true", help="more chunks follow; keep source pending")
sp.add_argument("--type", help="entity type (resolution)")
sp.add_argument("--entity", help="entity id or name (profile)")
sp.add_argument("--entity", help="entity id or name (profile/review)")
sp.add_argument("--reviewer", help="who reviewed (gold/review): agent (default) or human")
sp.set_defaults(fn=cmd_apply)
sp = sub.add_parser("extract", help="print the extraction packet for a source")
+34 -6
View File
@@ -90,6 +90,29 @@ class Evaluator:
"missed_entities": sorted(gold_ents - raw_names)[:10],
"extra_entities": sorted(raw_names - gold_ents)[:10],
})
# Auto-extend the scorer alias map: a canonical form whose raw mention matched a gold
# name is a scoring artifact, not a resolver bug — record canonical -> gold name.
auto_added = {}
if unrecognized:
amap = read_json(st.p("eval", "aliases.json"), {}) or {}
gold_names = {}
for gf in self.gold_files():
for e in (read_json(gf, {}) or {}).get("entities", []):
if e.get("name"):
gold_names[norm_name(e["name"])] = e["name"]
for cname in unrecognized:
eid = st.resolve_name(cname)
if not eid:
continue
for (alias, _t), aeid in st.aliases.items():
if aeid == eid and alias in gold_names and cname not in amap:
amap[cname] = gold_names[alias]
auto_added[cname] = gold_names[alias]
break
if auto_added:
write_json(st.p("eval", "aliases.json"), amap)
self.alias_map = {norm_name(k): v for k, v in amap.items()}
return self.evaluate() # rescore once with the extended map
report = {"ts": now_iso(), "docs": rows, "unrecognized_canonicals": sorted(unrecognized),
"schema_version": st.schema_version}
write_json(st.p("eval", "last_report.json"), report)
@@ -137,10 +160,12 @@ def doctor(store: Store) -> List[Dict[str, str]]:
checks.append({"item": name, "status": "ok" if ok else ("warn" if ok is None else "fail"),
"detail": detail, "fix": fix})
gold_n = len(list(st.p("eval", "gold").glob("*.json")))
gold_files = list(st.p("eval", "gold").glob("*.json"))
gold_n = len(gold_files)
reviewers = sorted({str((read_json(f, {}) or {}).get("_reviewed_by", "human")) for f in gold_files})
add("Gold set", gold_n >= 2 if gold_n else False,
f"{gold_n} gold file(s)" if gold_n else "no gold files → prompt changes are blind",
"add ≥2 hand-labelled files in .growmos/eval/gold/ (growmos gold-template <source>)")
(f"{gold_n} gold file(s), reviewed by: {', '.join(reviewers)}") if gold_n else "no gold files → prompt changes are blind",
"growmos next hands out gold packets until gold_min is met (or hand-write .growmos/eval/gold/*.json)")
amap = read_json(st.p("eval", "aliases.json"), None)
last = read_json(st.p("eval", "last_report.json"), {}) or {}
unrec = last.get("unrecognized_canonicals") or []
@@ -175,9 +200,12 @@ def doctor(store: Store) -> List[Dict[str, str]]:
days = (datetime.now(timezone.utc) - dt).days
except ValueError:
days = None
add("Human sample", (days is not None and days <= 7) if ls else False,
f"last sample {days} day(s) ago" if days is not None else "no one has sampled the graph yet",
"growmos sample — read one node's profile and check its edges against the sources")
who = "human" if st.state.get("last_sample_entity") is None else ("agent" if any(
r.get("kind") == "review" for r in st.state.get("runs", [])[-20:]) else "human")
add("Human sample", (days is not None and days <= int(st.config.get("review_days", 7))) if ls else False,
f"last review {days} day(s) ago ({who}; ok={st.state.get('last_sample_ok', '?')})" if days is not None
else "no one has reviewed a node yet",
"growmos next hands out a review packet every review_days (or growmos sample for a human pass)")
return checks
+54 -1
View File
@@ -24,7 +24,7 @@ from .store import Store
from .util import approx_tokens, truncate
TEMPLATE_DIR = Path(__file__).parent / "templates" / "prompts"
PROMPT_NAMES = ["extract", "resolve", "summarize", "query", "check"]
PROMPT_NAMES = ["extract", "resolve", "summarize", "query", "check", "gold", "review"]
def install_default_prompts(dest: Path, force: bool = False) -> List[str]:
@@ -120,6 +120,10 @@ def _shape(kind: str, entity_types: Optional[List[str]] = None) -> str:
if kind == "profile":
return ('{"summary": "2-3 paragraphs", "key_facts": ["3-5 atomic facts traceable to sources"],\n'
' "time_range": {"start": "YYYY|unknown", "end": "YYYY|ongoing|unknown"}}')
if kind == "gold":
return '{"entities": [{"name": "", "type": ""}], "relations": [{"source": "<name>", "target": "<name>"}]}'
if kind == "review":
return '{"ok": true, "issues": ["<edge id>: what is wrong"], "fixes": ["growmos link … / growmos remember … you ran or recommend"]}'
return "{}"
@@ -305,3 +309,52 @@ def check_packet(store: Store, content: str, hops: int = 2) -> Tuple[str, Dict[s
prompt = render(load_prompt(store, "check"), graph=triples or "(empty)", content=content)
meta = {"seeds": seeds, "nodes": len(nodes), "edges": total, "tokens": approx_tokens(prompt)}
return prompt, meta
def gold_packet(store: Store, source_id: str) -> Tuple[str, Dict[str, Any]]:
"""Ask the agent to produce (or correct) the gold answer for a source — the eval loop's ground truth."""
rec = store.sources[source_id]
text = store.source_text(source_id)
ents = [{"name": m["name"], "type": m["type"]} for m in store.mentions() if m.get("source") == source_id]
pairs = [{"source": store.entities[r["source"]]["name"], "target": store.entities[r["target"]]["name"]}
for r in store.relations.values() if source_id in r.get("sources", [])
and r["source"] in store.entities and r["target"] in store.entities]
prompt = render(load_prompt(store, "gold"), source_ref=rec["ref"], text=truncate(text, 20000),
extraction=json.dumps({"entities": ents, "relations": pairs}, ensure_ascii=False, indent=1),
entity_types=", ".join(store.entity_types))
stem = Path(rec["ref"]).stem or source_id
out_file = f".growmos/cache/gold_{source_id}.json"
apply_cmd = f"growmos apply gold {out_file} --source {source_id}"
header = _packet_header(f"gold set · {rec['ref']}", apply_cmd, S.GOLD_SCHEMA, out_file, kind="gold")
meta = {"source": source_id, "stem": stem, "apply": apply_cmd, "out_file": out_file, "tokens": approx_tokens(prompt)}
return header + prompt, meta
def review_packet(store: Store, eid: str, max_excerpt_chars: int = 1500) -> Tuple[str, Dict[str, Any]]:
"""The comprehension check: one node, its edges, and the sources those edges cite."""
g = Graph(store)
card = g.entity_card(eid)
ent = store.entities[eid]
sids: List[str] = []
for rid in g.out.get(eid, []) + g.inc.get(eid, []):
for sid in store.relations[rid].get("sources", []):
if sid not in sids:
sids.append(sid)
for sid in ent.get("sources", []):
if sid not in sids:
sids.append(sid)
excerpts = []
for sid in sids[:8]:
rec = store.sources.get(sid, {})
try:
text = store.source_text(sid)
except Exception:
text = ""
body = _excerpt_around(text, ent["name"], max_excerpt_chars) if text else "(note/session source — no text)"
excerpts.append(f"[{rec.get('ref', sid)}]\n{body}")
prompt = render(load_prompt(store, "review"), card=card, excerpts="\n\n".join(excerpts) or "(none)")
out_file = f".growmos/cache/review_{eid.replace('/', '__')}.json"
apply_cmd = f"growmos apply review {out_file} --entity \"{eid}\""
header = _packet_header(f"review · {ent['name']}", apply_cmd, S.REVIEW_SCHEMA, out_file, kind="review")
meta = {"entity": eid, "apply": apply_cmd, "out_file": out_file, "tokens": approx_tokens(prompt)}
return header + prompt, meta
+25
View File
@@ -129,6 +129,31 @@ PROFILE_SCHEMA: Dict[str, Any] = {
"additionalProperties": False,
}
GOLD_SCHEMA: Dict[str, Any] = {
"type": "object",
"properties": {
"entities": {"type": "array", "items": {"type": "object", "properties": {
"name": {"type": "string"}, "type": {"type": "string"}},
"required": ["name", "type"], "additionalProperties": False}},
"relations": {"type": "array", "items": {"type": "object", "properties": {
"source": {"type": "string"}, "target": {"type": "string"}},
"required": ["source", "target"], "additionalProperties": False}},
},
"required": ["entities", "relations"],
"additionalProperties": False,
}
REVIEW_SCHEMA: Dict[str, Any] = {
"type": "object",
"properties": {
"ok": {"type": "boolean"},
"issues": {"type": "array", "items": {"type": "string"}},
"fixes": {"type": "array", "items": {"type": "string"}},
},
"required": ["ok", "issues", "fixes"],
"additionalProperties": False,
}
ANSWER_SCHEMA: Dict[str, Any] = {
"type": "object",
"properties": {
+2
View File
@@ -90,6 +90,8 @@ class Store:
"chunk_chars": 6000,
"profile_min_degree": 3,
"resolve_batch_size": 80,
"gold_min": 2,
"review_days": 7,
"provider": {"name": "", "extract_model": "", "reason_model": ""},
}
st.schema = {
@@ -42,6 +42,13 @@ no invented facts.
**Query** — answer from the graph only, cite edges, flag gaps. On a private corpus only the
grounded answer works at all.
**Gold packet** — you are writing the reference answer for the eval loop: read the document, keep
only genuinely central entities and (source,target) pairs, add what extraction missed, remove what
is not stated. Precision first. Humans may later overwrite the file; say so is fine.
**Review packet** — the periodic comprehension check: verify every edge of one node against the
cited sources; fix what is wrong with `growmos link`/`remember`/`journal`, then report ok/issues.
## Applying results
Write the JSON to the `out_file` named in the packet header (or pipe on stdin) and run the
@@ -51,6 +58,8 @@ printed command, e.g.
growmos apply extraction .growmos/cache/extract_src_xxx_0.json --source src_xxx --chunk 0
growmos apply resolution .growmos/cache/resolve_person_0.json --type PERSON
growmos apply profile .growmos/cache/profile_x.json --entity "component/graph-store"
growmos apply gold .growmos/cache/gold_src_xxx.json --source src_xxx
growmos apply review .growmos/cache/review_x.json --entity "component/graph-store"
```
The CLI validates against the schema, drops dangling relations, gives unmatched names a
@@ -14,8 +14,9 @@ memory you read at the start of work and write to as you develop. Zero-config co
- `growmos link "<A>" "<predicate>" "<B>"` (short verb phrase predicates: "depends on", "replaces")
- `growmos journal "<what changed and why>"`
4. **Feed the organism** — run `growmos next`. It hands you a *task packet* (extraction / resolution /
profile) with the exact prompt, the JSON schema, and the `growmos apply …` command. Do the judgment
work yourself, write the JSON, apply it. Repeat until `growmos next` says the graph is up to date.
profile / gold set / review) with the exact prompt, the JSON shape, and the `growmos apply …` command.
Do the judgment work yourself, write the JSON, apply it. Repeat until `growmos next` says the graph is
up to date — that loop covers everything, including the evaluation gold set and the periodic node review.
Never invent facts not in the source; every relation must connect two extracted entities.
5. **Before claiming facts about the repo in a summary/report**`growmos check "<claim text>"` grounds
your claims against edges with provenance (evaluatoroptimizer loop).
@@ -23,5 +24,5 @@ memory you read at the start of work and write to as you develop. Zero-config co
Store files are plain JSONL under `.growmos/` — commit them with your code. Do not hand-edit
`entities.jsonl`/`relations.jsonl` (use the CLI); prompts in `.growmos/prompts/` are yours to tune.
More: `growmos --help`, docs at https://github.com/codician/growmos.
More: `growmos --help`, docs at https://github.com/codician-team/growmos.
<!-- growmos:end -->
+11
View File
@@ -0,0 +1,11 @@
You are building the gold set used to score the extraction pipeline. Read the document carefully and produce the reference answer: the entities that are genuinely central to what this document is about (with their types) and the relations between them as (source, target) pairs. Use the current extraction below only as a starting point — add what it missed, remove what is not actually stated or is merely incidental, and correct types. Precision matters more than recall: do not include an entity you cannot point to in the text.
<document source="{source_ref}">
{text}
</document>
<current_extraction>
{extraction}
</current_extraction>
Allowed entity types: {entity_types}
+11
View File
@@ -0,0 +1,11 @@
This is the periodic comprehension check on the knowledge graph (the playbook's "read a random node each day"). Below is one node with its profile, edges and provenance, followed by excerpts of the sources those edges cite. Verify each edge against the sources: is it stated there, with that direction and meaning? Is the description/profile faithful? Are there obviously missing central relations?
<node>
{card}
</node>
<sources>
{excerpts}
</sources>
Report ok=true only if every edge is supported. List concrete issues (edge id + what is wrong). List fixes as growmos commands you would run (e.g. growmos link "A" "pred" "B" --source docs/x.md), or apply them yourself before reporting.
+42
View File
@@ -138,6 +138,48 @@ class ApolloFixture(unittest.TestCase):
self.assertIn("growmos apply extraction", pkt)
self.assertEqual(meta["chunks"], 1)
def test_next_hands_out_gold_then_review_then_done(self):
import copy
tmp2 = Path(tempfile.mkdtemp(prefix="growmos-next-"))
shutil.copytree(self.tmp, tmp2 / "r")
root = tmp2 / "r"
st = Store(root).load()
for f in st.p("eval", "gold").glob("*.json"):
f.unlink()
for eid, deg in Graph(st).hubs(50):
if deg >= 3:
st.set_profile(eid, {"summary": "s", "key_facts": ["k"], "time_range": {"start": "u", "end": "u"}})
st.state["last_sample"] = None
st.save()
out = run_cli("--root", str(root), "next")
self.assertIn("gold set", out)
for _ in range(2):
meta = json.loads(run_cli("--root", str(root), "next", "--json"))["meta"]
run_cli("--root", str(root), "apply", "gold", "-", "--source", meta["source"],
stdin=json.dumps({"entities": [{"name": "Apollo 11", "type": "EVENT"}], "relations": []}))
out = run_cli("--root", str(root), "next")
self.assertIn("review", out)
meta = json.loads(run_cli("--root", str(root), "next", "--json"))["meta"]
run_cli("--root", str(root), "apply", "review", "-", "--entity", meta["entity"],
stdin=json.dumps({"ok": True, "issues": [], "fixes": []}))
out = run_cli("--root", str(root), "next")
self.assertIn("up to date", out)
checks = {c["item"]: c["status"] for c in doctor(Store(root).load())}
self.assertEqual(checks["Gold set"], "ok")
self.assertEqual(checks["Human sample"], "ok")
shutil.rmtree(tmp2, ignore_errors=True)
def test_eval_auto_extends_alias_map(self):
tmp2 = Path(tempfile.mkdtemp(prefix="growmos-alias-"))
shutil.copytree(self.tmp, tmp2 / "r")
st = Store(tmp2 / "r").load()
(st.p("eval", "aliases.json")).write_text("{}", encoding="utf-8")
rep = Evaluator(st).evaluate()
amap = read_json(st.p("eval", "aliases.json"))
self.assertEqual(amap.get("Neil Alden Armstrong"), "Neil Armstrong")
self.assertEqual(rep["unrecognized_canonicals"], [])
shutil.rmtree(tmp2, ignore_errors=True)
def test_export_formats(self):
from growmos.export import EXPORTERS
for name, fn in EXPORTERS.items():