2026-06-18 06:00:06 +00:00
#!/usr/bin/env python3
2026-06-18 05:49:07 +00:00
"""
2026-06-18 12:00:08 +00:00
agent . py — Vault Entity / Concept Extractor + Graph Builder ( v2 )
2026-06-18 06:00:06 +00:00
2026-06-18 12:00:08 +00:00
Pivot from generic topic - tags to a wikilink - driven graph :
( a ) extract entities / concepts from doc via Qwen vault - tagger
( b ) canonicalize each term → page name ( Title - Case ) + filename ( kebab )
( c ) decide which terms qualify for a page ( vocab OR ≥ 2 corpus mentions )
( d ) body - wikilink : insert [ [ Name ] ] at first prose occurrence only
( e ) related : bge - m3 nearest - docs ( unchanged )
( f ) tags : demoted to single primary_topic ( or none )
2026-06-18 06:00:06 +00:00
Usage :
python3 engine / lib / agent . py - - dry - run < path > # print diff, write nothing
python3 engine / lib / agent . py < path > # apply if confidence >= threshold
2026-06-18 12:00:08 +00:00
Importable : ollama_extract ( ) , tei_embed ( ) , process_doc ( ) , load_config ( ) , load_vocab ( )
2026-06-18 06:00:06 +00:00
stdlib only — no pip installs ; all HTTP via urllib .
2026-06-18 05:49:07 +00:00
"""
2026-06-18 06:00:06 +00:00
from __future__ import annotations
import argparse
import difflib
import json
import math
import os
import re
import sys
import urllib . error
import urllib . request
from datetime import datetime , timezone
from pathlib import Path
# ---------------------------------------------------------------------------
2026-06-18 12:00:08 +00:00
# Config loader (minimal stdlib YAML parser)
2026-06-18 06:00:06 +00:00
# ---------------------------------------------------------------------------
def load_config ( config_path : str | Path | None = None ) - > dict :
""" Load config.yaml; fall back to defaults if file is missing or unparseable. """
defaults : dict = {
" vault_dir " : " /home/zvx/projects/.ref/vault " ,
" engine_dir " : " /home/zvx/projects/.ref/engine " ,
" topic_categories " : [
" mesh " , " matrix " , " recon " , " media " , " auth " ,
" dns " , " vpn " , " storage " , " proxmox " , " ai " , " mail " ,
] ,
" models " : {
" tagger " : {
" ollama_endpoint " : " http://localhost:11434 " ,
" model " : " vault-tagger " ,
" temperature " : 0.1 ,
" format " : " json " ,
} ,
" embeddings " : {
" tei_endpoint " : " http://localhost:8090 " ,
" model " : " bge-m3 " ,
} ,
} ,
" frontmatter_schema " : {
" required " : [ " title " , " type " , " tags " , " updated " ] ,
" optional " : [ " aliases " , " related " , " status " ] ,
" types " : [ " reference " , " runbook " , " project " , " note " , " index " , " session " ] ,
} ,
" behavior " : {
" auto_apply " : True ,
" log_changes " : True ,
" changelog " : " /home/zvx/projects/.ref/engine/changelog.md " ,
" confidence_threshold " : 0.6 ,
} ,
}
if config_path is None :
here = Path ( __file__ ) . parent
candidates = [
here . parent / " config.yaml " ,
Path ( " /home/zvx/projects/.ref/engine/config.yaml " ) ,
]
for c in candidates :
if c . exists ( ) :
config_path = c
break
if config_path is None or not Path ( config_path ) . exists ( ) :
return defaults
cfg = dict ( defaults )
try :
with open ( config_path ) as f :
lines = f . readlines ( )
2026-06-18 05:49:07 +00:00
2026-06-18 06:00:06 +00:00
for line in lines :
m = re . match ( r ' ^(vault_dir|engine_dir): \ s*(.+) ' , line )
if m :
cfg [ m . group ( 1 ) ] = m . group ( 2 ) . strip ( )
2026-06-18 05:49:07 +00:00
2026-06-18 06:00:06 +00:00
in_topics = False
topics : list [ str ] = [ ]
for line in lines :
if re . match ( r ' ^topic_categories: ' , line ) :
in_topics = True
continue
if in_topics :
mm = re . match ( r ' ^ \ s+- \ s+( \ S+) ' , line )
if mm :
topics . append ( mm . group ( 1 ) . strip ( ) )
elif line . strip ( ) and not line . startswith ( ' ' ) :
in_topics = False
if topics :
cfg [ " topic_categories " ] = topics
2026-06-18 05:49:07 +00:00
2026-06-18 06:00:06 +00:00
in_models = False
in_tagger = False
in_embed = False
in_behavior = False
for line in lines :
if re . match ( r ' ^models: ' , line ) :
in_models = True
continue
if re . match ( r ' ^behavior: ' , line ) :
in_models = False
in_behavior = True
continue
if in_models :
if re . match ( r ' tagger: ' , line ) :
in_tagger = True
in_embed = False
elif re . match ( r ' embeddings: ' , line ) :
in_embed = True
in_tagger = False
elif not line . startswith ( ' ' ) :
in_models = False
2026-06-18 05:49:07 +00:00
2026-06-18 06:00:06 +00:00
if in_tagger :
mm = re . match ( r ' \ s+ollama_endpoint: \ s*( \ S+) ' , line )
if mm :
cfg [ " models " ] [ " tagger " ] [ " ollama_endpoint " ] = mm . group ( 1 ) . strip ( )
mm = re . match ( r ' \ s+model: \ s*( \ S+) ' , line )
if mm :
cfg [ " models " ] [ " tagger " ] [ " model " ] = mm . group ( 1 ) . strip ( )
mm = re . match ( r ' \ s+temperature: \ s*([ \ d.]+) ' , line )
if mm :
cfg [ " models " ] [ " tagger " ] [ " temperature " ] = float ( mm . group ( 1 ) )
2026-06-18 05:49:07 +00:00
2026-06-18 06:00:06 +00:00
if in_embed :
mm = re . match ( r ' \ s+tei_endpoint: \ s*( \ S+) ' , line )
if mm :
cfg [ " models " ] [ " embeddings " ] [ " tei_endpoint " ] = mm . group ( 1 ) . strip ( )
2026-06-18 05:49:07 +00:00
2026-06-18 06:00:06 +00:00
if in_behavior :
mm = re . match ( r ' \ s+auto_apply: \ s*( \ S+) ' , line )
if mm :
cfg [ " behavior " ] [ " auto_apply " ] = mm . group ( 1 ) . strip ( ) . lower ( ) == ' true '
mm = re . match ( r ' \ s+confidence_threshold: \ s*([ \ d.]+) ' , line )
if mm :
cfg [ " behavior " ] [ " confidence_threshold " ] = float ( mm . group ( 1 ) )
mm = re . match ( r ' \ s+changelog: \ s*(.+) ' , line )
if mm :
cfg [ " behavior " ] [ " changelog " ] = mm . group ( 1 ) . strip ( )
2026-06-18 05:49:07 +00:00
2026-06-18 06:00:06 +00:00
except Exception as e :
print ( f " [warn] Could not fully parse { config_path } : { e } ; using defaults " , file = sys . stderr )
2026-06-18 05:49:07 +00:00
2026-06-18 06:00:06 +00:00
return cfg
# ---------------------------------------------------------------------------
# Vocab loader
# ---------------------------------------------------------------------------
def load_vocab ( engine_dir : str | Path ) - > dict :
2026-06-18 12:00:08 +00:00
""" Load vocab.json. Returns dict with topic_categories, entities, acronyms. """
2026-06-18 06:00:06 +00:00
vocab_path = Path ( engine_dir ) / " vocab.json "
if not vocab_path . exists ( ) :
print ( f " [warn] vocab.json not found at { vocab_path } " , file = sys . stderr )
return { " topic_categories " : [ ] , " entities " : [ ] , " acronyms " : [ ] }
with open ( vocab_path ) as f :
return json . load ( f )
# ---------------------------------------------------------------------------
# Frontmatter parser (stdlib)
# ---------------------------------------------------------------------------
def parse_frontmatter ( text : str ) - > tuple [ dict , str ] :
2026-06-18 12:00:08 +00:00
""" Parse YAML frontmatter. Returns (fm_dict, body_text). """
2026-06-18 06:00:06 +00:00
fm : dict = { }
body = text
m = re . match ( r ' ^--- \ r? \ n(.*?) \ r? \ n--- \ r? \ n?(.*) ' , text , re . DOTALL )
if not m :
return fm , body
fm_raw = m . group ( 1 )
body = m . group ( 2 )
lines = fm_raw . splitlines ( )
i = 0
while i < len ( lines ) :
line = lines [ i ]
if not line . strip ( ) or line . strip ( ) . startswith ( ' # ' ) :
i + = 1
continue
kv = re . match ( r ' ^( \ w[ \ w-]*): \ s*(.*) ' , line )
if not kv :
i + = 1
continue
key = kv . group ( 1 )
val_raw = kv . group ( 2 ) . strip ( )
if val_raw . startswith ( ' [ ' ) :
inner = re . sub ( r ' ^ \ [| \ ]$ ' , ' ' , val_raw ) . strip ( )
if inner :
items = [ x . strip ( ) . strip ( ' " \' ' ) for x in inner . split ( ' , ' ) if x . strip ( ) ]
else :
items = [ ]
fm [ key ] = items
i + = 1
continue
if not val_raw :
lst = [ ]
j = i + 1
while j < len ( lines ) and re . match ( r ' ^ \ s+- \ s+(.*) ' , lines [ j ] ) :
item_m = re . match ( r ' ^ \ s+- \ s+(.*) ' , lines [ j ] )
lst . append ( item_m . group ( 1 ) . strip ( ) . strip ( ' " \' ' ) )
j + = 1
if lst :
fm [ key ] = lst
i = j
continue
val = val_raw . strip ( ' " \' ' )
try :
if re . match ( r ' ^ \ d+$ ' , val ) :
fm [ key ] = int ( val )
elif re . match ( r ' ^ \ d+ \ . \ d+$ ' , val ) :
fm [ key ] = float ( val )
else :
fm [ key ] = val
except Exception :
fm [ key ] = val
i + = 1
return fm , body
def render_frontmatter ( fm : dict ) - > str :
2026-06-18 12:00:08 +00:00
""" Render frontmatter dict to YAML string (between --- markers). """
2026-06-18 06:00:06 +00:00
lines = [ " --- " ]
key_order = [ " title " , " type " , " tags " , " aliases " , " related " , " updated " ]
written = set ( )
for key in key_order :
if key not in fm :
continue
written . add ( key )
val = fm [ key ]
if isinstance ( val , list ) :
if not val :
lines . append ( f " { key } : [] " )
else :
lines . append ( f " { key } : " )
for item in val :
lines . append ( f " - { item } " )
else :
sv = str ( val )
if any ( c in sv for c in ' :# {} []|>&*!,? ' ) :
lines . append ( f ' { key } : " { sv } " ' )
else :
lines . append ( f " { key } : { sv } " )
for key , val in fm . items ( ) :
if key in written :
continue
if isinstance ( val , list ) :
if not val :
lines . append ( f " { key } : [] " )
else :
lines . append ( f " { key } : " )
for item in val :
lines . append ( f " - { item } " )
else :
sv = str ( val )
if any ( c in sv for c in ' :# {} []|>&*!,? ' ) :
lines . append ( f ' { key } : " { sv } " ' )
else :
lines . append ( f " { key } : { sv } " )
lines . append ( " --- " )
return " \n " . join ( lines ) + " \n "
# ---------------------------------------------------------------------------
# HTTP helpers (urllib only)
# ---------------------------------------------------------------------------
def _http_post ( url : str , payload : dict , timeout : int = 120 ) - > dict | list :
data = json . dumps ( payload ) . encode ( " utf-8 " )
req = urllib . request . Request (
url ,
data = data ,
headers = { " Content-Type " : " application/json " } ,
method = " POST " ,
)
with urllib . request . urlopen ( req , timeout = timeout ) as resp :
return json . loads ( resp . read ( ) . decode ( " utf-8 " ) )
def _http_get ( url : str , timeout : int = 10 ) - > dict :
req = urllib . request . Request ( url , method = " GET " )
with urllib . request . urlopen ( req , timeout = timeout ) as resp :
return json . loads ( resp . read ( ) . decode ( " utf-8 " ) )
# ---------------------------------------------------------------------------
2026-06-18 12:00:08 +00:00
# Canonicalization helpers
2026-06-18 06:00:06 +00:00
# ---------------------------------------------------------------------------
2026-06-18 12:00:08 +00:00
def to_kebab ( s : str ) - > str :
""" Convert a string to lowercase-kebab-case for filenames. """
s = s . strip ( ) . lower ( )
s = re . sub ( r ' [ \ s_/]+ ' , ' - ' , s )
s = re . sub ( r ' [^a-z0-9 \ -] ' , ' ' , s )
s = re . sub ( r ' - { 2,} ' , ' - ' , s )
return s . strip ( ' - ' )
def to_title_case ( s : str ) - > str :
""" Convert kebab/slug to Title Case display name. """
# e.g. "raspberry-pi" -> "Raspberry Pi", "sso" -> "SSO" (handled by acronym list)
_FORCE_UPPER = { " sso " , " tls " , " dns " , " vpn " , " lxc " , " oidc " , " tcp " , " udp " ,
" api " , " url " , " ssh " , " bbs " , " lora " , " gui " , " ui " , " ram " ,
" cpu " , " gpu " , " nas " , " uuid " , " ip " , " os " }
parts = re . split ( r ' [-_ \ s]+ ' , s . strip ( ) )
result = [ ]
for p in parts :
if p . lower ( ) in _FORCE_UPPER :
result . append ( p . upper ( ) )
else :
result . append ( p . capitalize ( ) )
return ' ' . join ( result )
2026-06-18 06:00:06 +00:00
2026-06-18 12:00:08 +00:00
def canonicalize_term (
raw : str ,
vocab : dict ,
vault_dir : Path ,
existing_pages_cache : dict [ str , str ] | None = None ,
) - > tuple [ str , str , str ] :
"""
Given a raw term ( from LLM ) , return ( canonical_display , kebab_filename , match_type ) .
2026-06-18 06:00:06 +00:00
2026-06-18 12:00:08 +00:00
match_type : ' vocab-entity ' | ' vocab-alias ' | ' existing-page ' | ' new '
2026-06-18 06:00:06 +00:00
2026-06-18 12:00:08 +00:00
Checks ( in order ) :
1. Exact match against vocab entity names ( case - insensitive )
2. Alias match against vocab entities
3. Existing page in vault / entities / or vault / concepts / ( case - insensitive )
4. New term — invent canonical from the raw string
2026-06-18 06:00:06 +00:00
2026-06-18 12:00:08 +00:00
existing_pages_cache : { kebab_filename : display_name } — populated lazily .
"""
raw_stripped = raw . strip ( )
raw_lower = raw_stripped . lower ( ) . replace ( ' ' , ' - ' )
2026-06-18 06:00:06 +00:00
2026-06-18 12:00:08 +00:00
# Build vocab entity lookup: {lowercase_name_or_alias: (canonical_name, display)}
vocab_lookup : dict [ str , tuple [ str , str ] ] = { }
for ent in vocab . get ( " entities " , [ ] ) :
name = ent . get ( " name " , " " )
display = ent . get ( " display " , " " ) or to_title_case ( name )
vocab_lookup [ name . lower ( ) ] = ( name , display )
for alias in ent . get ( " aliases " , [ ] ) :
alias_s = str ( alias ) . lower ( )
if alias_s not in vocab_lookup :
vocab_lookup [ alias_s ] = ( name , display )
# Also normalise spaces to hyphens for matching
def _norm ( s : str ) - > str :
return re . sub ( r ' [ \ s_]+ ' , ' - ' , s . strip ( ) . lower ( ) )
raw_norm = _norm ( raw_stripped )
# 1. Vocab entity name match
if raw_norm in vocab_lookup :
name , display = vocab_lookup [ raw_norm ]
return display or to_title_case ( name ) , to_kebab ( name ) , ' vocab-entity '
# 2. Vocab alias match (substring normalised)
for alias_key , ( name , display ) in vocab_lookup . items ( ) :
if _norm ( alias_key ) == raw_norm :
return display or to_title_case ( name ) , to_kebab ( name ) , ' vocab-alias '
# 3. Existing page match
if existing_pages_cache is not None :
if raw_norm in existing_pages_cache :
existing_disp = existing_pages_cache [ raw_norm ]
return existing_disp , raw_norm , ' existing-page '
# also try without hyphens/spaces normalisation
for pg_key , pg_disp in existing_pages_cache . items ( ) :
if pg_key == raw_norm :
return pg_disp , pg_key , ' existing-page '
# 4. New term
kebab = to_kebab ( raw_stripped )
display = to_title_case ( raw_stripped )
return display , kebab , ' new '
def build_existing_pages_cache ( vault_dir : Path ) - > dict [ str , str ] :
"""
Scan vault / entities / and vault / concepts / for existing pages .
Returns { kebab_stem : display_title } ( case - insensitive key ) .
"""
cache : dict [ str , str ] = { }
for subdir in [ " entities " , " concepts " ] :
d = vault_dir / subdir
if not d . exists ( ) :
continue
for md in d . glob ( " *.md " ) :
stem = to_kebab ( md . stem )
# Try to extract title from frontmatter
try :
text = md . read_text ( errors = " replace " )
fm , _ = parse_frontmatter ( text )
display = fm . get ( " title " , " " ) or to_title_case ( md . stem )
except Exception :
display = to_title_case ( md . stem )
cache [ stem ] = str ( display )
return cache
# ---------------------------------------------------------------------------
# Corpus mention index
# ---------------------------------------------------------------------------
def build_corpus_mention_index (
vault_dir : Path ,
terms : list [ str ] , # canonical kebab names to scan for
vocab : dict ,
) - > dict [ str , list [ str ] ] :
"""
Deterministic substring scan of ALL vault / * . md files for each term .
Returns { term_kebab : [ list of doc paths that mention it ] } .
Matches : term name , aliases from vocab , surface variations ( spaces < - > hyphens ) .
Skips fenced code blocks and frontmatter in each doc for a cleaner count .
"""
# Build match patterns per term
term_patterns : dict [ str , list [ str ] ] = { }
vocab_by_kebab : dict [ str , dict ] = { }
for ent in vocab . get ( " entities " , [ ] ) :
kb = to_kebab ( ent . get ( " name " , " " ) )
vocab_by_kebab [ kb ] = ent
for term_kebab in terms :
patterns = [ term_kebab , term_kebab . replace ( ' - ' , ' ' ) ]
# Add vocab aliases if this is a vocab entity
if term_kebab in vocab_by_kebab :
for alias in vocab_by_kebab [ term_kebab ] . get ( " aliases " , [ ] ) :
alias_s = str ( alias ) . strip ( ) . lower ( )
if len ( alias_s ) > = 3 : # ignore very short aliases like IPs
patterns . append ( alias_s )
# Deduplicate
term_patterns [ term_kebab ] = list ( dict . fromkeys ( patterns ) )
mentions : dict [ str , list [ str ] ] = { t : [ ] for t in terms }
for md_path in vault_dir . rglob ( " *.md " ) :
if " archive " in md_path . parts :
continue
try :
raw = md_path . read_text ( errors = " replace " )
except Exception :
continue
# Strip frontmatter
fm_match = re . match ( r ' ^--- \ r? \ n.*? \ r? \ n--- \ r? \ n? ' , raw , re . DOTALL )
body = raw [ fm_match . end ( ) : ] if fm_match else raw
# Strip fenced code blocks
body_stripped = re . sub ( r ' ```.*?``` ' , ' ' , body , flags = re . DOTALL )
body_lower = body_stripped . lower ( )
rel = str ( md_path )
for term_kebab , patterns in term_patterns . items ( ) :
for pat in patterns :
if pat and pat in body_lower :
if rel not in mentions [ term_kebab ] :
mentions [ term_kebab ] . append ( rel )
break # one match per doc is enough
return mentions
# ---------------------------------------------------------------------------
# Ollama entity/concept extractor (v2 — NEW pipeline)
# ---------------------------------------------------------------------------
_EXTRACT_SYSTEM = """ You are the Echo6 vault entity/concept extractor. Given a markdown document and a reference list of known entity names + aliases, you extract structured metadata for building a knowledge graph. You output ONLY valid JSON matching the schema exactly. No prose, no markdown fences. """
def _build_extract_prompt ( doc_text : str , vocab : dict ) - > str :
"""
Build the extraction prompt with compact vocab reference and few - shot examples .
"""
# Compact entity reference: name (type) [aliases...]
entity_lines = [ ]
for ent in vocab . get ( " entities " , [ ] ) :
name = ent . get ( " name " , " " )
etype = ent . get ( " type " , " " )
aliases = [ str ( a ) for a in ent . get ( " aliases " , [ ] ) if not re . match ( r ' ^ \ d+ \ . \ d+ \ . \ d+ \ . \ d+ ' , str ( a ) ) and not re . match ( r ' ^100 \ . ' , str ( a ) ) ] [ : 3 ]
alias_str = f " [ { ' , ' . join ( aliases ) } ] " if aliases else " "
entity_lines . append ( f " - { name } ( { etype } ) { alias_str } " )
entity_block = " \n " . join ( entity_lines )
2026-06-18 06:00:06 +00:00
doc_truncated = doc_text [ : 6000 ]
if len ( doc_text ) > 6000 :
doc_truncated + = " \n \n [... truncated ...] "
2026-06-18 12:00:08 +00:00
return f """ { _EXTRACT_SYSTEM }
2026-06-18 06:00:06 +00:00
2026-06-18 12:00:08 +00:00
## Known entities (match canonical names where possible)
2026-06-18 06:00:06 +00:00
2026-06-18 12:00:08 +00:00
{ entity_block }
2026-06-18 06:00:06 +00:00
2026-06-18 12:00:08 +00:00
## Output schema
2026-06-18 06:00:06 +00:00
2026-06-18 12:00:08 +00:00
Return ONLY this JSON ( no fences , no extra keys ) :
{ {
" entities " : [ " canonical entity names from the list above that are genuinely present in this doc " ] ,
" concepts " : [ " specific technologies/protocols/patterns/hardware the doc is GENUINELY about — lowercase-kebab, new terms allowed " ] ,
" primary_topic " : " one broad topic string or null " ,
" confidence " : 0.0
} }
2026-06-18 06:00:06 +00:00
2026-06-18 12:00:08 +00:00
## Rules for concepts
2026-06-18 06:00:06 +00:00
2026-06-18 12:00:08 +00:00
- Concepts MUST be SPECIFIC and substantive : what is this doc actually about ?
- Good examples : raspberry - pi , meshtastic , lora , headscale , reverse - proxy , tls , dns , osint , searxng , docker , proxmox - lxc , tailscale , caddy , sso
- BAD examples ( reject these ) : infrastructure , configuration , setup , system , server , network , management , deployment , documentation
- Use lowercase - kebab . New concepts not in the entity list are allowed and encouraged .
- Do NOT conflate distinct systems : ARGUS is an OSINT platform ; RECON is its dashboard VM — they are separate .
- Limit to 3 - 8 concepts most central to the doc . Quality over quantity .
2026-06-18 06:00:06 +00:00
2026-06-18 12:00:08 +00:00
## Few-shot examples
2026-06-18 06:00:06 +00:00
2026-06-18 12:00:08 +00:00
Example 1 — meshtastic sidecar runbook :
entities : [ " mt-isr " , " aida-nebra " , " headscale " , " meshtasticd " , " meshtastic-cli " , " advbbs-project " , " meshmonitor " ]
concepts : [ " raspberry-pi " , " meshtastic " , " lora " , " headscale " , " tailscale " , " meshtasticd " , " systemd " ]
primary_topic : " mesh "
Example 2 — caddy & dns reference :
entities : [ " caddy " , " utility-caddy " , " headscale " , " headplane " , " mailcow " , " authentik " , " synapse " , " matrix-synapse " ]
concepts : [ " caddy " , " reverse-proxy " , " tls " , " dns " , " dnsmasq " , " split-dns " , " acme-sh " ]
primary_topic : " dns "
Example 3 — argus osint platform project :
entities : [ " argus " , " searxng " , " utility " , " cortex " ]
concepts : [ " osint " , " searxng " , " docker " , " proxmox-lxc " , " tailscale " , " lxc-container " , " intelligence-gathering " ]
primary_topic : " recon "
## Document to extract
{ doc_truncated }
Return ONLY the JSON object . """
def ollama_extract ( doc_text : str , vocab : dict , config : dict ) - > dict :
2026-06-18 05:49:07 +00:00
"""
2026-06-18 12:00:08 +00:00
POST to Ollama vault - tagger model with new entity / concept extraction prompt .
Returns dict : { { entities , concepts , primary_topic , confidence } } .
2026-06-18 06:00:06 +00:00
"""
tagger_cfg = config . get ( " models " , { } ) . get ( " tagger " , { } )
endpoint = tagger_cfg . get ( " ollama_endpoint " , " http://localhost:11434 " )
model = tagger_cfg . get ( " model " , " vault-tagger " )
temperature = tagger_cfg . get ( " temperature " , 0.1 )
2026-06-18 12:00:08 +00:00
prompt = _build_extract_prompt ( doc_text , vocab )
2026-06-18 06:00:06 +00:00
payload = {
" model " : model ,
" prompt " : prompt ,
" format " : " json " ,
" stream " : False ,
" options " : { " temperature " : temperature } ,
}
url = f " { endpoint } /api/generate "
try :
2026-06-18 12:00:08 +00:00
resp = _http_post ( url , payload , timeout = 180 )
2026-06-18 06:00:06 +00:00
except urllib . error . HTTPError as e :
raise RuntimeError ( f " Ollama HTTP { e . code } : { e . reason } " ) from e
except Exception as e :
raise RuntimeError ( f " Ollama request failed: { e } " ) from e
raw_response = resp . get ( " response " , " " ) if isinstance ( resp , dict ) else " "
try :
result = json . loads ( raw_response )
except json . JSONDecodeError :
m = re . search ( r ' \ { .* \ } ' , raw_response , re . DOTALL )
if m :
try :
result = json . loads ( m . group ( 0 ) )
except json . JSONDecodeError :
result = { }
else :
result = { }
2026-06-18 12:00:08 +00:00
# --- Validate + sanitize ---
# Generic concept blocklist (vague filler terms)
_CONCEPT_BLOCKLIST = {
" infrastructure " , " configuration " , " setup " , " system " , " server " ,
" network " , " management " , " deployment " , " documentation " , " service " ,
" installation " , " overview " , " reference " , " platform " , " solution " ,
" integration " , " process " , " feature " , " component " , " environment " ,
" hardware " , " software " , " security " , " monitoring " , " automation " ,
" container " , " services " ,
}
2026-06-18 06:00:06 +00:00
out : dict = {
" entities " : [ ] ,
2026-06-18 12:00:08 +00:00
" concepts " : [ ] ,
" primary_topic " : None ,
2026-06-18 06:00:06 +00:00
" confidence " : 0.0 ,
}
2026-06-18 12:00:08 +00:00
# Entities — validate against vocab names
valid_entity_names = { e . get ( " name " , " " ) . lower ( ) for e in vocab . get ( " entities " , [ ] ) }
if isinstance ( result . get ( " entities " ) , list ) :
for e in result [ " entities " ] :
if isinstance ( e , str ) and e . lower ( ) in valid_entity_names :
# Find canonical name (preserve original casing from vocab)
for ent in vocab . get ( " entities " , [ ] ) :
if ent . get ( " name " , " " ) . lower ( ) == e . lower ( ) :
if ent [ " name " ] not in out [ " entities " ] :
out [ " entities " ] . append ( ent [ " name " ] )
break
# Concepts — filter blocklist, validate format, deduplicate
_topic_cats = set ( vocab . get ( " topic_categories " , [ ] ) )
if isinstance ( result . get ( " concepts " ) , list ) :
for c in result [ " concepts " ] :
if not isinstance ( c , str ) :
continue
c_clean = to_kebab ( c . strip ( ) )
if not c_clean or len ( c_clean ) < 3 :
continue
if c_clean in _CONCEPT_BLOCKLIST :
continue
# Skip if it's just a topic_category name (redundant — goes in tags)
if c_clean in _topic_cats :
continue
# Also block if this is clearly a host IP or short abbreviation
if re . match ( r ' ^ \ d+ \ . \ d+ \ . \ d+ \ . \ d+$ ' , c_clean ) :
continue
if c_clean not in out [ " concepts " ] :
out [ " concepts " ] . append ( c_clean )
# Primary topic
if isinstance ( result . get ( " primary_topic " ) , str ) and result [ " primary_topic " ] . strip ( ) :
out [ " primary_topic " ] = result [ " primary_topic " ] . strip ( )
elif result . get ( " primary_topic " ) is None :
out [ " primary_topic " ] = None
# Confidence
try :
conf = float ( result . get ( " confidence " , 0.0 ) )
out [ " confidence " ] = max ( 0.0 , min ( 1.0 , conf ) )
except ( TypeError , ValueError ) :
out [ " confidence " ] = 0.0
return out
# ---------------------------------------------------------------------------
# Legacy ollama_tag — kept for backward compat with old CLI / sweep.sh
# ---------------------------------------------------------------------------
_SYSTEM_PROMPT = """ You are the Echo6 vault tagger. Given a markdown document and a controlled vocabulary (topic categories + an entity lexicon), you classify the document and extract structured metadata. You output ONLY valid JSON matching the requested schema. You never invent tags outside the provided vocabulary. """
def _build_vocab_block ( vocab : dict ) - > str :
cats = vocab . get ( " topic_categories " , [ ] )
entities = vocab . get ( " entities " , [ ] )
acronyms = vocab . get ( " acronyms " , [ ] )
lines = [ " ## Vocabulary " , " " , f " topic_categories (use ONLY these as tags): { ' , ' . join ( cats ) } " , " " , " ### Entities (names + aliases) " ]
for ent in entities :
aliases = ent . get ( " aliases " , [ ] )
alias_str = f " [ { ' , ' . join ( str ( a ) for a in aliases [ : 3 ] ) } ] " if aliases else " "
lines . append ( f " - { ent [ ' name ' ] } ( { ent [ ' type ' ] } ) { alias_str } " )
lines . extend ( [ " " , " ### Acronyms " ] )
for acro in acronyms :
exp = acro . get ( " expansion " ) or " unknown "
lines . append ( f " - { acro [ ' acronym ' ] } : { exp } " )
return " \n " . join ( lines )
def _build_prompt ( doc_text : str , vocab : dict ) - > str :
vocab_block = _build_vocab_block ( vocab )
doc_truncated = doc_text [ : 6000 ]
if len ( doc_text ) > 6000 :
doc_truncated + = " \n \n [... truncated ...] "
return f """ { _SYSTEM_PROMPT } \n \n { vocab_block } \n \n ## Document \n \n { doc_truncated } \n \n ## Task \n \n Classify this document and return ONLY a JSON object with these fields: \n - " type " : one of reference|runbook|project|note|index|session \n - " tags " : array of up to 3 strings, each MUST be in topic_categories; order most-relevant to least-relevant (primary topic first) \n - " entities " : array of entity names from the entity lexicon found in this doc \n - " glossary_proposals " : array of unknown acronyms/terms worth adding \n - " confidence " : float 0.0-1.0 \n \n Respond with ONLY the JSON. No prose, no markdown fences. """
def ollama_tag ( doc_text : str , vocab : dict , config : dict ) - > dict :
""" Legacy tagger — produces { type, tags, entities, glossary_proposals, confidence}. """
tagger_cfg = config . get ( " models " , { } ) . get ( " tagger " , { } )
endpoint = tagger_cfg . get ( " ollama_endpoint " , " http://localhost:11434 " )
model = tagger_cfg . get ( " model " , " vault-tagger " )
temperature = tagger_cfg . get ( " temperature " , 0.1 )
prompt = _build_prompt ( doc_text , vocab )
payload = { " model " : model , " prompt " : prompt , " format " : " json " , " stream " : False , " options " : { " temperature " : temperature } }
url = f " { endpoint } /api/generate "
try :
resp = _http_post ( url , payload , timeout = 120 )
except urllib . error . HTTPError as e :
raise RuntimeError ( f " Ollama HTTP { e . code } : { e . reason } " ) from e
except Exception as e :
raise RuntimeError ( f " Ollama request failed: { e } " ) from e
raw_response = resp . get ( " response " , " " ) if isinstance ( resp , dict ) else " "
try :
result = json . loads ( raw_response )
except json . JSONDecodeError :
m = re . search ( r ' \ { .* \ } ' , raw_response , re . DOTALL )
result = json . loads ( m . group ( 0 ) ) if m else { }
valid_types = { " reference " , " runbook " , " project " , " note " , " index " , " session " }
valid_cats = set ( vocab . get ( " topic_categories " , [ ] ) )
out : dict = { " type " : None , " tags " : [ ] , " entities " : [ ] , " glossary_proposals " : [ ] , " confidence " : 0.0 }
2026-06-18 06:00:06 +00:00
if isinstance ( result . get ( " type " ) , str ) and result [ " type " ] in valid_types :
out [ " type " ] = result [ " type " ]
if isinstance ( result . get ( " tags " ) , list ) :
out [ " tags " ] = [ t for t in result [ " tags " ] if isinstance ( t , str ) and t in valid_cats ]
if isinstance ( result . get ( " entities " ) , list ) :
out [ " entities " ] = [ e for e in result [ " entities " ] if isinstance ( e , str ) ]
if isinstance ( result . get ( " glossary_proposals " ) , list ) :
2026-06-18 12:00:08 +00:00
_GENERIC_ACRONYM_BLOCKLIST = { " MAS " , " TAK " , " OTS " , " OSINT " , " DEM " , " DM " , " E2EE " , " E2BE " , " OIDC " , " SAML " , " SSO " , " JWT " , " OAUTH " , " NVENC " , " DNAT " , " SNAT " , " ACL " , " CIDR " , " DHCP " , " NTP " , " SMTP " , " IMAP " , " LDAP " , " REST " , " GRPC " , " CORS " , " MQTT " , " RAID " , " ZFS " , " LVM " , " VLAN " , " NAT " , " WAF " , " CDN " , " UPS " , " PSU " , " ATAK " , " COT " , " RTT " , " TTL " , " MTU " , " RPC " , " SBC " , " PWA " }
_known_vocab_acronyms = { a [ " acronym " ] . upper ( ) for a in vocab . get ( " acronyms " , [ ] ) if isinstance ( a , dict ) }
_record_id_re = re . compile ( r ' ^[A-Za-z] \ d+$ ' )
filtered_proposals = [ ]
for p in result [ " glossary_proposals " ] :
if not isinstance ( p , str ) : continue
pu = p . upper ( ) . strip ( )
if len ( pu ) < 2 : continue
if pu in _known_vocab_acronyms : continue
if pu in _GENERIC_ACRONYM_BLOCKLIST : continue
if _record_id_re . match ( p . strip ( ) ) : continue
if p . strip ( ) == p . strip ( ) . lower ( ) : continue
filtered_proposals . append ( p )
out [ " glossary_proposals " ] = filtered_proposals
2026-06-18 06:00:06 +00:00
try :
conf = float ( result . get ( " confidence " , 0.0 ) )
out [ " confidence " ] = max ( 0.0 , min ( 1.0 , conf ) )
except ( TypeError , ValueError ) :
out [ " confidence " ] = 0.0
return out
# ---------------------------------------------------------------------------
# TEI embedding
# ---------------------------------------------------------------------------
def _detect_tei_route ( endpoint : str ) - > str :
for route in [ " /embed " , " /embeddings " ] :
try :
resp = _http_post ( f " { endpoint } { route } " , { " inputs " : " test " } , timeout = 10 )
if isinstance ( resp , list ) :
return route
except urllib . error . HTTPError as e :
if e . code not in ( 404 , 405 ) :
2026-06-18 12:00:08 +00:00
return route
2026-06-18 06:00:06 +00:00
except Exception :
continue
2026-06-18 12:00:08 +00:00
return " /embed "
2026-06-18 06:00:06 +00:00
_TEI_ROUTE_CACHE : dict [ str , str ] = { }
def tei_embed ( text : str , endpoint : str ) - > list [ float ] :
if endpoint not in _TEI_ROUTE_CACHE :
_TEI_ROUTE_CACHE [ endpoint ] = _detect_tei_route ( endpoint )
route = _TEI_ROUTE_CACHE [ endpoint ]
resp = _http_post ( f " { endpoint } { route } " , { " inputs " : text } , timeout = 30 )
if isinstance ( resp , list ) :
if resp and isinstance ( resp [ 0 ] , list ) :
2026-06-18 12:00:08 +00:00
return resp [ 0 ]
return resp
2026-06-18 06:00:06 +00:00
raise RuntimeError ( f " Unexpected TEI response type: { type ( resp ) } " )
# ---------------------------------------------------------------------------
# Embedding cache
# ---------------------------------------------------------------------------
def _cache_path ( engine_dir : Path ) - > Path :
return engine_dir / " .embcache.json "
def load_embed_cache ( engine_dir : Path ) - > dict :
p = _cache_path ( engine_dir )
if p . exists ( ) :
try :
with open ( p ) as f :
return json . load ( f )
except Exception :
return { }
return { }
def save_embed_cache ( cache : dict , engine_dir : Path ) - > None :
p = _cache_path ( engine_dir )
with open ( p , " w " ) as f :
json . dump ( cache , f )
2026-06-18 05:49:07 +00:00
2026-06-18 06:00:06 +00:00
def _embed_cache_key ( rel_path : str , mtime : float ) - > str :
return f " { rel_path } :: { mtime : .3f } "
2026-06-18 05:49:07 +00:00
2026-06-18 06:00:06 +00:00
def get_or_embed (
doc_path : Path ,
vault_dir : Path ,
tei_endpoint : str ,
cache : dict ,
doc_text : str | None = None ,
) - > tuple [ list [ float ] , bool ] :
rel = str ( doc_path . relative_to ( vault_dir ) )
mtime = doc_path . stat ( ) . st_mtime
key = _embed_cache_key ( rel , mtime )
if key in cache :
return cache [ key ] , True
if doc_text is None :
try :
doc_text = doc_path . read_text ( errors = " replace " )
except Exception as e :
raise RuntimeError ( f " Cannot read { doc_path } : { e } " ) from e
text_for_embed = doc_text [ : 4000 ]
vec = tei_embed ( text_for_embed , tei_endpoint )
cache [ key ] = vec
return vec , False
2026-06-18 05:49:07 +00:00
2026-06-18 06:00:06 +00:00
# ---------------------------------------------------------------------------
# Cosine similarity + related-doc finder
# ---------------------------------------------------------------------------
2026-06-18 05:49:07 +00:00
2026-06-18 06:00:06 +00:00
def cosine_sim ( a : list [ float ] , b : list [ float ] ) - > float :
if len ( a ) != len ( b ) :
return 0.0
dot = sum ( x * y for x , y in zip ( a , b ) )
mag_a = math . sqrt ( sum ( x * x for x in a ) )
mag_b = math . sqrt ( sum ( x * x for x in b ) )
if mag_a == 0 or mag_b == 0 :
return 0.0
return dot / ( mag_a * mag_b )
2026-06-18 05:49:07 +00:00
2026-06-18 06:00:06 +00:00
def find_related (
doc_path : Path ,
doc_vec : list [ float ] ,
vault_dir : Path ,
tei_endpoint : str ,
cache : dict ,
k : int = 5 ,
skip_archive : bool = True ,
) - > list [ str ] :
sims : list [ tuple [ float , str ] ] = [ ]
for md in vault_dir . rglob ( " *.md " ) :
if md == doc_path :
continue
if skip_archive and " archive " in md . parts :
continue
try :
vec , _ = get_or_embed ( md , vault_dir , tei_endpoint , cache )
except Exception :
continue
sim = cosine_sim ( doc_vec , vec )
sims . append ( ( sim , md . stem ) )
sims . sort ( reverse = True )
return [ f " [[ { stem } ]] " for _ , stem in sims [ : k ] ]
# ---------------------------------------------------------------------------
2026-06-18 12:00:08 +00:00
# Body wikilinking
2026-06-18 06:00:06 +00:00
# ---------------------------------------------------------------------------
2026-06-18 12:00:08 +00:00
def _build_skip_spans ( body : str ) - > list [ tuple [ int , int ] ] :
2026-06-18 06:00:06 +00:00
"""
2026-06-18 12:00:08 +00:00
Return list of ( start , end ) character spans that must NOT be modified :
- fenced code blocks ( ` ` ` . . . ` ` ` )
- inline code ( ` . . . ` )
- existing [ [ wikilinks ] ]
- markdown headings ( lines starting with #)
- URLs ( http / https / . . . )
2026-06-18 06:00:06 +00:00
"""
2026-06-18 12:00:08 +00:00
spans : list [ tuple [ int , int ] ] = [ ]
# Fenced code blocks
for m in re . finditer ( r ' ```.*?``` ' , body , re . DOTALL ) :
spans . append ( ( m . start ( ) , m . end ( ) ) )
# Inline code
for m in re . finditer ( r ' `[^` \ n]+` ' , body ) :
spans . append ( ( m . start ( ) , m . end ( ) ) )
# Existing wikilinks
for m in re . finditer ( r ' \ [ \ [.*? \ ] \ ] ' , body ) :
spans . append ( ( m . start ( ) , m . end ( ) ) )
# Headings (whole line)
for m in re . finditer ( r ' ^#+ \ s+.*$ ' , body , re . MULTILINE ) :
spans . append ( ( m . start ( ) , m . end ( ) ) )
# URLs
for m in re . finditer ( r ' https?:// \ S+ ' , body ) :
spans . append ( ( m . start ( ) , m . end ( ) ) )
# Domain-like patterns: word followed immediately by .echo6 / .com / .co / .org / etc.
# This prevents "mesh" from being linked in "mesh.echo6.co"
for m in re . finditer ( r ' \ b \ w[ \ w \ -]* \ . \ w+ \ b ' , body ) :
spans . append ( ( m . start ( ) , m . end ( ) ) )
return spans
def _in_skip_span ( pos : int , end : int , skip_spans : list [ tuple [ int , int ] ] ) - > bool :
""" Return True if [pos, end) overlaps any skip span. """
for s , e in skip_spans :
if pos < e and end > s :
return True
return False
def apply_wikilinks_to_body (
body : str ,
terms : list [ tuple [ str , str ] ] , # [(surface_text_to_match, canonical_display_name), ...]
) - > str :
"""
For each ( surface , canonical ) pair , find the FIRST prose occurrence of surface
( case - insensitive , word - boundary ) and replace it with [ [ Canonical Name | surface ] ]
( or [ [ Canonical Name ] ] if surface matches canonical case - insensitively ) .
Rules :
- Skip matches inside fenced code , inline code , headings , URLs , existing [ [ . . . ] ]
- Only FIRST occurrence per term
- Word - boundary match
- Idempotent ( won ' t double-link)
"""
skip_spans = _build_skip_spans ( body )
result = body
# Sort terms longest-first to avoid partial-match conflicts
sorted_terms = sorted ( terms , key = lambda t : len ( t [ 0 ] ) , reverse = True )
offset = 0 # cumulative offset as we insert characters
# We'll collect replacements as (original_pos, original_end, replacement_str)
# then apply them in order; recalculate skip_spans after each
replacements : list [ tuple [ int , int , str ] ] = [ ]
replaced_terms : set [ str ] = set ( )
for surface , canonical in sorted_terms :
if surface . lower ( ) in replaced_terms :
continue
# Build word-boundary regex
# Escape the surface text, allow spaces-or-hyphens to match either
esc = re . escape ( surface ) . replace ( r ' \ ' , r ' [ \ s \ -] ' ) . replace ( r ' \ - ' , r ' [ \ s \ -] ' )
# Negative lookahead includes '.' to avoid matching inside domain names
pattern = rf ' (?<![a-zA-Z0-9 \ -_ \ [ \ ]])( { esc } )(?![a-zA-Z0-9 \ -_ \ [ \ ] \ .]) '
# Search in result (which accumulates changes) — but track against ORIGINAL spans
# Simpler approach: search in body (original), check against skip_spans
for m in re . finditer ( pattern , body , re . IGNORECASE ) :
matched_text = m . group ( 1 )
start , end = m . start ( 1 ) , m . end ( 1 )
if _in_skip_span ( start , end , skip_spans ) :
continue
# Build replacement
canonical_display = canonical
# Use [[Canonical|surface]] only if surface differs from canonical (case-insensitive)
if matched_text . lower ( ) == canonical_display . lower ( ) :
wiki = f " [[ { canonical_display } ]] "
else :
wiki = f " [[ { canonical_display } | { matched_text } ]] "
replacements . append ( ( start , end , wiki ) )
replaced_terms . add ( surface . lower ( ) )
# Mark the replacement span as a skip span to prevent overlap
skip_spans . append ( ( start , end ) )
break # only first occurrence
# Apply replacements in reverse order (so offsets don't shift)
replacements . sort ( key = lambda x : x [ 0 ] , reverse = True )
result_chars = list ( body )
for start , end , wiki in replacements :
result_chars [ start : end ] = list ( wiki )
return " " . join ( result_chars )
2026-06-18 06:00:06 +00:00
# ---------------------------------------------------------------------------
# Type inference from folder
# ---------------------------------------------------------------------------
def infer_type_from_path ( doc_path : Path , vault_dir : Path ) - > str | None :
try :
rel = doc_path . relative_to ( vault_dir )
except ValueError :
return None
parts = rel . parts
if not parts :
return None
folder = parts [ 0 ] . lower ( ) if len ( parts ) > 1 else " "
filename = doc_path . stem . lower ( )
2026-06-18 12:00:08 +00:00
if filename == " index " :
2026-06-18 06:00:06 +00:00
return " index "
folder_map = {
" runbooks " : " runbook " ,
" projects " : " project " ,
" docs " : " reference " ,
" notes " : " note " ,
" session-resume " : " session " ,
}
2026-06-18 12:00:08 +00:00
result = folder_map . get ( folder )
if result is not None :
return result
if len ( parts ) == 1 and folder == " " :
return " reference "
return None
2026-06-18 06:00:06 +00:00
# ---------------------------------------------------------------------------
# Title extraction
# ---------------------------------------------------------------------------
def extract_title ( fm : dict , body : str , doc_path : Path ) - > str :
if fm . get ( " title " ) :
return str ( fm [ " title " ] )
m = re . search ( r ' ^# \ s+(.+) ' , body , re . MULTILINE )
if m :
return m . group ( 1 ) . strip ( )
stem = doc_path . stem
return re . sub ( r ' [-_]+ ' , ' ' , stem ) . title ( )
# ---------------------------------------------------------------------------
2026-06-18 12:00:08 +00:00
# Core process_doc (v2 — entity/concept pipeline)
2026-06-18 06:00:06 +00:00
# ---------------------------------------------------------------------------
def process_doc (
doc_path : Path | str ,
config : dict ,
vocab : dict ,
dry_run : bool = True ,
cache : dict | None = None ,
) - > dict :
"""
2026-06-18 12:00:08 +00:00
Process a single vault document through the new entity / concept pipeline .
Steps :
1. Extract entities + concepts via Qwen ( ollama_extract )
2. Canonicalize each term
3. Build corpus mention index → page - mint decisions
4. Body wikilinking ( dry - run : diff only )
5. related : bge - m3 nearest docs
6. tags : demote to single primary_topic
Returns a result dict . In dry_run = True , nothing is written .
2026-06-18 06:00:06 +00:00
"""
doc_path = Path ( doc_path ) . resolve ( )
vault_dir = Path ( config [ " vault_dir " ] ) . resolve ( )
engine_dir = Path ( config [ " engine_dir " ] ) . resolve ( )
tei_endpoint = config [ " models " ] [ " embeddings " ] [ " tei_endpoint " ]
threshold = config [ " behavior " ] [ " confidence_threshold " ]
if cache is None :
cache = load_embed_cache ( engine_dir )
# --- Read doc ---
try :
doc_text = doc_path . read_text ( errors = " replace " )
except Exception as e :
raise RuntimeError ( f " Cannot read { doc_path } : { e } " ) from e
existing_fm , body = parse_frontmatter ( doc_text )
2026-06-18 12:00:08 +00:00
# --- Step 1: Extract entities + concepts ---
print ( f " [extract] Calling vault-tagger on { doc_path . name } ... " , file = sys . stderr )
extract_result = ollama_extract ( doc_text , vocab , config )
print (
f " [extract] entities= { extract_result [ ' entities ' ] } "
f " concepts= { extract_result [ ' concepts ' ] } "
f " primary_topic= { extract_result [ ' primary_topic ' ] } "
f " confidence= { extract_result [ ' confidence ' ] : .2f } " ,
file = sys . stderr ,
)
2026-06-18 06:00:06 +00:00
2026-06-18 12:00:08 +00:00
# --- Step 2: Canonicalize each entity/concept ---
existing_pages = build_existing_pages_cache ( vault_dir )
all_terms : list [ tuple [ str , str , str , str ] ] = [ ] # (raw, display, kebab, match_type)
for ent_name in extract_result [ " entities " ] :
display , kebab , mtype = canonicalize_term ( ent_name , vocab , vault_dir , existing_pages )
all_terms . append ( ( ent_name , display , kebab , mtype ) )
for concept in extract_result [ " concepts " ] :
display , kebab , mtype = canonicalize_term ( concept , vocab , vault_dir , existing_pages )
all_terms . append ( ( concept , display , kebab , mtype ) )
# Deduplicate by kebab
seen_kebabs : set [ str ] = set ( )
unique_terms : list [ tuple [ str , str , str , str ] ] = [ ]
for raw , display , kebab , mtype in all_terms :
if kebab not in seen_kebabs :
seen_kebabs . add ( kebab )
unique_terms . append ( ( raw , display , kebab , mtype ) )
# --- Step 3: Corpus mention index + page-mint decisions ---
all_kebabs = [ t [ 2 ] for t in unique_terms ]
print ( f " [corpus] Scanning vault for { len ( all_kebabs ) } terms ... " , file = sys . stderr )
mention_index = build_corpus_mention_index ( vault_dir , all_kebabs , vocab )
# A term qualifies for a page if:
# - it's in vocab (entities) → vault/entities/<kebab>.md
# - OR concept mentioned in ≥2 docs → vault/concepts/<kebab>.md
vocab_entity_names = { e . get ( " name " , " " ) for e in vocab . get ( " entities " , [ ] ) }
mint_decisions : dict [ str , dict ] = { } # kebab → {display, page_path, reason, already_exists}
for raw , display , kebab , match_type in unique_terms :
is_vocab = ( raw in vocab_entity_names ) or ( match_type in ( ' vocab-entity ' , ' vocab-alias ' ) )
mention_count = len ( mention_index . get ( kebab , [ ] ) )
already_exists = kebab in existing_pages
if already_exists :
mint_decisions [ kebab ] = {
" display " : display , " already_exists " : True ,
" page_dir " : " existing " , " reason " : f " page already exists in vault " ,
" mention_count " : mention_count ,
}
elif is_vocab :
mint_decisions [ kebab ] = {
" display " : display , " already_exists " : False ,
" page_dir " : " entities " , " reason " : " vocab entity " ,
" mention_count " : mention_count ,
}
elif mention_count > = 2 :
mint_decisions [ kebab ] = {
" display " : display , " already_exists " : False ,
" page_dir " : " concepts " , " reason " : f " mentioned in { mention_count } docs " ,
" mention_count " : mention_count ,
}
else :
mint_decisions [ kebab ] = {
" display " : display , " already_exists " : False ,
" page_dir " : None ,
" reason " : f " only { mention_count } corpus mention(s) — below threshold " ,
" mention_count " : mention_count ,
}
# --- Step 4: Body wikilinking ---
# Build list of (surface_text, canonical_display) for terms that qualify for a page
wikilink_terms : list [ tuple [ str , str ] ] = [ ]
for raw , display , kebab , match_type in unique_terms :
decision = mint_decisions . get ( kebab , { } )
if decision . get ( " page_dir " ) is not None or decision . get ( " already_exists " ) :
# Use the raw name as surface text (what actually appears in the doc),
# canonical display as link target
wikilink_terms . append ( ( raw , display ) )
# Also add kebab variant and display variant as additional surfaces
# (so "raspberry-pi" links "Raspberry Pi" text too)
if display . lower ( ) != raw . lower ( ) :
wikilink_terms . append ( ( display , display ) )
# Add space variant of kebab
space_variant = kebab . replace ( ' - ' , ' ' )
if space_variant . lower ( ) not in { raw . lower ( ) , display . lower ( ) } :
wikilink_terms . append ( ( space_variant , display ) )
# Exclude self-referential links (term matches this doc's own stem/title)
doc_stem_kebab = to_kebab ( doc_path . stem )
doc_title_kebab = to_kebab ( str ( existing_fm . get ( " title " , doc_path . stem ) ) )
wikilink_terms_filtered = [
( surface , canonical )
for surface , canonical in wikilink_terms
if to_kebab ( surface ) not in ( doc_stem_kebab , doc_title_kebab )
and to_kebab ( canonical ) not in ( doc_stem_kebab , doc_title_kebab )
]
linked_body = apply_wikilinks_to_body ( body , wikilink_terms_filtered )
body_diff = " " . join ( difflib . unified_diff (
body . splitlines ( keepends = True ) ,
linked_body . splitlines ( keepends = True ) ,
fromfile = f " a/ { doc_path . name } (body — { len ( wikilink_terms_filtered ) } link candidates) " ,
tofile = f " b/ { doc_path . name } (body+wikilinks) " ,
n = 2 ,
) )
# --- Step 5: related links (bge-m3) ---
2026-06-18 06:00:06 +00:00
print ( f " [embed] Embedding { doc_path . name } ... " , file = sys . stderr )
2026-06-18 12:00:08 +00:00
related_links : list [ str ] = [ ]
doc_vec : list [ float ] = [ ]
2026-06-18 06:00:06 +00:00
try :
doc_vec , cache_hit = get_or_embed ( doc_path , vault_dir , tei_endpoint , cache , doc_text = doc_text )
if not cache_hit :
save_embed_cache ( cache , engine_dir )
2026-06-18 12:00:08 +00:00
print ( f " [embed] dim= { len ( doc_vec ) } cache_hit= { cache_hit } " , file = sys . stderr )
print ( f " [embed] Finding related docs ... " , file = sys . stderr )
2026-06-18 06:00:06 +00:00
related_links = find_related ( doc_path , doc_vec , vault_dir , tei_endpoint , cache , k = 5 )
2026-06-18 12:00:08 +00:00
save_embed_cache ( cache , engine_dir )
print ( f " [embed] Found { len ( related_links ) } related " , file = sys . stderr )
2026-06-18 06:00:06 +00:00
except Exception as e :
print ( f " [warn] Embedding/related failed: { e } " , file = sys . stderr )
2026-06-18 12:00:08 +00:00
# --- Step 6: Build new frontmatter ---
today = datetime . now ( timezone . utc ) . strftime ( " % Y- % m- %d " )
valid_types = set ( config . get ( " frontmatter_schema " , { } ) . get ( " types " , [ " reference " , " runbook " , " project " , " note " , " index " , " session " ] ) )
2026-06-18 06:00:06 +00:00
2026-06-18 12:00:08 +00:00
new_fm = dict ( existing_fm )
2026-06-18 06:00:06 +00:00
2026-06-18 12:00:08 +00:00
# title: existing wins
if not new_fm . get ( " title " ) :
new_fm [ " title " ] = extract_title ( existing_fm , body , doc_path )
# type: folder inference authoritative
inferred_type = infer_type_from_path ( doc_path , vault_dir )
if inferred_type :
new_fm [ " type " ] = inferred_type
elif str ( new_fm . get ( " type " , " " ) ) . strip ( ) not in valid_types :
new_fm [ " type " ] = " reference "
# tags: DEMOTED to single primary_topic (or none)
primary = extract_result . get ( " primary_topic " )
valid_cats = set ( vocab . get ( " topic_categories " , [ ] ) )
if primary and primary in valid_cats :
new_fm [ " tags " ] = [ primary ]
elif primary :
# primary_topic not in controlled vocab — store it but warn
print ( f " [warn] primary_topic ' { primary } ' not in topic_categories " , file = sys . stderr )
new_fm [ " tags " ] = [ ]
else :
new_fm [ " tags " ] = [ ]
# aliases: keep existing
if " aliases " not in new_fm :
new_fm [ " aliases " ] = [ ]
# related: embedding-derived links (only replace if missing or empty)
if related_links and not new_fm . get ( " related " ) :
new_fm [ " related " ] = related_links
elif related_links and new_fm . get ( " related " ) :
existing_related = list ( new_fm [ " related " ] )
for link in related_links :
if link not in existing_related :
existing_related . append ( link )
new_fm [ " related " ] = existing_related [ : 5 ]
2026-06-18 06:00:06 +00:00
2026-06-18 12:00:08 +00:00
# updated: always refresh
new_fm [ " updated " ] = today
# --- Build new doc text (FM + linked body) ---
new_fm_text = render_frontmatter ( new_fm )
new_doc_text = new_fm_text + linked_body
# Full doc diff (frontmatter + body)
full_diff = " " . join ( difflib . unified_diff (
doc_text . splitlines ( keepends = True ) ,
new_doc_text . splitlines ( keepends = True ) ,
fromfile = f " a/ { doc_path . name } " ,
tofile = f " b/ { doc_path . name } " ,
n = 2 ,
) )
# --- Apply (only if not dry_run) ---
2026-06-18 06:00:06 +00:00
applied = False
if not dry_run :
auto_apply = config [ " behavior " ] . get ( " auto_apply " , True )
2026-06-18 12:00:08 +00:00
if auto_apply and extract_result [ " confidence " ] > = threshold :
2026-06-18 06:00:06 +00:00
doc_path . write_text ( new_doc_text )
applied = True
print ( f " [apply] Wrote { doc_path } " , file = sys . stderr )
else :
2026-06-18 12:00:08 +00:00
print ( f " [skip] confidence= { extract_result [ ' confidence ' ] : .2f } < threshold= { threshold } " , file = sys . stderr )
2026-06-18 06:00:06 +00:00
return {
" path " : str ( doc_path ) ,
2026-06-18 12:00:08 +00:00
" extract_result " : extract_result ,
" all_terms " : unique_terms , # [(raw, display, kebab, match_type)]
" mint_decisions " : mint_decisions , # {kebab: {display, already_exists, page_dir, reason, mention_count}}
" wikilink_terms " : wikilink_terms ,
" body_diff " : body_diff ,
2026-06-18 06:00:06 +00:00
" related_links " : related_links ,
" proposed_frontmatter " : new_fm_text ,
2026-06-18 12:00:08 +00:00
" full_diff " : full_diff ,
2026-06-18 06:00:06 +00:00
" applied " : applied ,
" dry_run " : dry_run ,
" existing_fm " : existing_fm ,
" new_fm " : new_fm ,
}
# ---------------------------------------------------------------------------
# CLI
# ---------------------------------------------------------------------------
2026-06-18 05:49:07 +00:00
def main ( ) - > None :
2026-06-18 06:00:06 +00:00
parser = argparse . ArgumentParser (
2026-06-18 12:00:08 +00:00
description = " Vault entity/concept extractor — build graph via wikilinks + pages. "
2026-06-18 06:00:06 +00:00
)
parser . add_argument ( " path " , help = " Path to the vault markdown document " )
parser . add_argument (
" --dry-run " ,
action = " store_true " ,
2026-06-18 12:00:08 +00:00
help = " Print extraction, mint decisions, body diff — write nothing " ,
2026-06-18 06:00:06 +00:00
)
args = parser . parse_args ( )
here = Path ( __file__ ) . parent
config_path = here . parent / " config.yaml "
if not config_path . exists ( ) :
config_path = Path ( " /home/zvx/projects/.ref/engine/config.yaml " )
config = load_config ( config_path )
vocab = load_vocab ( config [ " engine_dir " ] )
doc_path = Path ( args . path ) . resolve ( )
if not doc_path . exists ( ) :
print ( f " [error] File not found: { doc_path } " , file = sys . stderr )
sys . exit ( 1 )
result = process_doc ( doc_path , config , vocab , dry_run = args . dry_run )
2026-06-18 12:00:08 +00:00
er = result [ " extract_result " ]
print ( " \n " + " = " * 72 )
print ( f " ENTITY/CONCEPT EXTRACTOR { ' DRY-RUN ' if args . dry_run else ' ' } REPORT " )
print ( f " Doc: { doc_path . name } " )
print ( " = " * 72 )
print ( " \n ### 1. Extracted entities + concepts " )
print ( f " entities : { er [ ' entities ' ] } " )
print ( f " concepts : { er [ ' concepts ' ] } " )
print ( f " primary_topic: { er [ ' primary_topic ' ] !r} " )
print ( f " confidence : { er [ ' confidence ' ] : .2f } " )
print ( " \n ### 2. Page-mint decisions " )
entities_to_mint = [ ]
concepts_to_mint = [ ]
already_exist = [ ]
below_threshold = [ ]
for kebab , info in result [ " mint_decisions " ] . items ( ) :
disp = info [ " display " ]
pdir = info [ " page_dir " ]
reason = info [ " reason " ]
exists = info [ " already_exists " ]
if exists :
already_exist . append ( f " [EXISTS ] { disp } ( { kebab } ) — { reason } " )
elif pdir == " entities " :
entities_to_mint . append ( f " [MINT→entities/] { disp } ( { kebab } ) — { reason } " )
elif pdir == " concepts " :
concepts_to_mint . append ( f " [MINT→concepts/] { disp } ( { kebab } ) — { reason } " )
else :
below_threshold . append ( f " [SKIP ] { disp } ( { kebab } ) — { reason } " )
if entities_to_mint :
print ( " --- Would mint as vault/entities/ --- " )
for line in entities_to_mint :
print ( line )
if concepts_to_mint :
print ( " --- Would mint as vault/concepts/ --- " )
for line in concepts_to_mint :
print ( line )
if already_exist :
print ( " --- Already has a page --- " )
for line in already_exist :
print ( line )
if below_threshold :
print ( " --- Below threshold (no page) --- " )
for line in below_threshold :
print ( line )
print ( " \n ### 3. Body wikilink diff " )
if result [ " body_diff " ] :
print ( result [ " body_diff " ] )
2026-06-18 06:00:06 +00:00
else :
2026-06-18 12:00:08 +00:00
print ( " (no changes — no qualifying terms found in prose) " )
print ( " \n ### 4. Proposed related links (bge-m3 top-5) " )
2026-06-18 06:00:06 +00:00
if result [ " related_links " ] :
for link in result [ " related_links " ] :
print ( f " { link } " )
else :
print ( " (none — embedding not available) " )
2026-06-18 12:00:08 +00:00
print ( " \n ### 5. Proposed frontmatter " )
2026-06-18 06:00:06 +00:00
print ( result [ " proposed_frontmatter " ] )
2026-06-18 12:00:08 +00:00
2026-06-18 06:00:06 +00:00
if args . dry_run :
2026-06-18 12:00:08 +00:00
print ( " \n [DRY-RUN] Nothing written. git status should show only engine/lib/agent.py modified. " )
2026-06-18 06:00:06 +00:00
else :
2026-06-18 12:00:08 +00:00
status = " applied " if result [ " applied " ] else " not applied (below threshold) "
2026-06-18 06:00:06 +00:00
print ( f " \n [APPLY] Status: { status } " )
2026-06-18 05:49:07 +00:00
if __name__ == " __main__ " :
main ( )