fix(skillgen): stamp only produced-output files in the update runbook (#2015)
The --update runbook's Step 9 stamped the entire detected corpus into the manifest, so a semantic file (doc/paper/image) whose chunk failed or was omitted was recorded as done and never re-queued on the next update — losing its content permanently. The runbook now builds the manifest the way the library extract path does: cli._stamped_manifest_files stamps only files that actually produced nodes/edges/hyperedges, dispatched-but-empty files have their stale hash cleared, and scan_corpus drops newly-excluded in-root rows. Applied to the Claude, Aider, and Devin skill bodies and the shared update reference; all 134 artifacts regenerated. gen.py gains a sanctioned monolith- diff predicate for the new stamping lines.
This commit is contained in:
@@ -551,15 +551,37 @@ from graphify.detect import save_manifest
|
||||
|
||||
# Save manifest for --update
|
||||
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\"))
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
# In --update mode, 'all_files' carries the full corpus; 'files' is the changed
|
||||
# subset. Full-rebuild mode populates only 'files', so the fallback handles that.
|
||||
# root= relativizes the manifest keys to the scan root (same base as the build),
|
||||
# so the on-disk manifest is portable across clones/machines and a later --update
|
||||
# matches cached files instead of missing every one (#1417).
|
||||
save_manifest(detect.get('all_files') or detect['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output:
|
||||
# a detected file whose chunk failed or was omitted must stay unstamped so the
|
||||
# next --update re-queues it, otherwise it is marked done and its content is lost
|
||||
# forever (#2015). This mirrors the library extract path exactly
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus); do not stamp the
|
||||
# raw corpus. Code files are always stamped (AST is deterministic); only semantic
|
||||
# types are gated on output.
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_corpus = detect.get('all_files') or detect['files']
|
||||
_manifest_files = _stamped_manifest_files(_corpus, extract, Path('INPUT_PATH'))
|
||||
# Files dispatched this run (the changed subset) but NOT stamped above still carry
|
||||
# a stale semantic_hash from a prior run; clear it so detect_incremental re-queues
|
||||
# them instead of reading them as unchanged (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in detect['files'].items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus (not the stamp-filtered subset) so in-root
|
||||
# files newly excluded since last run are dropped rather than masquerading as
|
||||
# deletions; untouched files' prior rows are still preserved (#1908).
|
||||
_scan = {f for fl in _corpus.values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
|
||||
# Update cumulative cost tracker
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
input_tok = extract.get('input_tokens', 0)
|
||||
output_tok = extract.get('output_tokens', 0)
|
||||
|
||||
|
||||
+11
-2
@@ -677,10 +677,19 @@ from graphify.detect import save_manifest
|
||||
|
||||
# Save manifest for --update
|
||||
detect = json.loads(Path('.graphify_detect.json').read_text())
|
||||
save_manifest(detect['files'], root='INPUT_PATH')
|
||||
extract = json.loads(Path('.graphify_extract.json').read_text())
|
||||
# Stamp only semantic files that produced output so a failed chunk is re-queued next run, not lost (#2015).
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_corpus = detect.get('all_files') or detect['files']
|
||||
_manifest_files = _stamped_manifest_files(_corpus, extract, Path('INPUT_PATH'))
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in detect['files'].items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
_scan = {f for fl in _corpus.values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
|
||||
# Update cumulative cost tracker
|
||||
extract = json.loads(Path('.graphify_extract.json').read_text())
|
||||
input_tok = extract.get('input_tokens', 0)
|
||||
output_tok = extract.get('output_tokens', 0)
|
||||
|
||||
|
||||
+24
-2
@@ -551,15 +551,37 @@ from graphify.detect import save_manifest
|
||||
|
||||
# Save manifest for --update
|
||||
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\"))
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
# In --update mode, 'all_files' carries the full corpus; 'files' is the changed
|
||||
# subset. Full-rebuild mode populates only 'files', so the fallback handles that.
|
||||
# root= relativizes the manifest keys to the scan root (same base as the build),
|
||||
# so the on-disk manifest is portable across clones/machines and a later --update
|
||||
# matches cached files instead of missing every one (#1417).
|
||||
save_manifest(detect.get('all_files') or detect['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output:
|
||||
# a detected file whose chunk failed or was omitted must stay unstamped so the
|
||||
# next --update re-queues it, otherwise it is marked done and its content is lost
|
||||
# forever (#2015). This mirrors the library extract path exactly
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus); do not stamp the
|
||||
# raw corpus. Code files are always stamped (AST is deterministic); only semantic
|
||||
# types are gated on output.
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_corpus = detect.get('all_files') or detect['files']
|
||||
_manifest_files = _stamped_manifest_files(_corpus, extract, Path('INPUT_PATH'))
|
||||
# Files dispatched this run (the changed subset) but NOT stamped above still carry
|
||||
# a stale semantic_hash from a prior run; clear it so detect_incremental re-queues
|
||||
# them instead of reading them as unchanged (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in detect['files'].items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus (not the stamp-filtered subset) so in-root
|
||||
# files newly excluded since last run are dropped rather than masquerading as
|
||||
# deletions; untouched files' prior rows are still preserved (#1908).
|
||||
_scan = {f for fl in _corpus.values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
|
||||
# Update cumulative cost tracker
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
input_tok = extract.get('input_tokens', 0)
|
||||
output_tok = extract.get('output_tokens', 0)
|
||||
|
||||
|
||||
+24
-2
@@ -554,15 +554,37 @@ from graphify.detect import save_manifest
|
||||
|
||||
# Save manifest for --update
|
||||
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\"))
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
# In --update mode, 'all_files' carries the full corpus; 'files' is the changed
|
||||
# subset. Full-rebuild mode populates only 'files', so the fallback handles that.
|
||||
# root= relativizes the manifest keys to the scan root (same base as the build),
|
||||
# so the on-disk manifest is portable across clones/machines and a later --update
|
||||
# matches cached files instead of missing every one (#1417).
|
||||
save_manifest(detect.get('all_files') or detect['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output:
|
||||
# a detected file whose chunk failed or was omitted must stay unstamped so the
|
||||
# next --update re-queues it, otherwise it is marked done and its content is lost
|
||||
# forever (#2015). This mirrors the library extract path exactly
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus); do not stamp the
|
||||
# raw corpus. Code files are always stamped (AST is deterministic); only semantic
|
||||
# types are gated on output.
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_corpus = detect.get('all_files') or detect['files']
|
||||
_manifest_files = _stamped_manifest_files(_corpus, extract, Path('INPUT_PATH'))
|
||||
# Files dispatched this run (the changed subset) but NOT stamped above still carry
|
||||
# a stale semantic_hash from a prior run; clear it so detect_incremental re-queues
|
||||
# them instead of reading them as unchanged (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in detect['files'].items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus (not the stamp-filtered subset) so in-root
|
||||
# files newly excluded since last run are dropped rather than masquerading as
|
||||
# deletions; untouched files' prior rows are still preserved (#1908).
|
||||
_scan = {f for fl in _corpus.values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
|
||||
# Update cumulative cost tracker
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
input_tok = extract.get('input_tokens', 0)
|
||||
output_tok = extract.get('output_tokens', 0)
|
||||
|
||||
|
||||
+24
-2
@@ -551,15 +551,37 @@ from graphify.detect import save_manifest
|
||||
|
||||
# Save manifest for --update
|
||||
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\"))
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
# In --update mode, 'all_files' carries the full corpus; 'files' is the changed
|
||||
# subset. Full-rebuild mode populates only 'files', so the fallback handles that.
|
||||
# root= relativizes the manifest keys to the scan root (same base as the build),
|
||||
# so the on-disk manifest is portable across clones/machines and a later --update
|
||||
# matches cached files instead of missing every one (#1417).
|
||||
save_manifest(detect.get('all_files') or detect['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output:
|
||||
# a detected file whose chunk failed or was omitted must stay unstamped so the
|
||||
# next --update re-queues it, otherwise it is marked done and its content is lost
|
||||
# forever (#2015). This mirrors the library extract path exactly
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus); do not stamp the
|
||||
# raw corpus. Code files are always stamped (AST is deterministic); only semantic
|
||||
# types are gated on output.
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_corpus = detect.get('all_files') or detect['files']
|
||||
_manifest_files = _stamped_manifest_files(_corpus, extract, Path('INPUT_PATH'))
|
||||
# Files dispatched this run (the changed subset) but NOT stamped above still carry
|
||||
# a stale semantic_hash from a prior run; clear it so detect_incremental re-queues
|
||||
# them instead of reading them as unchanged (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in detect['files'].items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus (not the stamp-filtered subset) so in-root
|
||||
# files newly excluded since last run are dropped rather than masquerading as
|
||||
# deletions; untouched files' prior rows are still preserved (#1908).
|
||||
_scan = {f for fl in _corpus.values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
|
||||
# Update cumulative cost tracker
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
input_tok = extract.get('input_tokens', 0)
|
||||
output_tok = extract.get('output_tokens', 0)
|
||||
|
||||
|
||||
@@ -554,15 +554,37 @@ from graphify.detect import save_manifest
|
||||
|
||||
# Save manifest for --update
|
||||
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\"))
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
# In --update mode, 'all_files' carries the full corpus; 'files' is the changed
|
||||
# subset. Full-rebuild mode populates only 'files', so the fallback handles that.
|
||||
# root= relativizes the manifest keys to the scan root (same base as the build),
|
||||
# so the on-disk manifest is portable across clones/machines and a later --update
|
||||
# matches cached files instead of missing every one (#1417).
|
||||
save_manifest(detect.get('all_files') or detect['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output:
|
||||
# a detected file whose chunk failed or was omitted must stay unstamped so the
|
||||
# next --update re-queues it, otherwise it is marked done and its content is lost
|
||||
# forever (#2015). This mirrors the library extract path exactly
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus); do not stamp the
|
||||
# raw corpus. Code files are always stamped (AST is deterministic); only semantic
|
||||
# types are gated on output.
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_corpus = detect.get('all_files') or detect['files']
|
||||
_manifest_files = _stamped_manifest_files(_corpus, extract, Path('INPUT_PATH'))
|
||||
# Files dispatched this run (the changed subset) but NOT stamped above still carry
|
||||
# a stale semantic_hash from a prior run; clear it so detect_incremental re-queues
|
||||
# them instead of reading them as unchanged (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in detect['files'].items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus (not the stamp-filtered subset) so in-root
|
||||
# files newly excluded since last run are dropped rather than masquerading as
|
||||
# deletions; untouched files' prior rows are still preserved (#1908).
|
||||
_scan = {f for fl in _corpus.values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
|
||||
# Update cumulative cost tracker
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
input_tok = extract.get('input_tokens', 0)
|
||||
output_tok = extract.get('output_tokens', 0)
|
||||
|
||||
|
||||
+11
-2
@@ -795,10 +795,19 @@ from graphify.detect import save_manifest
|
||||
|
||||
# Save manifest for --update
|
||||
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text())
|
||||
save_manifest(detect['files'], root='INPUT_PATH')
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text())
|
||||
# Stamp only semantic files that produced output so a failed chunk is re-queued next run, not lost (#2015).
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_corpus = detect.get('all_files') or detect['files']
|
||||
_manifest_files = _stamped_manifest_files(_corpus, extract, Path('INPUT_PATH'))
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in detect['files'].items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
_scan = {f for fl in _corpus.values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
|
||||
# Update cumulative cost tracker
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text())
|
||||
input_tok = extract.get('input_tokens', 0)
|
||||
output_tok = extract.get('output_tokens', 0)
|
||||
|
||||
|
||||
+24
-2
@@ -551,15 +551,37 @@ from graphify.detect import save_manifest
|
||||
|
||||
# Save manifest for --update
|
||||
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\"))
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
# In --update mode, 'all_files' carries the full corpus; 'files' is the changed
|
||||
# subset. Full-rebuild mode populates only 'files', so the fallback handles that.
|
||||
# root= relativizes the manifest keys to the scan root (same base as the build),
|
||||
# so the on-disk manifest is portable across clones/machines and a later --update
|
||||
# matches cached files instead of missing every one (#1417).
|
||||
save_manifest(detect.get('all_files') or detect['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output:
|
||||
# a detected file whose chunk failed or was omitted must stay unstamped so the
|
||||
# next --update re-queues it, otherwise it is marked done and its content is lost
|
||||
# forever (#2015). This mirrors the library extract path exactly
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus); do not stamp the
|
||||
# raw corpus. Code files are always stamped (AST is deterministic); only semantic
|
||||
# types are gated on output.
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_corpus = detect.get('all_files') or detect['files']
|
||||
_manifest_files = _stamped_manifest_files(_corpus, extract, Path('INPUT_PATH'))
|
||||
# Files dispatched this run (the changed subset) but NOT stamped above still carry
|
||||
# a stale semantic_hash from a prior run; clear it so detect_incremental re-queues
|
||||
# them instead of reading them as unchanged (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in detect['files'].items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus (not the stamp-filtered subset) so in-root
|
||||
# files newly excluded since last run are dropped rather than masquerading as
|
||||
# deletions; untouched files' prior rows are still preserved (#1908).
|
||||
_scan = {f for fl in _corpus.values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
|
||||
# Update cumulative cost tracker
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
input_tok = extract.get('input_tokens', 0)
|
||||
output_tok = extract.get('output_tokens', 0)
|
||||
|
||||
|
||||
+24
-2
@@ -554,15 +554,37 @@ from graphify.detect import save_manifest
|
||||
|
||||
# Save manifest for --update
|
||||
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\"))
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
# In --update mode, 'all_files' carries the full corpus; 'files' is the changed
|
||||
# subset. Full-rebuild mode populates only 'files', so the fallback handles that.
|
||||
# root= relativizes the manifest keys to the scan root (same base as the build),
|
||||
# so the on-disk manifest is portable across clones/machines and a later --update
|
||||
# matches cached files instead of missing every one (#1417).
|
||||
save_manifest(detect.get('all_files') or detect['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output:
|
||||
# a detected file whose chunk failed or was omitted must stay unstamped so the
|
||||
# next --update re-queues it, otherwise it is marked done and its content is lost
|
||||
# forever (#2015). This mirrors the library extract path exactly
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus); do not stamp the
|
||||
# raw corpus. Code files are always stamped (AST is deterministic); only semantic
|
||||
# types are gated on output.
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_corpus = detect.get('all_files') or detect['files']
|
||||
_manifest_files = _stamped_manifest_files(_corpus, extract, Path('INPUT_PATH'))
|
||||
# Files dispatched this run (the changed subset) but NOT stamped above still carry
|
||||
# a stale semantic_hash from a prior run; clear it so detect_incremental re-queues
|
||||
# them instead of reading them as unchanged (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in detect['files'].items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus (not the stamp-filtered subset) so in-root
|
||||
# files newly excluded since last run are dropped rather than masquerading as
|
||||
# deletions; untouched files' prior rows are still preserved (#1908).
|
||||
_scan = {f for fl in _corpus.values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
|
||||
# Update cumulative cost tracker
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
input_tok = extract.get('input_tokens', 0)
|
||||
output_tok = extract.get('output_tokens', 0)
|
||||
|
||||
|
||||
+24
-2
@@ -554,15 +554,37 @@ from graphify.detect import save_manifest
|
||||
|
||||
# Save manifest for --update
|
||||
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\"))
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
# In --update mode, 'all_files' carries the full corpus; 'files' is the changed
|
||||
# subset. Full-rebuild mode populates only 'files', so the fallback handles that.
|
||||
# root= relativizes the manifest keys to the scan root (same base as the build),
|
||||
# so the on-disk manifest is portable across clones/machines and a later --update
|
||||
# matches cached files instead of missing every one (#1417).
|
||||
save_manifest(detect.get('all_files') or detect['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output:
|
||||
# a detected file whose chunk failed or was omitted must stay unstamped so the
|
||||
# next --update re-queues it, otherwise it is marked done and its content is lost
|
||||
# forever (#2015). This mirrors the library extract path exactly
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus); do not stamp the
|
||||
# raw corpus. Code files are always stamped (AST is deterministic); only semantic
|
||||
# types are gated on output.
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_corpus = detect.get('all_files') or detect['files']
|
||||
_manifest_files = _stamped_manifest_files(_corpus, extract, Path('INPUT_PATH'))
|
||||
# Files dispatched this run (the changed subset) but NOT stamped above still carry
|
||||
# a stale semantic_hash from a prior run; clear it so detect_incremental re-queues
|
||||
# them instead of reading them as unchanged (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in detect['files'].items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus (not the stamp-filtered subset) so in-root
|
||||
# files newly excluded since last run are dropped rather than masquerading as
|
||||
# deletions; untouched files' prior rows are still preserved (#1908).
|
||||
_scan = {f for fl in _corpus.values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
|
||||
# Update cumulative cost tracker
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
input_tok = extract.get('input_tokens', 0)
|
||||
output_tok = extract.get('output_tokens', 0)
|
||||
|
||||
|
||||
@@ -546,15 +546,37 @@ from graphify.detect import save_manifest
|
||||
|
||||
# Save manifest for --update
|
||||
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\"))
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
# In --update mode, 'all_files' carries the full corpus; 'files' is the changed
|
||||
# subset. Full-rebuild mode populates only 'files', so the fallback handles that.
|
||||
# root= relativizes the manifest keys to the scan root (same base as the build),
|
||||
# so the on-disk manifest is portable across clones/machines and a later --update
|
||||
# matches cached files instead of missing every one (#1417).
|
||||
save_manifest(detect.get('all_files') or detect['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output:
|
||||
# a detected file whose chunk failed or was omitted must stay unstamped so the
|
||||
# next --update re-queues it, otherwise it is marked done and its content is lost
|
||||
# forever (#2015). This mirrors the library extract path exactly
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus); do not stamp the
|
||||
# raw corpus. Code files are always stamped (AST is deterministic); only semantic
|
||||
# types are gated on output.
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_corpus = detect.get('all_files') or detect['files']
|
||||
_manifest_files = _stamped_manifest_files(_corpus, extract, Path('INPUT_PATH'))
|
||||
# Files dispatched this run (the changed subset) but NOT stamped above still carry
|
||||
# a stale semantic_hash from a prior run; clear it so detect_incremental re-queues
|
||||
# them instead of reading them as unchanged (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in detect['files'].items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus (not the stamp-filtered subset) so in-root
|
||||
# files newly excluded since last run are dropped rather than masquerading as
|
||||
# deletions; untouched files' prior rows are still preserved (#1908).
|
||||
_scan = {f for fl in _corpus.values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
|
||||
# Update cumulative cost tracker
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
input_tok = extract.get('input_tokens', 0)
|
||||
output_tok = extract.get('output_tokens', 0)
|
||||
|
||||
|
||||
+24
-2
@@ -554,15 +554,37 @@ from graphify.detect import save_manifest
|
||||
|
||||
# Save manifest for --update
|
||||
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\"))
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
# In --update mode, 'all_files' carries the full corpus; 'files' is the changed
|
||||
# subset. Full-rebuild mode populates only 'files', so the fallback handles that.
|
||||
# root= relativizes the manifest keys to the scan root (same base as the build),
|
||||
# so the on-disk manifest is portable across clones/machines and a later --update
|
||||
# matches cached files instead of missing every one (#1417).
|
||||
save_manifest(detect.get('all_files') or detect['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output:
|
||||
# a detected file whose chunk failed or was omitted must stay unstamped so the
|
||||
# next --update re-queues it, otherwise it is marked done and its content is lost
|
||||
# forever (#2015). This mirrors the library extract path exactly
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus); do not stamp the
|
||||
# raw corpus. Code files are always stamped (AST is deterministic); only semantic
|
||||
# types are gated on output.
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_corpus = detect.get('all_files') or detect['files']
|
||||
_manifest_files = _stamped_manifest_files(_corpus, extract, Path('INPUT_PATH'))
|
||||
# Files dispatched this run (the changed subset) but NOT stamped above still carry
|
||||
# a stale semantic_hash from a prior run; clear it so detect_incremental re-queues
|
||||
# them instead of reading them as unchanged (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in detect['files'].items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus (not the stamp-filtered subset) so in-root
|
||||
# files newly excluded since last run are dropped rather than masquerading as
|
||||
# deletions; untouched files' prior rows are still preserved (#1908).
|
||||
_scan = {f for fl in _corpus.values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
|
||||
# Update cumulative cost tracker
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
input_tok = extract.get('input_tokens', 0)
|
||||
output_tok = extract.get('output_tokens', 0)
|
||||
|
||||
|
||||
+24
-2
@@ -552,15 +552,37 @@ from graphify.detect import save_manifest
|
||||
|
||||
# Save manifest for --update
|
||||
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\"))
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
# In --update mode, 'all_files' carries the full corpus; 'files' is the changed
|
||||
# subset. Full-rebuild mode populates only 'files', so the fallback handles that.
|
||||
# root= relativizes the manifest keys to the scan root (same base as the build),
|
||||
# so the on-disk manifest is portable across clones/machines and a later --update
|
||||
# matches cached files instead of missing every one (#1417).
|
||||
save_manifest(detect.get('all_files') or detect['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output:
|
||||
# a detected file whose chunk failed or was omitted must stay unstamped so the
|
||||
# next --update re-queues it, otherwise it is marked done and its content is lost
|
||||
# forever (#2015). This mirrors the library extract path exactly
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus); do not stamp the
|
||||
# raw corpus. Code files are always stamped (AST is deterministic); only semantic
|
||||
# types are gated on output.
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_corpus = detect.get('all_files') or detect['files']
|
||||
_manifest_files = _stamped_manifest_files(_corpus, extract, Path('INPUT_PATH'))
|
||||
# Files dispatched this run (the changed subset) but NOT stamped above still carry
|
||||
# a stale semantic_hash from a prior run; clear it so detect_incremental re-queues
|
||||
# them instead of reading them as unchanged (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in detect['files'].items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus (not the stamp-filtered subset) so in-root
|
||||
# files newly excluded since last run are dropped rather than masquerading as
|
||||
# deletions; untouched files' prior rows are still preserved (#1908).
|
||||
_scan = {f for fl in _corpus.values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
|
||||
# Update cumulative cost tracker
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
input_tok = extract.get('input_tokens', 0)
|
||||
output_tok = extract.get('output_tokens', 0)
|
||||
|
||||
|
||||
@@ -550,15 +550,37 @@ from graphify.detect import save_manifest
|
||||
|
||||
# Save manifest for --update
|
||||
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\"))
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
# In --update mode, 'all_files' carries the full corpus; 'files' is the changed
|
||||
# subset. Full-rebuild mode populates only 'files', so the fallback handles that.
|
||||
# root= relativizes the manifest keys to the scan root (same base as the build),
|
||||
# so the on-disk manifest is portable across clones/machines and a later --update
|
||||
# matches cached files instead of missing every one (#1417).
|
||||
save_manifest(detect.get('all_files') or detect['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output:
|
||||
# a detected file whose chunk failed or was omitted must stay unstamped so the
|
||||
# next --update re-queues it, otherwise it is marked done and its content is lost
|
||||
# forever (#2015). This mirrors the library extract path exactly
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus); do not stamp the
|
||||
# raw corpus. Code files are always stamped (AST is deterministic); only semantic
|
||||
# types are gated on output.
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_corpus = detect.get('all_files') or detect['files']
|
||||
_manifest_files = _stamped_manifest_files(_corpus, extract, Path('INPUT_PATH'))
|
||||
# Files dispatched this run (the changed subset) but NOT stamped above still carry
|
||||
# a stale semantic_hash from a prior run; clear it so detect_incremental re-queues
|
||||
# them instead of reading them as unchanged (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in detect['files'].items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus (not the stamp-filtered subset) so in-root
|
||||
# files newly excluded since last run are dropped rather than masquerading as
|
||||
# deletions; untouched files' prior rows are still preserved (#1908).
|
||||
_scan = {f for fl in _corpus.values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
|
||||
# Update cumulative cost tracker
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
input_tok = extract.get('input_tokens', 0)
|
||||
output_tok = extract.get('output_tokens', 0)
|
||||
|
||||
|
||||
@@ -576,15 +576,37 @@ from graphify.detect import save_manifest
|
||||
|
||||
# Save manifest for --update
|
||||
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\"))
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
# In --update mode, 'all_files' carries the full corpus; 'files' is the changed
|
||||
# subset. Full-rebuild mode populates only 'files', so the fallback handles that.
|
||||
# root= relativizes the manifest keys to the scan root (same base as the build),
|
||||
# so the on-disk manifest is portable across clones/machines and a later --update
|
||||
# matches cached files instead of missing every one (#1417).
|
||||
save_manifest(detect.get('all_files') or detect['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output:
|
||||
# a detected file whose chunk failed or was omitted must stay unstamped so the
|
||||
# next --update re-queues it, otherwise it is marked done and its content is lost
|
||||
# forever (#2015). This mirrors the library extract path exactly
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus); do not stamp the
|
||||
# raw corpus. Code files are always stamped (AST is deterministic); only semantic
|
||||
# types are gated on output.
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_corpus = detect.get('all_files') or detect['files']
|
||||
_manifest_files = _stamped_manifest_files(_corpus, extract, Path('INPUT_PATH'))
|
||||
# Files dispatched this run (the changed subset) but NOT stamped above still carry
|
||||
# a stale semantic_hash from a prior run; clear it so detect_incremental re-queues
|
||||
# them instead of reading them as unchanged (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in detect['files'].items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus (not the stamp-filtered subset) so in-root
|
||||
# files newly excluded since last run are dropped rather than masquerading as
|
||||
# deletions; untouched files' prior rows are still preserved (#1908).
|
||||
_scan = {f for fl in _corpus.values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
|
||||
# Update cumulative cost tracker
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
input_tok = extract.get('input_tokens', 0)
|
||||
output_tok = extract.get('output_tokens', 0)
|
||||
|
||||
|
||||
+24
-2
@@ -554,15 +554,37 @@ from graphify.detect import save_manifest
|
||||
|
||||
# Save manifest for --update
|
||||
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\"))
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
# In --update mode, 'all_files' carries the full corpus; 'files' is the changed
|
||||
# subset. Full-rebuild mode populates only 'files', so the fallback handles that.
|
||||
# root= relativizes the manifest keys to the scan root (same base as the build),
|
||||
# so the on-disk manifest is portable across clones/machines and a later --update
|
||||
# matches cached files instead of missing every one (#1417).
|
||||
save_manifest(detect.get('all_files') or detect['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output:
|
||||
# a detected file whose chunk failed or was omitted must stay unstamped so the
|
||||
# next --update re-queues it, otherwise it is marked done and its content is lost
|
||||
# forever (#2015). This mirrors the library extract path exactly
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus); do not stamp the
|
||||
# raw corpus. Code files are always stamped (AST is deterministic); only semantic
|
||||
# types are gated on output.
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_corpus = detect.get('all_files') or detect['files']
|
||||
_manifest_files = _stamped_manifest_files(_corpus, extract, Path('INPUT_PATH'))
|
||||
# Files dispatched this run (the changed subset) but NOT stamped above still carry
|
||||
# a stale semantic_hash from a prior run; clear it so detect_incremental re-queues
|
||||
# them instead of reading them as unchanged (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in detect['files'].items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus (not the stamp-filtered subset) so in-root
|
||||
# files newly excluded since last run are dropped rather than masquerading as
|
||||
# deletions; untouched files' prior rows are still preserved (#1908).
|
||||
_scan = {f for fl in _corpus.values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
|
||||
# Update cumulative cost tracker
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
input_tok = extract.get('input_tokens', 0)
|
||||
output_tok = extract.get('output_tokens', 0)
|
||||
|
||||
|
||||
@@ -142,7 +142,25 @@ print(f'[graphify update] Merged extraction written ({len(merged_out[\"nodes\"])
|
||||
# root= matches the build_merge call above so the manifest keys stay relative to
|
||||
# the scan root — portable across clones/machines, so --update keeps matching
|
||||
# cached files instead of missing every one after a move (#1417).
|
||||
save_manifest(incremental['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output
|
||||
# THIS run (new_extraction is this run's fresh extraction, read above before the
|
||||
# merge overwrote the file): a changed doc whose chunk failed must stay unstamped
|
||||
# so the next --update re-queues it, otherwise it is marked done and its content
|
||||
# is lost forever (#2015). Mirrors the library extract path
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus).
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_manifest_files = _stamped_manifest_files(incremental['files'], new_extraction, Path('INPUT_PATH'))
|
||||
# Changed semantic files dispatched this run but NOT stamped had their chunk fail
|
||||
# or be omitted; clear any stale semantic_hash so they are re-queued (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in incremental.get('new_files', {}).items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus so in-root files newly excluded since last run
|
||||
# are dropped rather than masquerading as deletions; untouched rows preserved (#1908).
|
||||
_scan = {f for fl in incremental['files'].values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
print('[graphify update] Manifest saved.')
|
||||
"
|
||||
```
|
||||
|
||||
@@ -142,7 +142,25 @@ print(f'[graphify update] Merged extraction written ({len(merged_out[\"nodes\"])
|
||||
# root= matches the build_merge call above so the manifest keys stay relative to
|
||||
# the scan root — portable across clones/machines, so --update keeps matching
|
||||
# cached files instead of missing every one after a move (#1417).
|
||||
save_manifest(incremental['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output
|
||||
# THIS run (new_extraction is this run's fresh extraction, read above before the
|
||||
# merge overwrote the file): a changed doc whose chunk failed must stay unstamped
|
||||
# so the next --update re-queues it, otherwise it is marked done and its content
|
||||
# is lost forever (#2015). Mirrors the library extract path
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus).
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_manifest_files = _stamped_manifest_files(incremental['files'], new_extraction, Path('INPUT_PATH'))
|
||||
# Changed semantic files dispatched this run but NOT stamped had their chunk fail
|
||||
# or be omitted; clear any stale semantic_hash so they are re-queued (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in incremental.get('new_files', {}).items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus so in-root files newly excluded since last run
|
||||
# are dropped rather than masquerading as deletions; untouched rows preserved (#1908).
|
||||
_scan = {f for fl in incremental['files'].values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
print('[graphify update] Manifest saved.')
|
||||
"
|
||||
```
|
||||
|
||||
@@ -142,7 +142,25 @@ print(f'[graphify update] Merged extraction written ({len(merged_out[\"nodes\"])
|
||||
# root= matches the build_merge call above so the manifest keys stay relative to
|
||||
# the scan root — portable across clones/machines, so --update keeps matching
|
||||
# cached files instead of missing every one after a move (#1417).
|
||||
save_manifest(incremental['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output
|
||||
# THIS run (new_extraction is this run's fresh extraction, read above before the
|
||||
# merge overwrote the file): a changed doc whose chunk failed must stay unstamped
|
||||
# so the next --update re-queues it, otherwise it is marked done and its content
|
||||
# is lost forever (#2015). Mirrors the library extract path
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus).
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_manifest_files = _stamped_manifest_files(incremental['files'], new_extraction, Path('INPUT_PATH'))
|
||||
# Changed semantic files dispatched this run but NOT stamped had their chunk fail
|
||||
# or be omitted; clear any stale semantic_hash so they are re-queued (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in incremental.get('new_files', {}).items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus so in-root files newly excluded since last run
|
||||
# are dropped rather than masquerading as deletions; untouched rows preserved (#1908).
|
||||
_scan = {f for fl in incremental['files'].values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
print('[graphify update] Manifest saved.')
|
||||
"
|
||||
```
|
||||
|
||||
@@ -142,7 +142,25 @@ print(f'[graphify update] Merged extraction written ({len(merged_out[\"nodes\"])
|
||||
# root= matches the build_merge call above so the manifest keys stay relative to
|
||||
# the scan root — portable across clones/machines, so --update keeps matching
|
||||
# cached files instead of missing every one after a move (#1417).
|
||||
save_manifest(incremental['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output
|
||||
# THIS run (new_extraction is this run's fresh extraction, read above before the
|
||||
# merge overwrote the file): a changed doc whose chunk failed must stay unstamped
|
||||
# so the next --update re-queues it, otherwise it is marked done and its content
|
||||
# is lost forever (#2015). Mirrors the library extract path
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus).
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_manifest_files = _stamped_manifest_files(incremental['files'], new_extraction, Path('INPUT_PATH'))
|
||||
# Changed semantic files dispatched this run but NOT stamped had their chunk fail
|
||||
# or be omitted; clear any stale semantic_hash so they are re-queued (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in incremental.get('new_files', {}).items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus so in-root files newly excluded since last run
|
||||
# are dropped rather than masquerading as deletions; untouched rows preserved (#1908).
|
||||
_scan = {f for fl in incremental['files'].values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
print('[graphify update] Manifest saved.')
|
||||
"
|
||||
```
|
||||
|
||||
@@ -142,7 +142,25 @@ print(f'[graphify update] Merged extraction written ({len(merged_out[\"nodes\"])
|
||||
# root= matches the build_merge call above so the manifest keys stay relative to
|
||||
# the scan root — portable across clones/machines, so --update keeps matching
|
||||
# cached files instead of missing every one after a move (#1417).
|
||||
save_manifest(incremental['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output
|
||||
# THIS run (new_extraction is this run's fresh extraction, read above before the
|
||||
# merge overwrote the file): a changed doc whose chunk failed must stay unstamped
|
||||
# so the next --update re-queues it, otherwise it is marked done and its content
|
||||
# is lost forever (#2015). Mirrors the library extract path
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus).
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_manifest_files = _stamped_manifest_files(incremental['files'], new_extraction, Path('INPUT_PATH'))
|
||||
# Changed semantic files dispatched this run but NOT stamped had their chunk fail
|
||||
# or be omitted; clear any stale semantic_hash so they are re-queued (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in incremental.get('new_files', {}).items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus so in-root files newly excluded since last run
|
||||
# are dropped rather than masquerading as deletions; untouched rows preserved (#1908).
|
||||
_scan = {f for fl in incremental['files'].values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
print('[graphify update] Manifest saved.')
|
||||
"
|
||||
```
|
||||
|
||||
@@ -142,7 +142,25 @@ print(f'[graphify update] Merged extraction written ({len(merged_out[\"nodes\"])
|
||||
# root= matches the build_merge call above so the manifest keys stay relative to
|
||||
# the scan root — portable across clones/machines, so --update keeps matching
|
||||
# cached files instead of missing every one after a move (#1417).
|
||||
save_manifest(incremental['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output
|
||||
# THIS run (new_extraction is this run's fresh extraction, read above before the
|
||||
# merge overwrote the file): a changed doc whose chunk failed must stay unstamped
|
||||
# so the next --update re-queues it, otherwise it is marked done and its content
|
||||
# is lost forever (#2015). Mirrors the library extract path
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus).
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_manifest_files = _stamped_manifest_files(incremental['files'], new_extraction, Path('INPUT_PATH'))
|
||||
# Changed semantic files dispatched this run but NOT stamped had their chunk fail
|
||||
# or be omitted; clear any stale semantic_hash so they are re-queued (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in incremental.get('new_files', {}).items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus so in-root files newly excluded since last run
|
||||
# are dropped rather than masquerading as deletions; untouched rows preserved (#1908).
|
||||
_scan = {f for fl in incremental['files'].values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
print('[graphify update] Manifest saved.')
|
||||
"
|
||||
```
|
||||
|
||||
@@ -142,7 +142,25 @@ print(f'[graphify update] Merged extraction written ({len(merged_out[\"nodes\"])
|
||||
# root= matches the build_merge call above so the manifest keys stay relative to
|
||||
# the scan root — portable across clones/machines, so --update keeps matching
|
||||
# cached files instead of missing every one after a move (#1417).
|
||||
save_manifest(incremental['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output
|
||||
# THIS run (new_extraction is this run's fresh extraction, read above before the
|
||||
# merge overwrote the file): a changed doc whose chunk failed must stay unstamped
|
||||
# so the next --update re-queues it, otherwise it is marked done and its content
|
||||
# is lost forever (#2015). Mirrors the library extract path
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus).
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_manifest_files = _stamped_manifest_files(incremental['files'], new_extraction, Path('INPUT_PATH'))
|
||||
# Changed semantic files dispatched this run but NOT stamped had their chunk fail
|
||||
# or be omitted; clear any stale semantic_hash so they are re-queued (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in incremental.get('new_files', {}).items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus so in-root files newly excluded since last run
|
||||
# are dropped rather than masquerading as deletions; untouched rows preserved (#1908).
|
||||
_scan = {f for fl in incremental['files'].values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
print('[graphify update] Manifest saved.')
|
||||
"
|
||||
```
|
||||
|
||||
@@ -142,7 +142,25 @@ print(f'[graphify update] Merged extraction written ({len(merged_out[\"nodes\"])
|
||||
# root= matches the build_merge call above so the manifest keys stay relative to
|
||||
# the scan root — portable across clones/machines, so --update keeps matching
|
||||
# cached files instead of missing every one after a move (#1417).
|
||||
save_manifest(incremental['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output
|
||||
# THIS run (new_extraction is this run's fresh extraction, read above before the
|
||||
# merge overwrote the file): a changed doc whose chunk failed must stay unstamped
|
||||
# so the next --update re-queues it, otherwise it is marked done and its content
|
||||
# is lost forever (#2015). Mirrors the library extract path
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus).
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_manifest_files = _stamped_manifest_files(incremental['files'], new_extraction, Path('INPUT_PATH'))
|
||||
# Changed semantic files dispatched this run but NOT stamped had their chunk fail
|
||||
# or be omitted; clear any stale semantic_hash so they are re-queued (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in incremental.get('new_files', {}).items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus so in-root files newly excluded since last run
|
||||
# are dropped rather than masquerading as deletions; untouched rows preserved (#1908).
|
||||
_scan = {f for fl in incremental['files'].values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
print('[graphify update] Manifest saved.')
|
||||
"
|
||||
```
|
||||
|
||||
@@ -142,7 +142,25 @@ print(f'[graphify update] Merged extraction written ({len(merged_out[\"nodes\"])
|
||||
# root= matches the build_merge call above so the manifest keys stay relative to
|
||||
# the scan root — portable across clones/machines, so --update keeps matching
|
||||
# cached files instead of missing every one after a move (#1417).
|
||||
save_manifest(incremental['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output
|
||||
# THIS run (new_extraction is this run's fresh extraction, read above before the
|
||||
# merge overwrote the file): a changed doc whose chunk failed must stay unstamped
|
||||
# so the next --update re-queues it, otherwise it is marked done and its content
|
||||
# is lost forever (#2015). Mirrors the library extract path
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus).
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_manifest_files = _stamped_manifest_files(incremental['files'], new_extraction, Path('INPUT_PATH'))
|
||||
# Changed semantic files dispatched this run but NOT stamped had their chunk fail
|
||||
# or be omitted; clear any stale semantic_hash so they are re-queued (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in incremental.get('new_files', {}).items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus so in-root files newly excluded since last run
|
||||
# are dropped rather than masquerading as deletions; untouched rows preserved (#1908).
|
||||
_scan = {f for fl in incremental['files'].values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
print('[graphify update] Manifest saved.')
|
||||
"
|
||||
```
|
||||
|
||||
@@ -142,7 +142,25 @@ print(f'[graphify update] Merged extraction written ({len(merged_out[\"nodes\"])
|
||||
# root= matches the build_merge call above so the manifest keys stay relative to
|
||||
# the scan root — portable across clones/machines, so --update keeps matching
|
||||
# cached files instead of missing every one after a move (#1417).
|
||||
save_manifest(incremental['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output
|
||||
# THIS run (new_extraction is this run's fresh extraction, read above before the
|
||||
# merge overwrote the file): a changed doc whose chunk failed must stay unstamped
|
||||
# so the next --update re-queues it, otherwise it is marked done and its content
|
||||
# is lost forever (#2015). Mirrors the library extract path
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus).
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_manifest_files = _stamped_manifest_files(incremental['files'], new_extraction, Path('INPUT_PATH'))
|
||||
# Changed semantic files dispatched this run but NOT stamped had their chunk fail
|
||||
# or be omitted; clear any stale semantic_hash so they are re-queued (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in incremental.get('new_files', {}).items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus so in-root files newly excluded since last run
|
||||
# are dropped rather than masquerading as deletions; untouched rows preserved (#1908).
|
||||
_scan = {f for fl in incremental['files'].values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
print('[graphify update] Manifest saved.')
|
||||
"
|
||||
```
|
||||
|
||||
@@ -142,7 +142,25 @@ print(f'[graphify update] Merged extraction written ({len(merged_out[\"nodes\"])
|
||||
# root= matches the build_merge call above so the manifest keys stay relative to
|
||||
# the scan root — portable across clones/machines, so --update keeps matching
|
||||
# cached files instead of missing every one after a move (#1417).
|
||||
save_manifest(incremental['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output
|
||||
# THIS run (new_extraction is this run's fresh extraction, read above before the
|
||||
# merge overwrote the file): a changed doc whose chunk failed must stay unstamped
|
||||
# so the next --update re-queues it, otherwise it is marked done and its content
|
||||
# is lost forever (#2015). Mirrors the library extract path
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus).
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_manifest_files = _stamped_manifest_files(incremental['files'], new_extraction, Path('INPUT_PATH'))
|
||||
# Changed semantic files dispatched this run but NOT stamped had their chunk fail
|
||||
# or be omitted; clear any stale semantic_hash so they are re-queued (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in incremental.get('new_files', {}).items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus so in-root files newly excluded since last run
|
||||
# are dropped rather than masquerading as deletions; untouched rows preserved (#1908).
|
||||
_scan = {f for fl in incremental['files'].values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
print('[graphify update] Manifest saved.')
|
||||
"
|
||||
```
|
||||
|
||||
@@ -142,7 +142,25 @@ print(f'[graphify update] Merged extraction written ({len(merged_out[\"nodes\"])
|
||||
# root= matches the build_merge call above so the manifest keys stay relative to
|
||||
# the scan root — portable across clones/machines, so --update keeps matching
|
||||
# cached files instead of missing every one after a move (#1417).
|
||||
save_manifest(incremental['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output
|
||||
# THIS run (new_extraction is this run's fresh extraction, read above before the
|
||||
# merge overwrote the file): a changed doc whose chunk failed must stay unstamped
|
||||
# so the next --update re-queues it, otherwise it is marked done and its content
|
||||
# is lost forever (#2015). Mirrors the library extract path
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus).
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_manifest_files = _stamped_manifest_files(incremental['files'], new_extraction, Path('INPUT_PATH'))
|
||||
# Changed semantic files dispatched this run but NOT stamped had their chunk fail
|
||||
# or be omitted; clear any stale semantic_hash so they are re-queued (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in incremental.get('new_files', {}).items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus so in-root files newly excluded since last run
|
||||
# are dropped rather than masquerading as deletions; untouched rows preserved (#1908).
|
||||
_scan = {f for fl in incremental['files'].values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
print('[graphify update] Manifest saved.')
|
||||
"
|
||||
```
|
||||
|
||||
@@ -142,7 +142,25 @@ print(f'[graphify update] Merged extraction written ({len(merged_out[\"nodes\"])
|
||||
# root= matches the build_merge call above so the manifest keys stay relative to
|
||||
# the scan root — portable across clones/machines, so --update keeps matching
|
||||
# cached files instead of missing every one after a move (#1417).
|
||||
save_manifest(incremental['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output
|
||||
# THIS run (new_extraction is this run's fresh extraction, read above before the
|
||||
# merge overwrote the file): a changed doc whose chunk failed must stay unstamped
|
||||
# so the next --update re-queues it, otherwise it is marked done and its content
|
||||
# is lost forever (#2015). Mirrors the library extract path
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus).
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_manifest_files = _stamped_manifest_files(incremental['files'], new_extraction, Path('INPUT_PATH'))
|
||||
# Changed semantic files dispatched this run but NOT stamped had their chunk fail
|
||||
# or be omitted; clear any stale semantic_hash so they are re-queued (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in incremental.get('new_files', {}).items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus so in-root files newly excluded since last run
|
||||
# are dropped rather than masquerading as deletions; untouched rows preserved (#1908).
|
||||
_scan = {f for fl in incremental['files'].values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
print('[graphify update] Manifest saved.')
|
||||
"
|
||||
```
|
||||
|
||||
@@ -142,7 +142,25 @@ print(f'[graphify update] Merged extraction written ({len(merged_out[\"nodes\"])
|
||||
# root= matches the build_merge call above so the manifest keys stay relative to
|
||||
# the scan root — portable across clones/machines, so --update keeps matching
|
||||
# cached files instead of missing every one after a move (#1417).
|
||||
save_manifest(incremental['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output
|
||||
# THIS run (new_extraction is this run's fresh extraction, read above before the
|
||||
# merge overwrote the file): a changed doc whose chunk failed must stay unstamped
|
||||
# so the next --update re-queues it, otherwise it is marked done and its content
|
||||
# is lost forever (#2015). Mirrors the library extract path
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus).
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_manifest_files = _stamped_manifest_files(incremental['files'], new_extraction, Path('INPUT_PATH'))
|
||||
# Changed semantic files dispatched this run but NOT stamped had their chunk fail
|
||||
# or be omitted; clear any stale semantic_hash so they are re-queued (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in incremental.get('new_files', {}).items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus so in-root files newly excluded since last run
|
||||
# are dropped rather than masquerading as deletions; untouched rows preserved (#1908).
|
||||
_scan = {f for fl in incremental['files'].values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
print('[graphify update] Manifest saved.')
|
||||
"
|
||||
```
|
||||
|
||||
@@ -551,15 +551,37 @@ from graphify.detect import save_manifest
|
||||
|
||||
# Save manifest for --update
|
||||
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\"))
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
# In --update mode, 'all_files' carries the full corpus; 'files' is the changed
|
||||
# subset. Full-rebuild mode populates only 'files', so the fallback handles that.
|
||||
# root= relativizes the manifest keys to the scan root (same base as the build),
|
||||
# so the on-disk manifest is portable across clones/machines and a later --update
|
||||
# matches cached files instead of missing every one (#1417).
|
||||
save_manifest(detect.get('all_files') or detect['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output:
|
||||
# a detected file whose chunk failed or was omitted must stay unstamped so the
|
||||
# next --update re-queues it, otherwise it is marked done and its content is lost
|
||||
# forever (#2015). This mirrors the library extract path exactly
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus); do not stamp the
|
||||
# raw corpus. Code files are always stamped (AST is deterministic); only semantic
|
||||
# types are gated on output.
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_corpus = detect.get('all_files') or detect['files']
|
||||
_manifest_files = _stamped_manifest_files(_corpus, extract, Path('INPUT_PATH'))
|
||||
# Files dispatched this run (the changed subset) but NOT stamped above still carry
|
||||
# a stale semantic_hash from a prior run; clear it so detect_incremental re-queues
|
||||
# them instead of reading them as unchanged (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in detect['files'].items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus (not the stamp-filtered subset) so in-root
|
||||
# files newly excluded since last run are dropped rather than masquerading as
|
||||
# deletions; untouched files' prior rows are still preserved (#1908).
|
||||
_scan = {f for fl in _corpus.values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
|
||||
# Update cumulative cost tracker
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
input_tok = extract.get('input_tokens', 0)
|
||||
output_tok = extract.get('output_tokens', 0)
|
||||
|
||||
|
||||
@@ -677,10 +677,19 @@ from graphify.detect import save_manifest
|
||||
|
||||
# Save manifest for --update
|
||||
detect = json.loads(Path('.graphify_detect.json').read_text())
|
||||
save_manifest(detect['files'], root='INPUT_PATH')
|
||||
extract = json.loads(Path('.graphify_extract.json').read_text())
|
||||
# Stamp only semantic files that produced output so a failed chunk is re-queued next run, not lost (#2015).
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_corpus = detect.get('all_files') or detect['files']
|
||||
_manifest_files = _stamped_manifest_files(_corpus, extract, Path('INPUT_PATH'))
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in detect['files'].items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
_scan = {f for fl in _corpus.values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
|
||||
# Update cumulative cost tracker
|
||||
extract = json.loads(Path('.graphify_extract.json').read_text())
|
||||
input_tok = extract.get('input_tokens', 0)
|
||||
output_tok = extract.get('output_tokens', 0)
|
||||
|
||||
|
||||
@@ -551,15 +551,37 @@ from graphify.detect import save_manifest
|
||||
|
||||
# Save manifest for --update
|
||||
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\"))
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
# In --update mode, 'all_files' carries the full corpus; 'files' is the changed
|
||||
# subset. Full-rebuild mode populates only 'files', so the fallback handles that.
|
||||
# root= relativizes the manifest keys to the scan root (same base as the build),
|
||||
# so the on-disk manifest is portable across clones/machines and a later --update
|
||||
# matches cached files instead of missing every one (#1417).
|
||||
save_manifest(detect.get('all_files') or detect['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output:
|
||||
# a detected file whose chunk failed or was omitted must stay unstamped so the
|
||||
# next --update re-queues it, otherwise it is marked done and its content is lost
|
||||
# forever (#2015). This mirrors the library extract path exactly
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus); do not stamp the
|
||||
# raw corpus. Code files are always stamped (AST is deterministic); only semantic
|
||||
# types are gated on output.
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_corpus = detect.get('all_files') or detect['files']
|
||||
_manifest_files = _stamped_manifest_files(_corpus, extract, Path('INPUT_PATH'))
|
||||
# Files dispatched this run (the changed subset) but NOT stamped above still carry
|
||||
# a stale semantic_hash from a prior run; clear it so detect_incremental re-queues
|
||||
# them instead of reading them as unchanged (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in detect['files'].items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus (not the stamp-filtered subset) so in-root
|
||||
# files newly excluded since last run are dropped rather than masquerading as
|
||||
# deletions; untouched files' prior rows are still preserved (#1908).
|
||||
_scan = {f for fl in _corpus.values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
|
||||
# Update cumulative cost tracker
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
input_tok = extract.get('input_tokens', 0)
|
||||
output_tok = extract.get('output_tokens', 0)
|
||||
|
||||
|
||||
@@ -554,15 +554,37 @@ from graphify.detect import save_manifest
|
||||
|
||||
# Save manifest for --update
|
||||
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\"))
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
# In --update mode, 'all_files' carries the full corpus; 'files' is the changed
|
||||
# subset. Full-rebuild mode populates only 'files', so the fallback handles that.
|
||||
# root= relativizes the manifest keys to the scan root (same base as the build),
|
||||
# so the on-disk manifest is portable across clones/machines and a later --update
|
||||
# matches cached files instead of missing every one (#1417).
|
||||
save_manifest(detect.get('all_files') or detect['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output:
|
||||
# a detected file whose chunk failed or was omitted must stay unstamped so the
|
||||
# next --update re-queues it, otherwise it is marked done and its content is lost
|
||||
# forever (#2015). This mirrors the library extract path exactly
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus); do not stamp the
|
||||
# raw corpus. Code files are always stamped (AST is deterministic); only semantic
|
||||
# types are gated on output.
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_corpus = detect.get('all_files') or detect['files']
|
||||
_manifest_files = _stamped_manifest_files(_corpus, extract, Path('INPUT_PATH'))
|
||||
# Files dispatched this run (the changed subset) but NOT stamped above still carry
|
||||
# a stale semantic_hash from a prior run; clear it so detect_incremental re-queues
|
||||
# them instead of reading them as unchanged (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in detect['files'].items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus (not the stamp-filtered subset) so in-root
|
||||
# files newly excluded since last run are dropped rather than masquerading as
|
||||
# deletions; untouched files' prior rows are still preserved (#1908).
|
||||
_scan = {f for fl in _corpus.values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
|
||||
# Update cumulative cost tracker
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
input_tok = extract.get('input_tokens', 0)
|
||||
output_tok = extract.get('output_tokens', 0)
|
||||
|
||||
|
||||
@@ -551,15 +551,37 @@ from graphify.detect import save_manifest
|
||||
|
||||
# Save manifest for --update
|
||||
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\"))
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
# In --update mode, 'all_files' carries the full corpus; 'files' is the changed
|
||||
# subset. Full-rebuild mode populates only 'files', so the fallback handles that.
|
||||
# root= relativizes the manifest keys to the scan root (same base as the build),
|
||||
# so the on-disk manifest is portable across clones/machines and a later --update
|
||||
# matches cached files instead of missing every one (#1417).
|
||||
save_manifest(detect.get('all_files') or detect['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output:
|
||||
# a detected file whose chunk failed or was omitted must stay unstamped so the
|
||||
# next --update re-queues it, otherwise it is marked done and its content is lost
|
||||
# forever (#2015). This mirrors the library extract path exactly
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus); do not stamp the
|
||||
# raw corpus. Code files are always stamped (AST is deterministic); only semantic
|
||||
# types are gated on output.
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_corpus = detect.get('all_files') or detect['files']
|
||||
_manifest_files = _stamped_manifest_files(_corpus, extract, Path('INPUT_PATH'))
|
||||
# Files dispatched this run (the changed subset) but NOT stamped above still carry
|
||||
# a stale semantic_hash from a prior run; clear it so detect_incremental re-queues
|
||||
# them instead of reading them as unchanged (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in detect['files'].items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus (not the stamp-filtered subset) so in-root
|
||||
# files newly excluded since last run are dropped rather than masquerading as
|
||||
# deletions; untouched files' prior rows are still preserved (#1908).
|
||||
_scan = {f for fl in _corpus.values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
|
||||
# Update cumulative cost tracker
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
input_tok = extract.get('input_tokens', 0)
|
||||
output_tok = extract.get('output_tokens', 0)
|
||||
|
||||
|
||||
@@ -554,15 +554,37 @@ from graphify.detect import save_manifest
|
||||
|
||||
# Save manifest for --update
|
||||
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\"))
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
# In --update mode, 'all_files' carries the full corpus; 'files' is the changed
|
||||
# subset. Full-rebuild mode populates only 'files', so the fallback handles that.
|
||||
# root= relativizes the manifest keys to the scan root (same base as the build),
|
||||
# so the on-disk manifest is portable across clones/machines and a later --update
|
||||
# matches cached files instead of missing every one (#1417).
|
||||
save_manifest(detect.get('all_files') or detect['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output:
|
||||
# a detected file whose chunk failed or was omitted must stay unstamped so the
|
||||
# next --update re-queues it, otherwise it is marked done and its content is lost
|
||||
# forever (#2015). This mirrors the library extract path exactly
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus); do not stamp the
|
||||
# raw corpus. Code files are always stamped (AST is deterministic); only semantic
|
||||
# types are gated on output.
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_corpus = detect.get('all_files') or detect['files']
|
||||
_manifest_files = _stamped_manifest_files(_corpus, extract, Path('INPUT_PATH'))
|
||||
# Files dispatched this run (the changed subset) but NOT stamped above still carry
|
||||
# a stale semantic_hash from a prior run; clear it so detect_incremental re-queues
|
||||
# them instead of reading them as unchanged (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in detect['files'].items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus (not the stamp-filtered subset) so in-root
|
||||
# files newly excluded since last run are dropped rather than masquerading as
|
||||
# deletions; untouched files' prior rows are still preserved (#1908).
|
||||
_scan = {f for fl in _corpus.values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
|
||||
# Update cumulative cost tracker
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
input_tok = extract.get('input_tokens', 0)
|
||||
output_tok = extract.get('output_tokens', 0)
|
||||
|
||||
|
||||
@@ -795,10 +795,19 @@ from graphify.detect import save_manifest
|
||||
|
||||
# Save manifest for --update
|
||||
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text())
|
||||
save_manifest(detect['files'], root='INPUT_PATH')
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text())
|
||||
# Stamp only semantic files that produced output so a failed chunk is re-queued next run, not lost (#2015).
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_corpus = detect.get('all_files') or detect['files']
|
||||
_manifest_files = _stamped_manifest_files(_corpus, extract, Path('INPUT_PATH'))
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in detect['files'].items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
_scan = {f for fl in _corpus.values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
|
||||
# Update cumulative cost tracker
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text())
|
||||
input_tok = extract.get('input_tokens', 0)
|
||||
output_tok = extract.get('output_tokens', 0)
|
||||
|
||||
|
||||
@@ -551,15 +551,37 @@ from graphify.detect import save_manifest
|
||||
|
||||
# Save manifest for --update
|
||||
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\"))
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
# In --update mode, 'all_files' carries the full corpus; 'files' is the changed
|
||||
# subset. Full-rebuild mode populates only 'files', so the fallback handles that.
|
||||
# root= relativizes the manifest keys to the scan root (same base as the build),
|
||||
# so the on-disk manifest is portable across clones/machines and a later --update
|
||||
# matches cached files instead of missing every one (#1417).
|
||||
save_manifest(detect.get('all_files') or detect['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output:
|
||||
# a detected file whose chunk failed or was omitted must stay unstamped so the
|
||||
# next --update re-queues it, otherwise it is marked done and its content is lost
|
||||
# forever (#2015). This mirrors the library extract path exactly
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus); do not stamp the
|
||||
# raw corpus. Code files are always stamped (AST is deterministic); only semantic
|
||||
# types are gated on output.
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_corpus = detect.get('all_files') or detect['files']
|
||||
_manifest_files = _stamped_manifest_files(_corpus, extract, Path('INPUT_PATH'))
|
||||
# Files dispatched this run (the changed subset) but NOT stamped above still carry
|
||||
# a stale semantic_hash from a prior run; clear it so detect_incremental re-queues
|
||||
# them instead of reading them as unchanged (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in detect['files'].items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus (not the stamp-filtered subset) so in-root
|
||||
# files newly excluded since last run are dropped rather than masquerading as
|
||||
# deletions; untouched files' prior rows are still preserved (#1908).
|
||||
_scan = {f for fl in _corpus.values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
|
||||
# Update cumulative cost tracker
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
input_tok = extract.get('input_tokens', 0)
|
||||
output_tok = extract.get('output_tokens', 0)
|
||||
|
||||
|
||||
@@ -554,15 +554,37 @@ from graphify.detect import save_manifest
|
||||
|
||||
# Save manifest for --update
|
||||
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\"))
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
# In --update mode, 'all_files' carries the full corpus; 'files' is the changed
|
||||
# subset. Full-rebuild mode populates only 'files', so the fallback handles that.
|
||||
# root= relativizes the manifest keys to the scan root (same base as the build),
|
||||
# so the on-disk manifest is portable across clones/machines and a later --update
|
||||
# matches cached files instead of missing every one (#1417).
|
||||
save_manifest(detect.get('all_files') or detect['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output:
|
||||
# a detected file whose chunk failed or was omitted must stay unstamped so the
|
||||
# next --update re-queues it, otherwise it is marked done and its content is lost
|
||||
# forever (#2015). This mirrors the library extract path exactly
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus); do not stamp the
|
||||
# raw corpus. Code files are always stamped (AST is deterministic); only semantic
|
||||
# types are gated on output.
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_corpus = detect.get('all_files') or detect['files']
|
||||
_manifest_files = _stamped_manifest_files(_corpus, extract, Path('INPUT_PATH'))
|
||||
# Files dispatched this run (the changed subset) but NOT stamped above still carry
|
||||
# a stale semantic_hash from a prior run; clear it so detect_incremental re-queues
|
||||
# them instead of reading them as unchanged (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in detect['files'].items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus (not the stamp-filtered subset) so in-root
|
||||
# files newly excluded since last run are dropped rather than masquerading as
|
||||
# deletions; untouched files' prior rows are still preserved (#1908).
|
||||
_scan = {f for fl in _corpus.values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
|
||||
# Update cumulative cost tracker
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
input_tok = extract.get('input_tokens', 0)
|
||||
output_tok = extract.get('output_tokens', 0)
|
||||
|
||||
|
||||
@@ -554,15 +554,37 @@ from graphify.detect import save_manifest
|
||||
|
||||
# Save manifest for --update
|
||||
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\"))
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
# In --update mode, 'all_files' carries the full corpus; 'files' is the changed
|
||||
# subset. Full-rebuild mode populates only 'files', so the fallback handles that.
|
||||
# root= relativizes the manifest keys to the scan root (same base as the build),
|
||||
# so the on-disk manifest is portable across clones/machines and a later --update
|
||||
# matches cached files instead of missing every one (#1417).
|
||||
save_manifest(detect.get('all_files') or detect['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output:
|
||||
# a detected file whose chunk failed or was omitted must stay unstamped so the
|
||||
# next --update re-queues it, otherwise it is marked done and its content is lost
|
||||
# forever (#2015). This mirrors the library extract path exactly
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus); do not stamp the
|
||||
# raw corpus. Code files are always stamped (AST is deterministic); only semantic
|
||||
# types are gated on output.
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_corpus = detect.get('all_files') or detect['files']
|
||||
_manifest_files = _stamped_manifest_files(_corpus, extract, Path('INPUT_PATH'))
|
||||
# Files dispatched this run (the changed subset) but NOT stamped above still carry
|
||||
# a stale semantic_hash from a prior run; clear it so detect_incremental re-queues
|
||||
# them instead of reading them as unchanged (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in detect['files'].items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus (not the stamp-filtered subset) so in-root
|
||||
# files newly excluded since last run are dropped rather than masquerading as
|
||||
# deletions; untouched files' prior rows are still preserved (#1908).
|
||||
_scan = {f for fl in _corpus.values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
|
||||
# Update cumulative cost tracker
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
input_tok = extract.get('input_tokens', 0)
|
||||
output_tok = extract.get('output_tokens', 0)
|
||||
|
||||
|
||||
@@ -546,15 +546,37 @@ from graphify.detect import save_manifest
|
||||
|
||||
# Save manifest for --update
|
||||
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\"))
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
# In --update mode, 'all_files' carries the full corpus; 'files' is the changed
|
||||
# subset. Full-rebuild mode populates only 'files', so the fallback handles that.
|
||||
# root= relativizes the manifest keys to the scan root (same base as the build),
|
||||
# so the on-disk manifest is portable across clones/machines and a later --update
|
||||
# matches cached files instead of missing every one (#1417).
|
||||
save_manifest(detect.get('all_files') or detect['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output:
|
||||
# a detected file whose chunk failed or was omitted must stay unstamped so the
|
||||
# next --update re-queues it, otherwise it is marked done and its content is lost
|
||||
# forever (#2015). This mirrors the library extract path exactly
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus); do not stamp the
|
||||
# raw corpus. Code files are always stamped (AST is deterministic); only semantic
|
||||
# types are gated on output.
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_corpus = detect.get('all_files') or detect['files']
|
||||
_manifest_files = _stamped_manifest_files(_corpus, extract, Path('INPUT_PATH'))
|
||||
# Files dispatched this run (the changed subset) but NOT stamped above still carry
|
||||
# a stale semantic_hash from a prior run; clear it so detect_incremental re-queues
|
||||
# them instead of reading them as unchanged (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in detect['files'].items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus (not the stamp-filtered subset) so in-root
|
||||
# files newly excluded since last run are dropped rather than masquerading as
|
||||
# deletions; untouched files' prior rows are still preserved (#1908).
|
||||
_scan = {f for fl in _corpus.values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
|
||||
# Update cumulative cost tracker
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
input_tok = extract.get('input_tokens', 0)
|
||||
output_tok = extract.get('output_tokens', 0)
|
||||
|
||||
|
||||
@@ -554,15 +554,37 @@ from graphify.detect import save_manifest
|
||||
|
||||
# Save manifest for --update
|
||||
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\"))
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
# In --update mode, 'all_files' carries the full corpus; 'files' is the changed
|
||||
# subset. Full-rebuild mode populates only 'files', so the fallback handles that.
|
||||
# root= relativizes the manifest keys to the scan root (same base as the build),
|
||||
# so the on-disk manifest is portable across clones/machines and a later --update
|
||||
# matches cached files instead of missing every one (#1417).
|
||||
save_manifest(detect.get('all_files') or detect['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output:
|
||||
# a detected file whose chunk failed or was omitted must stay unstamped so the
|
||||
# next --update re-queues it, otherwise it is marked done and its content is lost
|
||||
# forever (#2015). This mirrors the library extract path exactly
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus); do not stamp the
|
||||
# raw corpus. Code files are always stamped (AST is deterministic); only semantic
|
||||
# types are gated on output.
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_corpus = detect.get('all_files') or detect['files']
|
||||
_manifest_files = _stamped_manifest_files(_corpus, extract, Path('INPUT_PATH'))
|
||||
# Files dispatched this run (the changed subset) but NOT stamped above still carry
|
||||
# a stale semantic_hash from a prior run; clear it so detect_incremental re-queues
|
||||
# them instead of reading them as unchanged (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in detect['files'].items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus (not the stamp-filtered subset) so in-root
|
||||
# files newly excluded since last run are dropped rather than masquerading as
|
||||
# deletions; untouched files' prior rows are still preserved (#1908).
|
||||
_scan = {f for fl in _corpus.values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
|
||||
# Update cumulative cost tracker
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
input_tok = extract.get('input_tokens', 0)
|
||||
output_tok = extract.get('output_tokens', 0)
|
||||
|
||||
|
||||
@@ -552,15 +552,37 @@ from graphify.detect import save_manifest
|
||||
|
||||
# Save manifest for --update
|
||||
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\"))
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
# In --update mode, 'all_files' carries the full corpus; 'files' is the changed
|
||||
# subset. Full-rebuild mode populates only 'files', so the fallback handles that.
|
||||
# root= relativizes the manifest keys to the scan root (same base as the build),
|
||||
# so the on-disk manifest is portable across clones/machines and a later --update
|
||||
# matches cached files instead of missing every one (#1417).
|
||||
save_manifest(detect.get('all_files') or detect['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output:
|
||||
# a detected file whose chunk failed or was omitted must stay unstamped so the
|
||||
# next --update re-queues it, otherwise it is marked done and its content is lost
|
||||
# forever (#2015). This mirrors the library extract path exactly
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus); do not stamp the
|
||||
# raw corpus. Code files are always stamped (AST is deterministic); only semantic
|
||||
# types are gated on output.
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_corpus = detect.get('all_files') or detect['files']
|
||||
_manifest_files = _stamped_manifest_files(_corpus, extract, Path('INPUT_PATH'))
|
||||
# Files dispatched this run (the changed subset) but NOT stamped above still carry
|
||||
# a stale semantic_hash from a prior run; clear it so detect_incremental re-queues
|
||||
# them instead of reading them as unchanged (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in detect['files'].items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus (not the stamp-filtered subset) so in-root
|
||||
# files newly excluded since last run are dropped rather than masquerading as
|
||||
# deletions; untouched files' prior rows are still preserved (#1908).
|
||||
_scan = {f for fl in _corpus.values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
|
||||
# Update cumulative cost tracker
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
input_tok = extract.get('input_tokens', 0)
|
||||
output_tok = extract.get('output_tokens', 0)
|
||||
|
||||
|
||||
@@ -550,15 +550,37 @@ from graphify.detect import save_manifest
|
||||
|
||||
# Save manifest for --update
|
||||
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\"))
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
# In --update mode, 'all_files' carries the full corpus; 'files' is the changed
|
||||
# subset. Full-rebuild mode populates only 'files', so the fallback handles that.
|
||||
# root= relativizes the manifest keys to the scan root (same base as the build),
|
||||
# so the on-disk manifest is portable across clones/machines and a later --update
|
||||
# matches cached files instead of missing every one (#1417).
|
||||
save_manifest(detect.get('all_files') or detect['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output:
|
||||
# a detected file whose chunk failed or was omitted must stay unstamped so the
|
||||
# next --update re-queues it, otherwise it is marked done and its content is lost
|
||||
# forever (#2015). This mirrors the library extract path exactly
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus); do not stamp the
|
||||
# raw corpus. Code files are always stamped (AST is deterministic); only semantic
|
||||
# types are gated on output.
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_corpus = detect.get('all_files') or detect['files']
|
||||
_manifest_files = _stamped_manifest_files(_corpus, extract, Path('INPUT_PATH'))
|
||||
# Files dispatched this run (the changed subset) but NOT stamped above still carry
|
||||
# a stale semantic_hash from a prior run; clear it so detect_incremental re-queues
|
||||
# them instead of reading them as unchanged (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in detect['files'].items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus (not the stamp-filtered subset) so in-root
|
||||
# files newly excluded since last run are dropped rather than masquerading as
|
||||
# deletions; untouched files' prior rows are still preserved (#1908).
|
||||
_scan = {f for fl in _corpus.values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
|
||||
# Update cumulative cost tracker
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
input_tok = extract.get('input_tokens', 0)
|
||||
output_tok = extract.get('output_tokens', 0)
|
||||
|
||||
|
||||
@@ -576,15 +576,37 @@ from graphify.detect import save_manifest
|
||||
|
||||
# Save manifest for --update
|
||||
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\"))
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
# In --update mode, 'all_files' carries the full corpus; 'files' is the changed
|
||||
# subset. Full-rebuild mode populates only 'files', so the fallback handles that.
|
||||
# root= relativizes the manifest keys to the scan root (same base as the build),
|
||||
# so the on-disk manifest is portable across clones/machines and a later --update
|
||||
# matches cached files instead of missing every one (#1417).
|
||||
save_manifest(detect.get('all_files') or detect['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output:
|
||||
# a detected file whose chunk failed or was omitted must stay unstamped so the
|
||||
# next --update re-queues it, otherwise it is marked done and its content is lost
|
||||
# forever (#2015). This mirrors the library extract path exactly
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus); do not stamp the
|
||||
# raw corpus. Code files are always stamped (AST is deterministic); only semantic
|
||||
# types are gated on output.
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_corpus = detect.get('all_files') or detect['files']
|
||||
_manifest_files = _stamped_manifest_files(_corpus, extract, Path('INPUT_PATH'))
|
||||
# Files dispatched this run (the changed subset) but NOT stamped above still carry
|
||||
# a stale semantic_hash from a prior run; clear it so detect_incremental re-queues
|
||||
# them instead of reading them as unchanged (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in detect['files'].items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus (not the stamp-filtered subset) so in-root
|
||||
# files newly excluded since last run are dropped rather than masquerading as
|
||||
# deletions; untouched files' prior rows are still preserved (#1908).
|
||||
_scan = {f for fl in _corpus.values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
|
||||
# Update cumulative cost tracker
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
input_tok = extract.get('input_tokens', 0)
|
||||
output_tok = extract.get('output_tokens', 0)
|
||||
|
||||
|
||||
@@ -554,15 +554,37 @@ from graphify.detect import save_manifest
|
||||
|
||||
# Save manifest for --update
|
||||
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\"))
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
# In --update mode, 'all_files' carries the full corpus; 'files' is the changed
|
||||
# subset. Full-rebuild mode populates only 'files', so the fallback handles that.
|
||||
# root= relativizes the manifest keys to the scan root (same base as the build),
|
||||
# so the on-disk manifest is portable across clones/machines and a later --update
|
||||
# matches cached files instead of missing every one (#1417).
|
||||
save_manifest(detect.get('all_files') or detect['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output:
|
||||
# a detected file whose chunk failed or was omitted must stay unstamped so the
|
||||
# next --update re-queues it, otherwise it is marked done and its content is lost
|
||||
# forever (#2015). This mirrors the library extract path exactly
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus); do not stamp the
|
||||
# raw corpus. Code files are always stamped (AST is deterministic); only semantic
|
||||
# types are gated on output.
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_corpus = detect.get('all_files') or detect['files']
|
||||
_manifest_files = _stamped_manifest_files(_corpus, extract, Path('INPUT_PATH'))
|
||||
# Files dispatched this run (the changed subset) but NOT stamped above still carry
|
||||
# a stale semantic_hash from a prior run; clear it so detect_incremental re-queues
|
||||
# them instead of reading them as unchanged (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in detect['files'].items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus (not the stamp-filtered subset) so in-root
|
||||
# files newly excluded since last run are dropped rather than masquerading as
|
||||
# deletions; untouched files' prior rows are still preserved (#1908).
|
||||
_scan = {f for fl in _corpus.values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
|
||||
# Update cumulative cost tracker
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
input_tok = extract.get('input_tokens', 0)
|
||||
output_tok = extract.get('output_tokens', 0)
|
||||
|
||||
|
||||
@@ -142,7 +142,25 @@ print(f'[graphify update] Merged extraction written ({len(merged_out[\"nodes\"])
|
||||
# root= matches the build_merge call above so the manifest keys stay relative to
|
||||
# the scan root — portable across clones/machines, so --update keeps matching
|
||||
# cached files instead of missing every one after a move (#1417).
|
||||
save_manifest(incremental['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output
|
||||
# THIS run (new_extraction is this run's fresh extraction, read above before the
|
||||
# merge overwrote the file): a changed doc whose chunk failed must stay unstamped
|
||||
# so the next --update re-queues it, otherwise it is marked done and its content
|
||||
# is lost forever (#2015). Mirrors the library extract path
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus).
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_manifest_files = _stamped_manifest_files(incremental['files'], new_extraction, Path('INPUT_PATH'))
|
||||
# Changed semantic files dispatched this run but NOT stamped had their chunk fail
|
||||
# or be omitted; clear any stale semantic_hash so they are re-queued (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in incremental.get('new_files', {}).items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus so in-root files newly excluded since last run
|
||||
# are dropped rather than masquerading as deletions; untouched rows preserved (#1908).
|
||||
_scan = {f for fl in incremental['files'].values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
print('[graphify update] Manifest saved.')
|
||||
"
|
||||
```
|
||||
|
||||
@@ -142,7 +142,25 @@ print(f'[graphify update] Merged extraction written ({len(merged_out[\"nodes\"])
|
||||
# root= matches the build_merge call above so the manifest keys stay relative to
|
||||
# the scan root — portable across clones/machines, so --update keeps matching
|
||||
# cached files instead of missing every one after a move (#1417).
|
||||
save_manifest(incremental['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output
|
||||
# THIS run (new_extraction is this run's fresh extraction, read above before the
|
||||
# merge overwrote the file): a changed doc whose chunk failed must stay unstamped
|
||||
# so the next --update re-queues it, otherwise it is marked done and its content
|
||||
# is lost forever (#2015). Mirrors the library extract path
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus).
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_manifest_files = _stamped_manifest_files(incremental['files'], new_extraction, Path('INPUT_PATH'))
|
||||
# Changed semantic files dispatched this run but NOT stamped had their chunk fail
|
||||
# or be omitted; clear any stale semantic_hash so they are re-queued (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in incremental.get('new_files', {}).items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus so in-root files newly excluded since last run
|
||||
# are dropped rather than masquerading as deletions; untouched rows preserved (#1908).
|
||||
_scan = {f for fl in incremental['files'].values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
print('[graphify update] Manifest saved.')
|
||||
"
|
||||
```
|
||||
|
||||
@@ -142,7 +142,25 @@ print(f'[graphify update] Merged extraction written ({len(merged_out[\"nodes\"])
|
||||
# root= matches the build_merge call above so the manifest keys stay relative to
|
||||
# the scan root — portable across clones/machines, so --update keeps matching
|
||||
# cached files instead of missing every one after a move (#1417).
|
||||
save_manifest(incremental['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output
|
||||
# THIS run (new_extraction is this run's fresh extraction, read above before the
|
||||
# merge overwrote the file): a changed doc whose chunk failed must stay unstamped
|
||||
# so the next --update re-queues it, otherwise it is marked done and its content
|
||||
# is lost forever (#2015). Mirrors the library extract path
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus).
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_manifest_files = _stamped_manifest_files(incremental['files'], new_extraction, Path('INPUT_PATH'))
|
||||
# Changed semantic files dispatched this run but NOT stamped had their chunk fail
|
||||
# or be omitted; clear any stale semantic_hash so they are re-queued (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in incremental.get('new_files', {}).items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus so in-root files newly excluded since last run
|
||||
# are dropped rather than masquerading as deletions; untouched rows preserved (#1908).
|
||||
_scan = {f for fl in incremental['files'].values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
print('[graphify update] Manifest saved.')
|
||||
"
|
||||
```
|
||||
|
||||
@@ -142,7 +142,25 @@ print(f'[graphify update] Merged extraction written ({len(merged_out[\"nodes\"])
|
||||
# root= matches the build_merge call above so the manifest keys stay relative to
|
||||
# the scan root — portable across clones/machines, so --update keeps matching
|
||||
# cached files instead of missing every one after a move (#1417).
|
||||
save_manifest(incremental['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output
|
||||
# THIS run (new_extraction is this run's fresh extraction, read above before the
|
||||
# merge overwrote the file): a changed doc whose chunk failed must stay unstamped
|
||||
# so the next --update re-queues it, otherwise it is marked done and its content
|
||||
# is lost forever (#2015). Mirrors the library extract path
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus).
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_manifest_files = _stamped_manifest_files(incremental['files'], new_extraction, Path('INPUT_PATH'))
|
||||
# Changed semantic files dispatched this run but NOT stamped had their chunk fail
|
||||
# or be omitted; clear any stale semantic_hash so they are re-queued (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in incremental.get('new_files', {}).items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus so in-root files newly excluded since last run
|
||||
# are dropped rather than masquerading as deletions; untouched rows preserved (#1908).
|
||||
_scan = {f for fl in incremental['files'].values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
print('[graphify update] Manifest saved.')
|
||||
"
|
||||
```
|
||||
|
||||
@@ -142,7 +142,25 @@ print(f'[graphify update] Merged extraction written ({len(merged_out[\"nodes\"])
|
||||
# root= matches the build_merge call above so the manifest keys stay relative to
|
||||
# the scan root — portable across clones/machines, so --update keeps matching
|
||||
# cached files instead of missing every one after a move (#1417).
|
||||
save_manifest(incremental['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output
|
||||
# THIS run (new_extraction is this run's fresh extraction, read above before the
|
||||
# merge overwrote the file): a changed doc whose chunk failed must stay unstamped
|
||||
# so the next --update re-queues it, otherwise it is marked done and its content
|
||||
# is lost forever (#2015). Mirrors the library extract path
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus).
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_manifest_files = _stamped_manifest_files(incremental['files'], new_extraction, Path('INPUT_PATH'))
|
||||
# Changed semantic files dispatched this run but NOT stamped had their chunk fail
|
||||
# or be omitted; clear any stale semantic_hash so they are re-queued (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in incremental.get('new_files', {}).items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus so in-root files newly excluded since last run
|
||||
# are dropped rather than masquerading as deletions; untouched rows preserved (#1908).
|
||||
_scan = {f for fl in incremental['files'].values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
print('[graphify update] Manifest saved.')
|
||||
"
|
||||
```
|
||||
|
||||
@@ -142,7 +142,25 @@ print(f'[graphify update] Merged extraction written ({len(merged_out[\"nodes\"])
|
||||
# root= matches the build_merge call above so the manifest keys stay relative to
|
||||
# the scan root — portable across clones/machines, so --update keeps matching
|
||||
# cached files instead of missing every one after a move (#1417).
|
||||
save_manifest(incremental['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output
|
||||
# THIS run (new_extraction is this run's fresh extraction, read above before the
|
||||
# merge overwrote the file): a changed doc whose chunk failed must stay unstamped
|
||||
# so the next --update re-queues it, otherwise it is marked done and its content
|
||||
# is lost forever (#2015). Mirrors the library extract path
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus).
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_manifest_files = _stamped_manifest_files(incremental['files'], new_extraction, Path('INPUT_PATH'))
|
||||
# Changed semantic files dispatched this run but NOT stamped had their chunk fail
|
||||
# or be omitted; clear any stale semantic_hash so they are re-queued (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in incremental.get('new_files', {}).items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus so in-root files newly excluded since last run
|
||||
# are dropped rather than masquerading as deletions; untouched rows preserved (#1908).
|
||||
_scan = {f for fl in incremental['files'].values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
print('[graphify update] Manifest saved.')
|
||||
"
|
||||
```
|
||||
|
||||
@@ -142,7 +142,25 @@ print(f'[graphify update] Merged extraction written ({len(merged_out[\"nodes\"])
|
||||
# root= matches the build_merge call above so the manifest keys stay relative to
|
||||
# the scan root — portable across clones/machines, so --update keeps matching
|
||||
# cached files instead of missing every one after a move (#1417).
|
||||
save_manifest(incremental['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output
|
||||
# THIS run (new_extraction is this run's fresh extraction, read above before the
|
||||
# merge overwrote the file): a changed doc whose chunk failed must stay unstamped
|
||||
# so the next --update re-queues it, otherwise it is marked done and its content
|
||||
# is lost forever (#2015). Mirrors the library extract path
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus).
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_manifest_files = _stamped_manifest_files(incremental['files'], new_extraction, Path('INPUT_PATH'))
|
||||
# Changed semantic files dispatched this run but NOT stamped had their chunk fail
|
||||
# or be omitted; clear any stale semantic_hash so they are re-queued (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in incremental.get('new_files', {}).items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus so in-root files newly excluded since last run
|
||||
# are dropped rather than masquerading as deletions; untouched rows preserved (#1908).
|
||||
_scan = {f for fl in incremental['files'].values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
print('[graphify update] Manifest saved.')
|
||||
"
|
||||
```
|
||||
|
||||
@@ -142,7 +142,25 @@ print(f'[graphify update] Merged extraction written ({len(merged_out[\"nodes\"])
|
||||
# root= matches the build_merge call above so the manifest keys stay relative to
|
||||
# the scan root — portable across clones/machines, so --update keeps matching
|
||||
# cached files instead of missing every one after a move (#1417).
|
||||
save_manifest(incremental['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output
|
||||
# THIS run (new_extraction is this run's fresh extraction, read above before the
|
||||
# merge overwrote the file): a changed doc whose chunk failed must stay unstamped
|
||||
# so the next --update re-queues it, otherwise it is marked done and its content
|
||||
# is lost forever (#2015). Mirrors the library extract path
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus).
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_manifest_files = _stamped_manifest_files(incremental['files'], new_extraction, Path('INPUT_PATH'))
|
||||
# Changed semantic files dispatched this run but NOT stamped had their chunk fail
|
||||
# or be omitted; clear any stale semantic_hash so they are re-queued (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in incremental.get('new_files', {}).items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus so in-root files newly excluded since last run
|
||||
# are dropped rather than masquerading as deletions; untouched rows preserved (#1908).
|
||||
_scan = {f for fl in incremental['files'].values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
print('[graphify update] Manifest saved.')
|
||||
"
|
||||
```
|
||||
|
||||
@@ -142,7 +142,25 @@ print(f'[graphify update] Merged extraction written ({len(merged_out[\"nodes\"])
|
||||
# root= matches the build_merge call above so the manifest keys stay relative to
|
||||
# the scan root — portable across clones/machines, so --update keeps matching
|
||||
# cached files instead of missing every one after a move (#1417).
|
||||
save_manifest(incremental['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output
|
||||
# THIS run (new_extraction is this run's fresh extraction, read above before the
|
||||
# merge overwrote the file): a changed doc whose chunk failed must stay unstamped
|
||||
# so the next --update re-queues it, otherwise it is marked done and its content
|
||||
# is lost forever (#2015). Mirrors the library extract path
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus).
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_manifest_files = _stamped_manifest_files(incremental['files'], new_extraction, Path('INPUT_PATH'))
|
||||
# Changed semantic files dispatched this run but NOT stamped had their chunk fail
|
||||
# or be omitted; clear any stale semantic_hash so they are re-queued (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in incremental.get('new_files', {}).items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus so in-root files newly excluded since last run
|
||||
# are dropped rather than masquerading as deletions; untouched rows preserved (#1908).
|
||||
_scan = {f for fl in incremental['files'].values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
print('[graphify update] Manifest saved.')
|
||||
"
|
||||
```
|
||||
|
||||
@@ -142,7 +142,25 @@ print(f'[graphify update] Merged extraction written ({len(merged_out[\"nodes\"])
|
||||
# root= matches the build_merge call above so the manifest keys stay relative to
|
||||
# the scan root — portable across clones/machines, so --update keeps matching
|
||||
# cached files instead of missing every one after a move (#1417).
|
||||
save_manifest(incremental['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output
|
||||
# THIS run (new_extraction is this run's fresh extraction, read above before the
|
||||
# merge overwrote the file): a changed doc whose chunk failed must stay unstamped
|
||||
# so the next --update re-queues it, otherwise it is marked done and its content
|
||||
# is lost forever (#2015). Mirrors the library extract path
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus).
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_manifest_files = _stamped_manifest_files(incremental['files'], new_extraction, Path('INPUT_PATH'))
|
||||
# Changed semantic files dispatched this run but NOT stamped had their chunk fail
|
||||
# or be omitted; clear any stale semantic_hash so they are re-queued (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in incremental.get('new_files', {}).items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus so in-root files newly excluded since last run
|
||||
# are dropped rather than masquerading as deletions; untouched rows preserved (#1908).
|
||||
_scan = {f for fl in incremental['files'].values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
print('[graphify update] Manifest saved.')
|
||||
"
|
||||
```
|
||||
|
||||
@@ -142,7 +142,25 @@ print(f'[graphify update] Merged extraction written ({len(merged_out[\"nodes\"])
|
||||
# root= matches the build_merge call above so the manifest keys stay relative to
|
||||
# the scan root — portable across clones/machines, so --update keeps matching
|
||||
# cached files instead of missing every one after a move (#1417).
|
||||
save_manifest(incremental['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output
|
||||
# THIS run (new_extraction is this run's fresh extraction, read above before the
|
||||
# merge overwrote the file): a changed doc whose chunk failed must stay unstamped
|
||||
# so the next --update re-queues it, otherwise it is marked done and its content
|
||||
# is lost forever (#2015). Mirrors the library extract path
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus).
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_manifest_files = _stamped_manifest_files(incremental['files'], new_extraction, Path('INPUT_PATH'))
|
||||
# Changed semantic files dispatched this run but NOT stamped had their chunk fail
|
||||
# or be omitted; clear any stale semantic_hash so they are re-queued (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in incremental.get('new_files', {}).items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus so in-root files newly excluded since last run
|
||||
# are dropped rather than masquerading as deletions; untouched rows preserved (#1908).
|
||||
_scan = {f for fl in incremental['files'].values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
print('[graphify update] Manifest saved.')
|
||||
"
|
||||
```
|
||||
|
||||
@@ -142,7 +142,25 @@ print(f'[graphify update] Merged extraction written ({len(merged_out[\"nodes\"])
|
||||
# root= matches the build_merge call above so the manifest keys stay relative to
|
||||
# the scan root — portable across clones/machines, so --update keeps matching
|
||||
# cached files instead of missing every one after a move (#1417).
|
||||
save_manifest(incremental['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output
|
||||
# THIS run (new_extraction is this run's fresh extraction, read above before the
|
||||
# merge overwrote the file): a changed doc whose chunk failed must stay unstamped
|
||||
# so the next --update re-queues it, otherwise it is marked done and its content
|
||||
# is lost forever (#2015). Mirrors the library extract path
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus).
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_manifest_files = _stamped_manifest_files(incremental['files'], new_extraction, Path('INPUT_PATH'))
|
||||
# Changed semantic files dispatched this run but NOT stamped had their chunk fail
|
||||
# or be omitted; clear any stale semantic_hash so they are re-queued (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in incremental.get('new_files', {}).items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus so in-root files newly excluded since last run
|
||||
# are dropped rather than masquerading as deletions; untouched rows preserved (#1908).
|
||||
_scan = {f for fl in incremental['files'].values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
print('[graphify update] Manifest saved.')
|
||||
"
|
||||
```
|
||||
|
||||
@@ -142,7 +142,25 @@ print(f'[graphify update] Merged extraction written ({len(merged_out[\"nodes\"])
|
||||
# root= matches the build_merge call above so the manifest keys stay relative to
|
||||
# the scan root — portable across clones/machines, so --update keeps matching
|
||||
# cached files instead of missing every one after a move (#1417).
|
||||
save_manifest(incremental['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output
|
||||
# THIS run (new_extraction is this run's fresh extraction, read above before the
|
||||
# merge overwrote the file): a changed doc whose chunk failed must stay unstamped
|
||||
# so the next --update re-queues it, otherwise it is marked done and its content
|
||||
# is lost forever (#2015). Mirrors the library extract path
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus).
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_manifest_files = _stamped_manifest_files(incremental['files'], new_extraction, Path('INPUT_PATH'))
|
||||
# Changed semantic files dispatched this run but NOT stamped had their chunk fail
|
||||
# or be omitted; clear any stale semantic_hash so they are re-queued (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in incremental.get('new_files', {}).items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus so in-root files newly excluded since last run
|
||||
# are dropped rather than masquerading as deletions; untouched rows preserved (#1908).
|
||||
_scan = {f for fl in incremental['files'].values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
print('[graphify update] Manifest saved.')
|
||||
"
|
||||
```
|
||||
|
||||
@@ -142,7 +142,25 @@ print(f'[graphify update] Merged extraction written ({len(merged_out[\"nodes\"])
|
||||
# root= matches the build_merge call above so the manifest keys stay relative to
|
||||
# the scan root — portable across clones/machines, so --update keeps matching
|
||||
# cached files instead of missing every one after a move (#1417).
|
||||
save_manifest(incremental['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output
|
||||
# THIS run (new_extraction is this run's fresh extraction, read above before the
|
||||
# merge overwrote the file): a changed doc whose chunk failed must stay unstamped
|
||||
# so the next --update re-queues it, otherwise it is marked done and its content
|
||||
# is lost forever (#2015). Mirrors the library extract path
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus).
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_manifest_files = _stamped_manifest_files(incremental['files'], new_extraction, Path('INPUT_PATH'))
|
||||
# Changed semantic files dispatched this run but NOT stamped had their chunk fail
|
||||
# or be omitted; clear any stale semantic_hash so they are re-queued (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in incremental.get('new_files', {}).items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus so in-root files newly excluded since last run
|
||||
# are dropped rather than masquerading as deletions; untouched rows preserved (#1908).
|
||||
_scan = {f for fl in incremental['files'].values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
print('[graphify update] Manifest saved.')
|
||||
"
|
||||
```
|
||||
|
||||
@@ -677,10 +677,19 @@ from graphify.detect import save_manifest
|
||||
|
||||
# Save manifest for --update
|
||||
detect = json.loads(Path('.graphify_detect.json').read_text())
|
||||
save_manifest(detect['files'], root='INPUT_PATH')
|
||||
extract = json.loads(Path('.graphify_extract.json').read_text())
|
||||
# Stamp only semantic files that produced output so a failed chunk is re-queued next run, not lost (#2015).
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_corpus = detect.get('all_files') or detect['files']
|
||||
_manifest_files = _stamped_manifest_files(_corpus, extract, Path('INPUT_PATH'))
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in detect['files'].items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
_scan = {f for fl in _corpus.values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
|
||||
# Update cumulative cost tracker
|
||||
extract = json.loads(Path('.graphify_extract.json').read_text())
|
||||
input_tok = extract.get('input_tokens', 0)
|
||||
output_tok = extract.get('output_tokens', 0)
|
||||
|
||||
|
||||
@@ -489,15 +489,37 @@ from graphify.detect import save_manifest
|
||||
|
||||
# Save manifest for --update
|
||||
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\"))
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
# In --update mode, 'all_files' carries the full corpus; 'files' is the changed
|
||||
# subset. Full-rebuild mode populates only 'files', so the fallback handles that.
|
||||
# root= relativizes the manifest keys to the scan root (same base as the build),
|
||||
# so the on-disk manifest is portable across clones/machines and a later --update
|
||||
# matches cached files instead of missing every one (#1417).
|
||||
save_manifest(detect.get('all_files') or detect['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output:
|
||||
# a detected file whose chunk failed or was omitted must stay unstamped so the
|
||||
# next --update re-queues it, otherwise it is marked done and its content is lost
|
||||
# forever (#2015). This mirrors the library extract path exactly
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus); do not stamp the
|
||||
# raw corpus. Code files are always stamped (AST is deterministic); only semantic
|
||||
# types are gated on output.
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_corpus = detect.get('all_files') or detect['files']
|
||||
_manifest_files = _stamped_manifest_files(_corpus, extract, Path('INPUT_PATH'))
|
||||
# Files dispatched this run (the changed subset) but NOT stamped above still carry
|
||||
# a stale semantic_hash from a prior run; clear it so detect_incremental re-queues
|
||||
# them instead of reading them as unchanged (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in detect['files'].items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus (not the stamp-filtered subset) so in-root
|
||||
# files newly excluded since last run are dropped rather than masquerading as
|
||||
# deletions; untouched files' prior rows are still preserved (#1908).
|
||||
_scan = {f for fl in _corpus.values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
|
||||
# Update cumulative cost tracker
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
input_tok = extract.get('input_tokens', 0)
|
||||
output_tok = extract.get('output_tokens', 0)
|
||||
|
||||
|
||||
@@ -795,10 +795,19 @@ from graphify.detect import save_manifest
|
||||
|
||||
# Save manifest for --update
|
||||
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text())
|
||||
save_manifest(detect['files'], root='INPUT_PATH')
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text())
|
||||
# Stamp only semantic files that produced output so a failed chunk is re-queued next run, not lost (#2015).
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_corpus = detect.get('all_files') or detect['files']
|
||||
_manifest_files = _stamped_manifest_files(_corpus, extract, Path('INPUT_PATH'))
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in detect['files'].items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
_scan = {f for fl in _corpus.values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
|
||||
# Update cumulative cost tracker
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text())
|
||||
input_tok = extract.get('input_tokens', 0)
|
||||
output_tok = extract.get('output_tokens', 0)
|
||||
|
||||
|
||||
@@ -142,7 +142,25 @@ print(f'[graphify update] Merged extraction written ({len(merged_out[\"nodes\"])
|
||||
# root= matches the build_merge call above so the manifest keys stay relative to
|
||||
# the scan root — portable across clones/machines, so --update keeps matching
|
||||
# cached files instead of missing every one after a move (#1417).
|
||||
save_manifest(incremental['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output
|
||||
# THIS run (new_extraction is this run's fresh extraction, read above before the
|
||||
# merge overwrote the file): a changed doc whose chunk failed must stay unstamped
|
||||
# so the next --update re-queues it, otherwise it is marked done and its content
|
||||
# is lost forever (#2015). Mirrors the library extract path
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus).
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_manifest_files = _stamped_manifest_files(incremental['files'], new_extraction, Path('INPUT_PATH'))
|
||||
# Changed semantic files dispatched this run but NOT stamped had their chunk fail
|
||||
# or be omitted; clear any stale semantic_hash so they are re-queued (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in incremental.get('new_files', {}).items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus so in-root files newly excluded since last run
|
||||
# are dropped rather than masquerading as deletions; untouched rows preserved (#1908).
|
||||
_scan = {f for fl in incremental['files'].values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
print('[graphify update] Manifest saved.')
|
||||
"
|
||||
```
|
||||
|
||||
@@ -851,6 +851,34 @@ def _is_manifest_root_fix_line(line: str) -> bool:
|
||||
return "save_manifest(" in line and "import" not in line
|
||||
|
||||
|
||||
def _is_manifest_stamp_fix_line(line: str) -> bool:
|
||||
"""Whether a line is part of the manifest over-stamping fix (#2015).
|
||||
|
||||
Step 9 stamped the whole detected corpus, so a semantic file whose chunk
|
||||
failed (or was omitted) was marked done and never re-queued on the next
|
||||
``--update`` — its content lost forever. The manifest is now built with
|
||||
``cli._stamped_manifest_files`` (only files that actually produced output)
|
||||
plus ``clear_semantic``/``scan_corpus``, mirroring the native
|
||||
``graphify extract`` path. The rooted ``save_manifest`` call itself is
|
||||
covered by ``_is_manifest_root_fix_line``; these are the added helper import
|
||||
and derivation lines, plus the single ``#2015`` explanatory comment.
|
||||
"""
|
||||
stripped = line.strip()
|
||||
return (
|
||||
"_stamped_manifest_files" in stripped
|
||||
or stripped.startswith((
|
||||
"_corpus =",
|
||||
"_manifest_files =",
|
||||
"_sem_types =",
|
||||
"_dispatched =",
|
||||
"_stamped =",
|
||||
"_cleared =",
|
||||
"_scan =",
|
||||
))
|
||||
or (stripped.startswith("#") and "#2015" in stripped)
|
||||
)
|
||||
|
||||
|
||||
def _is_no_api_key_fix_line(line: str) -> bool:
|
||||
"""Whether a line is part of the "no API key required" clarity (#1461).
|
||||
|
||||
@@ -931,6 +959,7 @@ _SANCTIONED_MONOLITH_DIFFS = (
|
||||
_is_cache_unlink_fix_line,
|
||||
_is_zero_node_guard_fix_line,
|
||||
_is_manifest_root_fix_line,
|
||||
_is_manifest_stamp_fix_line,
|
||||
_is_no_api_key_fix_line,
|
||||
_is_shebang_allowlist_fix_line,
|
||||
_is_obsidian_usage_comment_line,
|
||||
|
||||
Reference in New Issue
Block a user