Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion README.md
Original file line number Diff line number Diff line change
Expand Up @@ -171,7 +171,7 @@ OpenKB commands fall into two layers: the **wiki foundation** (compile + manage
| Command | Description |
| ------------------------------------------------------------ | --------------------------------------------------------------------------------------- |
| `openkb init` | Initialize a new knowledge base (interactive) |
| <code>openkb&nbsp;add&nbsp;&lt;file_or_dir_or_URL&gt;</code> | Add files, directories, or URLs and compile to wiki (URL content type is auto-detected) |
| <code>openkb&nbsp;add&nbsp;[file_or_dir_or_URL]</code> | Add files, directories, or URLs and compile to wiki (URL content type is auto-detected); omit the argument to process `raw/` recursively with a summary |
| `openkb list` | List indexed documents and concepts |
| `openkb status` | Show knowledge base stats |
| `openkb watch` | Watch `raw/` and auto-compile new files |
Expand Down
146 changes: 118 additions & 28 deletions openkb/cli.py
Original file line number Diff line number Diff line change
Expand Up @@ -452,6 +452,59 @@ def add_single_file(
return _add_single_file_locked(file_path, kb_dir, stage=stage, bundle=bundle)


def _delete_if_auto_cleanup_enabled(
file_path: Path, status: Literal["added", "skipped", "failed"], config: dict
) -> bool:
"""Delete file if auto_delete_added_files is enabled and ingestion succeeded/skipped.

Deletes on both "added" (successful ingestion) and "skipped" (duplicate already
in KB) to keep raw/ directory clean. Preserves files on "failed" to allow retries.

Args:
file_path: Path to the file to potentially delete.
status: Result status from add_single_file ("added", "skipped", or "failed").
config: Configuration dict (typically from resolve_effective_config).

Returns:
True if file was deleted, False otherwise.
"""
if status in ("added", "skipped") and config.get("auto_delete_added_files", False):
try:
file_path.unlink(missing_ok=True)
return True
except Exception as exc:
logger.warning(f"Failed to delete {file_path.name}: {exc}")
return False
return False


def _cleanup_empty_directories(start_dir: Path) -> int:
"""Recursively delete empty directories under start_dir.

Walks from deepest subdirectories up, deleting directories that become
empty after file cleanup.

Args:
start_dir: Root directory to clean up (e.g., kb_dir / "raw").

Returns:
Number of directories deleted.
"""
deleted_count = 0
try:
for directory in sorted(start_dir.rglob("*"), key=lambda p: len(p.parts), reverse=True):
if directory.is_dir() and directory != start_dir:
try:
if not list(directory.iterdir()):
directory.rmdir()
deleted_count += 1
except OSError:
pass
except Exception as exc:
logger.warning(f"Error during directory cleanup: {exc}")
return deleted_count


def _add_single_file_locked(
file_path: Path, kb_dir: Path, *, stage: bool = True, bundle=None
) -> Literal["added", "skipped", "failed"]:
Expand Down Expand Up @@ -1083,9 +1136,20 @@ def add(ctx, path, from_pageindex_cloud):
magic-byte sniff) are saved as ``.pdf``; HTML responses are run
through trafilatura's main-content extractor and saved as ``.md``.

If PATH is omitted (and --from-pageindex-cloud is not used), the KB's
``raw/`` directory is used instead: all supported files under it are
walked recursively and added, an aggregated summary (added/skipped/
failed/deleted counts) is printed at the end, and — if
``auto_delete_added_files`` is enabled — now-empty subdirectories under
``raw/`` are cleaned up.

Alternatively, pass --from-pageindex-cloud <DOC_ID> to import a document
that is already indexed in PageIndex Cloud, with no local file. Requires
the PAGEINDEX_API_KEY environment variable.

If ``auto_delete_added_files`` is enabled in config.yaml, files are
automatically deleted after ingestion (both on successful addition and
on skip/duplicate).
"""
kb_dir = _find_kb_dir(ctx.obj.get("kb_dir_override"))
if kb_dir is None:
Expand All @@ -1102,33 +1166,41 @@ def add(ctx, path, from_pageindex_cloud):
ctx.exit(1)
return

if path is None:
click.echo("Provide a PATH or use --from-pageindex-cloud <DOC_ID>.")
return

# URL ingest: download into raw/ first, then call add_single_file explicitly.
# Keep staged conversion enabled so converted source artifacts do not touch
# the live KB before the mutation snapshot exists. The tri-state outcome
# still lets us clean up the just-downloaded raw file on dedup.
from openkb.url_ingest import looks_like_url, fetch_url_to_raw
config = resolve_effective_config(kb_dir)[0]

if looks_like_url(path):
fetched = fetch_url_to_raw(path, kb_dir)
if fetched is None:
if path is None:
# No PATH given: default to the KB's raw/ directory (mirrors the
# former standalone `add-all` command).
target = kb_dir / "raw"
if not target.is_dir():
click.echo(f"No raw/ directory found at {target}")
return
else:
# URL ingest: download into raw/ first, then call add_single_file explicitly.
# Keep staged conversion enabled so converted source artifacts do not touch
# the live KB before the mutation snapshot exists. The tri-state outcome
# still lets us clean up the just-downloaded raw file on dedup.
from openkb.url_ingest import looks_like_url, fetch_url_to_raw

if looks_like_url(path):
fetched = fetch_url_to_raw(path, kb_dir)
if fetched is None:
return
outcome = add_single_file(fetched, kb_dir)
# Only clean up on dedup-skip. On "failed" we keep the file so
# the user can retry (e.g. transient LLM error during compile)
# without re-downloading — and so they don't lose data when
# indexing has already succeeded but compilation didn't.
if outcome == "skipped":
fetched.unlink(missing_ok=True)
else:
_delete_if_auto_cleanup_enabled(fetched, outcome, config)
return
outcome = add_single_file(fetched, kb_dir)
# Only clean up on dedup-skip. On "failed" we keep the file so
# the user can retry (e.g. transient LLM error during compile)
# without re-downloading — and so they don't lose data when
# indexing has already succeeded but compilation didn't.
if outcome == "skipped":
fetched.unlink(missing_ok=True)
return

target = Path(path)
if not target.exists():
click.echo(f"Path does not exist: {path}")
return
target = Path(path)
if not target.exists():
click.echo(f"Path does not exist: {path}")
return

if target.is_dir():
files = [
Expand All @@ -1137,21 +1209,39 @@ def add(ctx, path, from_pageindex_cloud):
if f.is_file() and f.suffix.lower() in SUPPORTED_EXTENSIONS
]
if not files:
click.echo(f"No supported files found in {path}.")
click.echo(f"No supported files found in {target}.")
return
total = len(files)
click.echo(f"Found {total} supported file(s) in {path}.")
added = skipped = failed = deleted = dirs_deleted = 0
click.echo(f"Found {total} supported file(s) in {target}.")
for i, f in enumerate(files, 1):
click.echo(f"\n[{i}/{total}] ", nl=False)
add_single_file(f, kb_dir)
outcome = add_single_file(f, kb_dir)
if outcome == "added":
added += 1
elif outcome == "skipped":
skipped += 1
else:
failed += 1
if _delete_if_auto_cleanup_enabled(f, outcome, config):
deleted += 1

if config.get("auto_delete_added_files", False):
dirs_deleted = _cleanup_empty_directories(target)

summary = f"Added: {added}, Skipped: {skipped}, Failed: {failed}, Deleted: {deleted}"
if dirs_deleted > 0:
summary += f", Empty dirs cleaned: {dirs_deleted}"
click.echo(f"\n\nSummary: {summary}")
else:
if target.suffix.lower() not in SUPPORTED_EXTENSIONS:
click.echo(
f"Unsupported file type: {target.suffix}. "
f"Supported: {', '.join(sorted(SUPPORTED_EXTENSIONS))}"
)
return
add_single_file(target, kb_dir)
outcome = add_single_file(target, kb_dir)
_delete_if_auto_cleanup_enabled(target, outcome, config)


def _stream_to_tty() -> bool:
Expand Down
1 change: 1 addition & 0 deletions openkb/config.py
Original file line number Diff line number Diff line change
Expand Up @@ -36,6 +36,7 @@
# global/KB list overrides it wholesale; resolve_entity_types cleans the
# effective value on read.
"entity_types": list(DEFAULT_ENTITY_TYPES),
"auto_delete_added_files": False,
}

GLOBAL_CONFIG_DIR = Path.home() / ".config" / "openkb"
Expand Down
77 changes: 75 additions & 2 deletions tests/test_add_command.py
Original file line number Diff line number Diff line change
Expand Up @@ -284,6 +284,30 @@ def test_add_directory_stops_after_dirty_rollback(self, tmp_path):
mock_add.assert_called_once()
assert mock_add.call_args.args[0].name == "a.md"

def test_add_directory_prints_summary_and_cleans_empty_dirs(self, tmp_path):
kb_dir = self._setup_kb(tmp_path)
(kb_dir / ".openkb" / "config.yaml").write_text(
"model: gpt-4o-mini\nauto_delete_added_files: true\n", encoding="utf-8"
)
docs_dir = tmp_path / "docs"
sub_dir = docs_dir / "sub"
sub_dir.mkdir(parents=True)
doc = sub_dir / "a.md"
doc.write_text("# A")

runner = CliRunner()
with (
patch("openkb.cli.add_single_file", return_value="added"),
patch("openkb.cli._find_kb_dir", return_value=kb_dir),
):
result = runner.invoke(cli, ["add", str(docs_dir)])
assert not doc.exists()
assert not sub_dir.exists()
assert (
"Summary: Added: 1, Skipped: 0, Failed: 0, Deleted: 1, Empty dirs cleaned: 1"
in result.output
)

def test_add_unsupported_extension(self, tmp_path):
kb_dir = self._setup_kb(tmp_path)
doc = tmp_path / "file.xyz"
Expand Down Expand Up @@ -448,12 +472,61 @@ def test_add_rejects_path_and_cloud_together(self, tmp_path):
mock_imp.assert_not_called()
mock_add.assert_not_called()

def test_add_requires_path_or_cloud(self, tmp_path):
def test_add_no_path_processes_raw_dir_by_default(self, tmp_path):
kb_dir = self._setup_kb(tmp_path)
(kb_dir / "raw" / "a.md").write_text("# A")
(kb_dir / "raw" / "b.txt").write_text("B content")
(kb_dir / "raw" / "ignore.xyz").write_text("skip me")

runner = CliRunner()
with (
patch("openkb.cli.add_single_file", return_value="added") as mock_add,
patch("openkb.cli._find_kb_dir", return_value=kb_dir),
):
result = runner.invoke(cli, ["add"])
assert mock_add.call_count == 2
called_names = {call.args[0].name for call in mock_add.call_args_list}
assert called_names == {"a.md", "b.txt"}
assert "Summary: Added: 2, Skipped: 0, Failed: 0, Deleted: 0" in result.output

def test_add_no_path_empty_raw_dir_reports_no_files(self, tmp_path):
kb_dir = self._setup_kb(tmp_path)
runner = CliRunner()
with patch("openkb.cli._find_kb_dir", return_value=kb_dir):
result = runner.invoke(cli, ["add"])
assert "Provide a PATH" in result.output
assert "No supported files found" in result.output

def test_add_no_path_no_cloud_missing_raw_dir_errors(self, tmp_path):
# KB without a raw/ directory (e.g. deleted by the user).
openkb_dir = tmp_path / ".openkb"
openkb_dir.mkdir()
(openkb_dir / "config.yaml").write_text("model: gpt-4o-mini\n")
(openkb_dir / "hashes.json").write_text(json.dumps({}))

runner = CliRunner()
with patch("openkb.cli._find_kb_dir", return_value=tmp_path):
result = runner.invoke(cli, ["add"])
assert "No raw/ directory found" in result.output

def test_add_no_path_auto_cleanup_deletes_files_and_empty_dirs(self, tmp_path):
kb_dir = self._setup_kb(tmp_path)
(kb_dir / ".openkb" / "config.yaml").write_text(
"model: gpt-4o-mini\nauto_delete_added_files: true\n", encoding="utf-8"
)
sub_dir = kb_dir / "raw" / "sub"
sub_dir.mkdir()
doc = sub_dir / "a.md"
doc.write_text("# A")

runner = CliRunner()
with (
patch("openkb.cli.add_single_file", return_value="added"),
patch("openkb.cli._find_kb_dir", return_value=kb_dir),
):
result = runner.invoke(cli, ["add"])
assert not doc.exists()
assert not sub_dir.exists()
assert "Empty dirs cleaned: 1" in result.output


class TestImportFromPageindexCloud:
Expand Down