@@ -414,8 +414,10 @@ def test_fetch_pdf_picks_unique_name_when_target_exists(tmp_path):
414414 assert result .read_bytes () == body
415415
416416
417- def test_fetch_html_picks_unique_name_when_target_exists (tmp_path ):
418- """Two blog posts both titled 'Introduction' must NOT collide."""
417+ def test_fetch_html_picks_unique_name_when_target_exists (tmp_path , capsys ):
418+ """Two blog posts both titled 'Introduction' must NOT collide. The
419+ user-facing 'Saved: ...' echo must also reflect the renamed path —
420+ otherwise the message lies about where the file actually went."""
419421 raw_dir = tmp_path / "raw"
420422 raw_dir .mkdir ()
421423 (raw_dir / "Introduction.md" ).write_text ("first blog post body" )
@@ -435,6 +437,8 @@ def test_fetch_html_picks_unique_name_when_target_exists(tmp_path):
435437 assert (raw_dir / "Introduction.md" ).read_text () == "first blog post body"
436438 assert result == raw_dir / "Introduction_2.md"
437439 assert result .read_text () == second_md
440+ out = capsys .readouterr ().out
441+ assert "Saved: raw/Introduction_2.md" in out
438442
439443
440444def test_fetch_pdf_uses_post_redirect_url_for_filename (tmp_path ):
@@ -458,10 +462,10 @@ def test_fetch_pdf_uses_post_redirect_url_for_filename(tmp_path):
458462 assert result .name == "great-paper.pdf"
459463
460464
461- def test_add_single_file_returns_true_on_success (tmp_path ):
462- """The new bool return contract: True when the file was actually
463- indexed. Used by the URL-ingest cleanup path to decide whether the
464- just-downloaded file in raw/ should be unlinked ."""
465+ def test_add_single_file_returns_added_on_success (tmp_path ):
466+ """Tri-state return contract: ``"added"`` when the file was newly
467+ indexed. URL-ingest uses this to decide whether to keep / unlink
468+ the just-downloaded file."""
465469 from openkb .cli import add_single_file
466470 from openkb .converter import ConvertResult
467471
@@ -487,12 +491,12 @@ def test_add_single_file_returns_true_on_success(tmp_path):
487491
488492 with patch ("openkb.cli.convert_document" , return_value = mock_result ), \
489493 patch ("openkb.cli.asyncio.run" ):
490- added = add_single_file (doc , tmp_path )
494+ outcome = add_single_file (doc , tmp_path )
491495
492- assert added is True
496+ assert outcome == "added"
493497
494498
495- def test_add_single_file_returns_false_on_skip (tmp_path ):
499+ def test_add_single_file_returns_skipped_on_dedup (tmp_path ):
496500 from openkb .cli import add_single_file
497501 from openkb .converter import ConvertResult
498502
@@ -505,14 +509,48 @@ def test_add_single_file_returns_false_on_skip(tmp_path):
505509
506510 skipped = ConvertResult (skipped = True )
507511 with patch ("openkb.cli.convert_document" , return_value = skipped ):
508- added = add_single_file (doc , tmp_path )
512+ outcome = add_single_file (doc , tmp_path )
509513
510- assert added is False
514+ assert outcome == "skipped"
515+
516+
517+ def test_add_single_file_returns_failed_on_pipeline_error (tmp_path ):
518+ """A pipeline failure (e.g. transient LLM error during compilation)
519+ must be distinguishable from dedup-skip, so URL-ingest can preserve
520+ the raw file for retry instead of deleting it."""
521+ from openkb .cli import add_single_file
522+ from openkb .converter import ConvertResult
523+
524+ (tmp_path / ".openkb" ).mkdir ()
525+ (tmp_path / ".openkb" / "config.yaml" ).write_text ("model: gpt-4o-mini\n " )
526+ (tmp_path / ".openkb" / "hashes.json" ).write_text ("{}" )
527+ (tmp_path / "raw" ).mkdir ()
528+ (tmp_path / "wiki" / "summaries" ).mkdir (parents = True )
529+ (tmp_path / "wiki" / "sources" ).mkdir (parents = True )
530+ (tmp_path / "wiki" / "log.md" ).write_text ("" )
531+
532+ doc = tmp_path / "raw" / "x.md"
533+ doc .write_text ("# Hello" )
534+ source_path = tmp_path / "wiki" / "sources" / "x.md"
535+ source_path .write_text ("# Hello" )
536+
537+ mock_result = ConvertResult (
538+ raw_path = doc , source_path = source_path ,
539+ is_long_doc = False , file_hash = "cafe" * 16 ,
540+ )
541+
542+ # Make both compile attempts raise to drive the failure path.
543+ with patch ("openkb.cli.convert_document" , return_value = mock_result ), \
544+ patch ("openkb.cli.asyncio.run" , side_effect = RuntimeError ("LLM 503" )), \
545+ patch ("openkb.cli.time.sleep" ):
546+ outcome = add_single_file (doc , tmp_path )
547+
548+ assert outcome == "failed"
511549
512550
513551def test_url_ingest_cleans_up_orphan_on_dedup_skip (tmp_path , monkeypatch ):
514552 """End-to-end: when the URL-fetched file is already in the registry,
515- add_single_file returns False and the CLI must unlink it from raw/
553+ add_single_file returns "skipped" and the CLI unlinks it from raw/
516554 so the user doesn't accumulate untracked duplicates."""
517555 from click .testing import CliRunner
518556 from openkb .cli import cli
@@ -541,3 +579,44 @@ def test_url_ingest_cleans_up_orphan_on_dedup_skip(tmp_path, monkeypatch):
541579 assert "[SKIP]" in result .output
542580 # Orphan cleanup: the URL-fetched file must be gone from raw/.
543581 assert not fetched_path .exists ()
582+
583+
584+ def test_url_ingest_keeps_raw_file_on_pipeline_failure (tmp_path ):
585+ """The point of the tri-state return: a pipeline failure (e.g. LLM
586+ timeout during compilation) must NOT delete the downloaded file —
587+ the user can retry without re-downloading, and we don't lose data
588+ when indexing has already succeeded but compilation hasn't."""
589+ from click .testing import CliRunner
590+ from openkb .cli import cli
591+ from openkb .converter import ConvertResult
592+
593+ (tmp_path / ".openkb" ).mkdir ()
594+ (tmp_path / ".openkb" / "config.yaml" ).write_text ("model: gpt-4o-mini\n " )
595+ (tmp_path / ".openkb" / "hashes.json" ).write_text ("{}" )
596+ (tmp_path / "raw" ).mkdir ()
597+ (tmp_path / "wiki" / "summaries" ).mkdir (parents = True )
598+ (tmp_path / "wiki" / "sources" ).mkdir (parents = True )
599+ (tmp_path / "wiki" / "log.md" ).write_text ("" )
600+
601+ fetched_path = tmp_path / "raw" / "paper.pdf"
602+ fetched_path .write_bytes (b"%PDF-fake" )
603+ source_path = tmp_path / "wiki" / "sources" / "paper.md"
604+ source_path .write_text ("# fake" )
605+
606+ mock_result = ConvertResult (
607+ raw_path = fetched_path , source_path = source_path ,
608+ is_long_doc = False , file_hash = "cafe" * 16 ,
609+ )
610+
611+ runner = CliRunner ()
612+ with patch ("openkb.cli._find_kb_dir" , return_value = tmp_path ), \
613+ patch ("openkb.url_ingest.fetch_url_to_raw" , return_value = fetched_path ), \
614+ patch ("openkb.cli.convert_document" , return_value = mock_result ), \
615+ patch ("openkb.cli.asyncio.run" , side_effect = RuntimeError ("LLM 503" )), \
616+ patch ("openkb.cli.time.sleep" ):
617+ result = runner .invoke (cli , ["add" , "https://example.com/paper.pdf" ])
618+
619+ assert result .exit_code == 0 , result .output
620+ assert "[ERROR] Compilation failed" in result .output
621+ # The raw file must be preserved so the user can retry.
622+ assert fetched_path .exists ()
0 commit comments