diff --git a/docs/development/lexical-retrieval.md b/docs/development/lexical-retrieval.md new file mode 100644 index 0000000000..9f4f6e8730 --- /dev/null +++ b/docs/development/lexical-retrieval.md @@ -0,0 +1,45 @@ +# Shared lexical retrieval / 共享词法检索 + +`loopx.lexical_retrieval.score_bm25(documents, query)` is a standard-library-only +scorer, reused by the self-repair lookup and the financial-evidence consumer. +It accepts document strings and returns one score plus matched terms per input, +and corpus-wide unmatched query terms. It keeps input positions; the consumer +owns filtering, stable keys, tie-breaks, exact-match precedence and pagination. + +`loopx.lexical_retrieval.score_bm25(documents, query)` 仅依赖标准库,自修复查询和 +金融证据消费者共用它。输入文档字符串,返回与每个输入对应的分数、匹配词项,以及整个 +语料未命中的查询词项。保留输入位置;筛选、稳定身份、同分排序、准确匹配优先级和分页 +由消费者负责。 + +```python +from loopx.lexical_retrieval import score_bm25 + +result = score_bm25(["lease recovery proof", "display refresh"], "recovery proof") +assert result.documents[0].score > result.documents[1].score +``` + +The fixed algorithm is BM25 with k1=1.2, b=.75 and positive Lucene IDF. +Casefolded Unicode word tokenization splits underscores. Repeated query terms +do not boost scores. No synonyms, embeddings, segmentation service, discovery, +storage, configuration, network or Goal authority are introduced. Ranking is +corpus-relative; a score does not certify a fact, risk, investment edge or trade. +Historical consumers must supply only their visible corpus: future documents +would otherwise change document frequencies even if hidden from the results. + +固定算法为 BM25,k1=1.2、b=.75、Lucene 正 IDF。Unicode 词项经 casefold, +下划线分隔;重复查询词项不增加权重。不引入同义词、嵌入、分词服务、发现、存储、配置、 +网络或 Goal 权威。分数相对于语料,不认证事实、风险、投资优势或交易资格。历史查询的 +消费者必须只传入可见语料,否则未来文档即使不展示,也会改变文档频率。 + +The repair skill retains its parser, exact pattern/code precedence and complete +guidance expansion. The existing workflow installer and wheel bundle the +canonical scorer beside the script, so isolated execution still works without +LoopX on PATH. The packaged copy is a build/install artifact, not another source +implementation. Finance retains canonical instruments, literal match rules, +source/clock/lifecycle boundaries and decision authority in its own consumer. +No new automatic capability hook or UI/Lark configuration is enabled. + +自修复 skill 保留解析器、准确 pattern/code 优先级和完整正文展开。现有 workflow +installer 和 wheel 将规范评分器打包到脚本旁,隔离运行仍无需 PATH 中的 LoopX。 +打包副本是构建/安装产物,不是第二份实现。金融消费者保留规范资产、字面匹配规则、 +信源/时点/lifecycle 边界和决策权威。不启用新的自动 capability hook 或 UI/Lark 配置。 diff --git a/loopx/lexical_retrieval.py b/loopx/lexical_retrieval.py new file mode 100644 index 0000000000..260573171f --- /dev/null +++ b/loopx/lexical_retrieval.py @@ -0,0 +1,52 @@ +"""Dependency-free lexical scoring; callers own identity, authority and selection. + +No storage, network, Goal state or source discovery. Scores are corpus-relative +recall order, never confidence. Kept standalone for the packaged repair skill. +""" + +from __future__ import annotations + +from collections import Counter +from math import log1p +import re +from typing import Iterable, NamedTuple + + +class BM25Hit(NamedTuple): + score: float + matched_terms: tuple[str, ...] + + +class BM25Scores(NamedTuple): + # One entry per input document; callers retain their own keys/tie-breaks. + documents: tuple[BM25Hit, ...] + unmatched_terms: tuple[str, ...] + + +def lexical_tokens(text: str) -> list[str]: + """Casefold Unicode words; split underscore identifiers, without aliases.""" + return re.findall(r"[^\W_]+", text.casefold()) + + +def score_bm25(documents: Iterable[str], query: str) -> BM25Scores: + """BM25 k1=1.2, b=.75, positive Lucene IDF; no query-frequency boost. + + Every input receives a score, including zero-match documents. Filtering, + pagination, exact-ID precedence and stable ordering belong to the caller. + """ + terms = set(lexical_tokens(query)) + counts = [Counter(lexical_tokens(text)) for text in documents] + lengths = [sum(document.values()) for document in counts] + average = sum(lengths) / len(lengths) if lengths else 1 + frequencies = Counter(term for document in counts for term in document) + scores = [] + for document, length in zip(counts, lengths): + matched = tuple(sorted(terms & document.keys())) + score = sum( + log1p((len(counts) - frequencies[term] + .5) / (frequencies[term] + .5)) + * document[term] * 2.2 + / (document[term] + 1.2 * (.25 + .75 * length / (average or 1))) + for term in matched + ) + scores.append(BM25Hit(score, matched)) + return BM25Scores(tuple(scores), tuple(sorted(terms - frequencies.keys()))) diff --git a/loopx/workflow_skill_install.py b/loopx/workflow_skill_install.py index 6b0c49f77a..23c483fb06 100644 --- a/loopx/workflow_skill_install.py +++ b/loopx/workflow_skill_install.py @@ -45,11 +45,12 @@ _INSTALL_LOCK_STEM = ".loopx-workflow-skills" -def _valid_source_root(path: Path) -> bool: +def _valid_source_root(path: Path, *, require_bundled_scorer: bool = False) -> bool: return all( (path / skill_id / "SKILL.md").is_file() for skill_id in PACKAGED_HOST_SKILL_IDS - ) + ) and (not require_bundled_scorer or + (path / "loopx-self-repair/scripts/lexical_retrieval.py").is_file()) def resolve_workflow_skill_source() -> dict[str, Any]: @@ -69,7 +70,7 @@ def resolve_workflow_skill_source() -> dict[str, Any]: bundle_root / "share" / "loopx" / "skills", bundle_root / "skills", ) - if _valid_source_root(candidate) + if _valid_source_root(candidate, require_bundled_scorer=True) ), None, ) @@ -166,6 +167,16 @@ def _exclusive_install_lock(skills_dir: Path) -> Iterator[None]: def _install_one_skill(source: Path, target: Path) -> str: + # Wheel/frozen skill data already includes this resource. A source install + # materializes the same canonical file, then uses the existing hash/atomic + # install path so repeated installs and drift readback include the scorer. + if source.name == "loopx-self-repair" and not (source / "scripts/lexical_retrieval.py").is_file(): + with tempfile.TemporaryDirectory(prefix="loopx-repair-skill-") as staging: + bundle = Path(staging) / source.name + shutil.copytree(source, bundle) + shutil.copy2(Path(__file__).with_name("lexical_retrieval.py"), + bundle / "scripts/lexical_retrieval.py") + return _install_one_skill(bundle, target) ignored = (SKILL_VERSION_MARKER_FILENAME,) if target.is_dir() and hash_skill_tree( source, ignored_relative_paths=ignored diff --git a/pyproject.toml b/pyproject.toml index 99b3f2fd2d..b07a1d2f26 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -131,6 +131,7 @@ include = ["loopx*"] ] "share/loopx/skills/loopx-self-repair/scripts" = [ "skills/loopx-self-repair/scripts/find_pattern.py", + "loopx/lexical_retrieval.py", ] [tool.pytest.ini_options] diff --git a/skills/loopx-self-repair/scripts/find_pattern.py b/skills/loopx-self-repair/scripts/find_pattern.py index 02acafd490..55158ca2ec 100644 --- a/skills/loopx-self-repair/scripts/find_pattern.py +++ b/skills/loopx-self-repair/scripts/find_pattern.py @@ -3,9 +3,8 @@ from __future__ import annotations import argparse -from collections import Counter +import importlib.util import json -from math import log1p from pathlib import Path import re @@ -13,6 +12,16 @@ CATALOG = Path(__file__).resolve().parents[1] / "references" / "repair-patterns.md" FIELDS = ("pattern", "symptoms", "evidence", "likely_root", "durable_repair") +# The installer/wheel bundles the canonical stdlib-only module beside this +# script. Source checkout execution reads that same file from the package. +_scorer_path = Path(__file__).with_name("lexical_retrieval.py") +if not _scorer_path.is_file(): + _scorer_path = Path(__file__).resolve().parents[3] / "loopx" / "lexical_retrieval.py" +_spec = importlib.util.spec_from_file_location("repair_lexical_retrieval", _scorer_path) +assert _spec is not None and _spec.loader is not None +_scorer = importlib.util.module_from_spec(_spec) +_spec.loader.exec_module(_scorer) + def read_patterns(path: Path) -> list[dict[str, str]]: """Keep complete table cells and prose appendices; never silently drop a row.""" @@ -48,25 +57,15 @@ def read_patterns(path: Path) -> list[dict[str, str]]: def search_patterns( patterns: list[dict[str, str]], query: str, *, offset: int, limit: int ) -> dict[str, object]: - # BM25 (k1=1.2, b=0.75), with Lucene's positive IDF. Whole identifiers - # remain addressable by --id; tokenization also exposes their components. - terms = set(re.findall(r"[^\W_]+", query.casefold())) - documents = [Counter(re.findall(r"[^\W_]+", "\n".join(row.values()).casefold())) for row in patterns] - lengths = [sum(doc.values()) for doc in documents] - average = sum(lengths) / len(lengths) if lengths else 1 - frequencies = Counter(term for doc in documents for term in doc) + texts = ["\n".join(row.values()) for row in patterns] + ranking = _scorer.score_bm25(texts, query) scored = [] - for row, document, length in zip(patterns, documents, lengths): - matched_terms = sorted(terms & document.keys()) - score = sum( - log1p((len(patterns) - frequencies[term] + 0.5) / (frequencies[term] + 0.5)) - * document[term] * 2.2 - / (document[term] + 1.2 * (0.25 + 0.75 * length / (average or 1))) - for term in matched_terms - ) + for row, text, hit in zip(patterns, texts, ranking.documents): + score = hit.score + matched_terms = list(hit.matched_terms) exact_id = row["pattern"].casefold() == query.strip().casefold() exact_code = any(query.strip().casefold() == code.casefold() - for code in re.findall(r"`([^`\n]+)`", "\n".join(row.values()))) + for code in re.findall(r"`([^`\n]+)`", text)) if score > 0 or exact_id or not query: scored.append((exact_id, exact_code, score, row, matched_terms)) # Stable id tie-break keeps pagination reproducible; scores are not confidence. @@ -82,7 +81,7 @@ def search_patterns( "ok": True, "query": query, "match_mode": "bm25_exact_first" if query else "catalog_order", - "unmatched_terms": sorted(terms - frequencies.keys()), + "unmatched_terms": list(ranking.unmatched_terms), "total_matches": len(matched), "offset": offset, "next_offset": end if end < len(matched) else None, diff --git a/tests/test_lexical_retrieval.py b/tests/test_lexical_retrieval.py new file mode 100644 index 0000000000..1dfbfc584f --- /dev/null +++ b/tests/test_lexical_retrieval.py @@ -0,0 +1,28 @@ +"""Neutral scoring and caller-owned relevance boundaries.""" + +from loopx.lexical_retrieval import lexical_tokens, score_bm25 + + +def test_rare_multiple_terms_beat_repeated_generic_word(): + result = score_bm25(["lease recovery proof", "lease " * 20, "unrelated"], + "lease recovery absent") + assert result.documents[0].score > result.documents[1].score > 0 + assert result.documents[2].score == 0 + assert result.documents[0].matched_terms == ("lease", "recovery") + assert result.unmatched_terms == ("absent",) + + +def test_zero_empty_and_generator_corpora_keep_input_identity(): + assert score_bm25([], "missing").documents == () + assert score_bm25([], "missing").unmatched_terms == ("missing",) + result = score_bm25((text for text in ["", "cash", "cash"]), "cash") + assert len(result.documents) == 3 + assert result.documents[0].score == 0 + assert result.documents[1] == result.documents[2] + assert all(hit.score == 0 for hit in score_bm25(["cash"], "").documents) + + +def test_tokens_are_literal_and_query_repetition_does_not_boost_scores(): + assert lexical_tokens("STALE_Proof [现金] C++") == ["stale", "proof", "现金", "c"] + assert score_bm25(["cash"], "cash cash") == score_bm25(["cash"], "cash") + assert score_bm25(["equity:ABC"], "CompanyName").documents[0].score == 0 diff --git a/tests/test_repair_pattern_lookup.py b/tests/test_repair_pattern_lookup.py index 5202a26179..becd06f543 100644 --- a/tests/test_repair_pattern_lookup.py +++ b/tests/test_repair_pattern_lookup.py @@ -115,6 +115,13 @@ def test_real_install_delivers_lookup_and_runs_without_loopx_on_path(tmp_path): assert installed_skill_summary((destination,))["loopx-self-repair"]["required_phrases"] installed = destination / "loopx-self-repair" / "scripts" / "find_pattern.py" + bundled = installed.with_name("lexical_retrieval.py") + assert bundled.read_bytes() == (ROOT / "loopx/lexical_retrieval.py").read_bytes() + again = subprocess.run([ + sys.executable, "-m", "loopx.cli", "--format", "json", "workflow-skills", + "--install", "--skills-dir", str(destination), + ], cwd=ROOT, capture_output=True, text=True, check=True) + assert json.loads(again.stdout)["installed"]["loopx-self-repair"] == "unchanged" result = subprocess.run([ sys.executable, "-I", str(installed), "--query", "closeout recovery", ], cwd=tmp_path, env={"PATH": str(tmp_path)}, capture_output=True, text=True, check=True) @@ -130,6 +137,7 @@ def test_real_install_delivers_lookup_and_runs_without_loopx_on_path(tmp_path): with (ROOT / "pyproject.toml").open("rb") as stream: data = tomllib.load(stream)["tool"]["setuptools"]["data-files"] included = {file for files in data.values() for file in files} + assert "loopx/lexical_retrieval.py" in included for path in SKILL.rglob("*"): if path.is_file() and "__pycache__" not in path.parts: assert str(path.relative_to(ROOT)) in included diff --git a/tests/test_workflow_skill_install.py b/tests/test_workflow_skill_install.py index cdeff7bc84..e3ff0e17d4 100644 --- a/tests/test_workflow_skill_install.py +++ b/tests/test_workflow_skill_install.py @@ -378,6 +378,8 @@ def test_frozen_bundle_install_lifecycle( bundled_skills = bundle / layout for skill_id in PACKAGED_HOST_SKILL_IDS: shutil.copytree(canonical / skill_id, bundled_skills / skill_id) + shutil.copy2(Path(install_module.__file__).with_name("lexical_retrieval.py"), + bundled_skills / "loopx-self-repair/scripts/lexical_retrieval.py") monkeypatch.setattr(sys, "frozen", True, raising=False) if with_meipass: monkeypatch.setattr(sys, "_MEIPASS", str(bundle), raising=False) @@ -412,7 +414,7 @@ def unexpected_distribution(name: str) -> None: target / skill_id, ignored_relative_paths=(SKILL_VERSION_MARKER_FILENAME,), ) == install_module.hash_skill_tree( - canonical / skill_id, + bundled_skills / skill_id, ignored_relative_paths=(SKILL_VERSION_MARKER_FILENAME,), ) repeated = workflow_skill_install(skills_dir=target, execute=True) @@ -429,6 +431,18 @@ def unexpected_distribution(name: str) -> None: assert sorted(removed["result"]["removed"]) == sorted(ARK_MANAGED_AGENT_REQUIRED_SKILL_IDS) +def test_frozen_bundle_missing_shared_scorer_does_not_borrow_ambient_code(tmp_path, monkeypatch): + canonical = Path(resolve_workflow_skill_source()["skills_root"]) + bundle = tmp_path / "incomplete bundle" + for skill_id in PACKAGED_HOST_SKILL_IDS: + shutil.copytree(canonical / skill_id, bundle / "skills" / skill_id) + monkeypatch.setattr(sys, "frozen", True, raising=False) + monkeypatch.setattr(sys, "_MEIPASS", str(bundle), raising=False) + result = workflow_skill_install(skills_dir=tmp_path / "host skills", execute=True) + assert result["ok"] is False + assert result["source"]["kind"] == "missing" + + @pytest.mark.parametrize("partial", [False, True]) def test_frozen_missing_data_does_not_fall_back_to_checkout( tmp_path: Path, monkeypatch: pytest.MonkeyPatch, partial: bool, @@ -469,6 +483,10 @@ def test_frozen_bundle_layout_precedence( path = tmp_path / layout / skill_id / "SKILL.md" path.parent.mkdir(parents=True) path.write_text(f"# {layout}: {skill_id}\n", encoding="utf-8") + if not layout.startswith("share/") or complete_wheel_layout: + scorer = tmp_path / layout / "loopx-self-repair/scripts/lexical_retrieval.py" + scorer.parent.mkdir(parents=True) + scorer.write_text("# bundled scorer\n", encoding="utf-8") monkeypatch.setattr(sys, "frozen", True, raising=False) monkeypatch.setattr(sys, "_MEIPASS", str(tmp_path), raising=False) source = resolve_workflow_skill_source()