Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
45 changes: 45 additions & 0 deletions docs/development/lexical-retrieval.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,45 @@
# Shared lexical retrieval / 共享词法检索

`loopx.lexical_retrieval.score_bm25(documents, query)` is a standard-library-only
scorer, reused by the self-repair lookup and the financial-evidence consumer.
It accepts document strings and returns one score plus matched terms per input,
and corpus-wide unmatched query terms. It keeps input positions; the consumer
owns filtering, stable keys, tie-breaks, exact-match precedence and pagination.

`loopx.lexical_retrieval.score_bm25(documents, query)` 仅依赖标准库,自修复查询和
金融证据消费者共用它。输入文档字符串,返回与每个输入对应的分数、匹配词项,以及整个
语料未命中的查询词项。保留输入位置;筛选、稳定身份、同分排序、准确匹配优先级和分页
由消费者负责。

```python
from loopx.lexical_retrieval import score_bm25

result = score_bm25(["lease recovery proof", "display refresh"], "recovery proof")
assert result.documents[0].score > result.documents[1].score
```

The fixed algorithm is BM25 with k1=1.2, b=.75 and positive Lucene IDF.
Casefolded Unicode word tokenization splits underscores. Repeated query terms
do not boost scores. No synonyms, embeddings, segmentation service, discovery,
storage, configuration, network or Goal authority are introduced. Ranking is
corpus-relative; a score does not certify a fact, risk, investment edge or trade.
Historical consumers must supply only their visible corpus: future documents
would otherwise change document frequencies even if hidden from the results.

固定算法为 BM25,k1=1.2、b=.75、Lucene 正 IDF。Unicode 词项经 casefold,
下划线分隔;重复查询词项不增加权重。不引入同义词、嵌入、分词服务、发现、存储、配置、
网络或 Goal 权威。分数相对于语料,不认证事实、风险、投资优势或交易资格。历史查询的
消费者必须只传入可见语料,否则未来文档即使不展示,也会改变文档频率。

The repair skill retains its parser, exact pattern/code precedence and complete
guidance expansion. The existing workflow installer and wheel bundle the
canonical scorer beside the script, so isolated execution still works without
LoopX on PATH. The packaged copy is a build/install artifact, not another source
implementation. Finance retains canonical instruments, literal match rules,
source/clock/lifecycle boundaries and decision authority in its own consumer.
No new automatic capability hook or UI/Lark configuration is enabled.

自修复 skill 保留解析器、准确 pattern/code 优先级和完整正文展开。现有 workflow
installer 和 wheel 将规范评分器打包到脚本旁,隔离运行仍无需 PATH 中的 LoopX。
打包副本是构建/安装产物,不是第二份实现。金融消费者保留规范资产、字面匹配规则、
信源/时点/lifecycle 边界和决策权威。不启用新的自动 capability hook 或 UI/Lark 配置。
52 changes: 52 additions & 0 deletions loopx/lexical_retrieval.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,52 @@
"""Dependency-free lexical scoring; callers own identity, authority and selection.

No storage, network, Goal state or source discovery. Scores are corpus-relative
recall order, never confidence. Kept standalone for the packaged repair skill.
"""

from __future__ import annotations

from collections import Counter
from math import log1p
import re
from typing import Iterable, NamedTuple


class BM25Hit(NamedTuple):
score: float
matched_terms: tuple[str, ...]


class BM25Scores(NamedTuple):
# One entry per input document; callers retain their own keys/tie-breaks.
documents: tuple[BM25Hit, ...]
unmatched_terms: tuple[str, ...]


def lexical_tokens(text: str) -> list[str]:
"""Casefold Unicode words; split underscore identifiers, without aliases."""
return re.findall(r"[^\W_]+", text.casefold())


def score_bm25(documents: Iterable[str], query: str) -> BM25Scores:
"""BM25 k1=1.2, b=.75, positive Lucene IDF; no query-frequency boost.

Every input receives a score, including zero-match documents. Filtering,
pagination, exact-ID precedence and stable ordering belong to the caller.
"""
terms = set(lexical_tokens(query))
counts = [Counter(lexical_tokens(text)) for text in documents]
lengths = [sum(document.values()) for document in counts]
average = sum(lengths) / len(lengths) if lengths else 1
frequencies = Counter(term for document in counts for term in document)
scores = []
for document, length in zip(counts, lengths):
matched = tuple(sorted(terms & document.keys()))
score = sum(
log1p((len(counts) - frequencies[term] + .5) / (frequencies[term] + .5))
* document[term] * 2.2
/ (document[term] + 1.2 * (.25 + .75 * length / (average or 1)))
for term in matched
)
scores.append(BM25Hit(score, matched))
return BM25Scores(tuple(scores), tuple(sorted(terms - frequencies.keys())))
17 changes: 14 additions & 3 deletions loopx/workflow_skill_install.py
Original file line number Diff line number Diff line change
Expand Up @@ -45,11 +45,12 @@
_INSTALL_LOCK_STEM = ".loopx-workflow-skills"


def _valid_source_root(path: Path) -> bool:
def _valid_source_root(path: Path, *, require_bundled_scorer: bool = False) -> bool:
return all(
(path / skill_id / "SKILL.md").is_file()
for skill_id in PACKAGED_HOST_SKILL_IDS
)
) and (not require_bundled_scorer or
(path / "loopx-self-repair/scripts/lexical_retrieval.py").is_file())


def resolve_workflow_skill_source() -> dict[str, Any]:
Expand All @@ -69,7 +70,7 @@ def resolve_workflow_skill_source() -> dict[str, Any]:
bundle_root / "share" / "loopx" / "skills",
bundle_root / "skills",
)
if _valid_source_root(candidate)
if _valid_source_root(candidate, require_bundled_scorer=True)
),
None,
)
Expand Down Expand Up @@ -166,6 +167,16 @@ def _exclusive_install_lock(skills_dir: Path) -> Iterator[None]:


def _install_one_skill(source: Path, target: Path) -> str:
# Wheel/frozen skill data already includes this resource. A source install
# materializes the same canonical file, then uses the existing hash/atomic
# install path so repeated installs and drift readback include the scorer.
if source.name == "loopx-self-repair" and not (source / "scripts/lexical_retrieval.py").is_file():
with tempfile.TemporaryDirectory(prefix="loopx-repair-skill-") as staging:
bundle = Path(staging) / source.name
shutil.copytree(source, bundle)
shutil.copy2(Path(__file__).with_name("lexical_retrieval.py"),
bundle / "scripts/lexical_retrieval.py")
return _install_one_skill(bundle, target)
ignored = (SKILL_VERSION_MARKER_FILENAME,)
if target.is_dir() and hash_skill_tree(
source, ignored_relative_paths=ignored
Expand Down
1 change: 1 addition & 0 deletions pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -131,6 +131,7 @@ include = ["loopx*"]
]
"share/loopx/skills/loopx-self-repair/scripts" = [
"skills/loopx-self-repair/scripts/find_pattern.py",
"loopx/lexical_retrieval.py",
]

[tool.pytest.ini_options]
Expand Down
37 changes: 18 additions & 19 deletions skills/loopx-self-repair/scripts/find_pattern.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,16 +3,25 @@
from __future__ import annotations

import argparse
from collections import Counter
import importlib.util
import json
from math import log1p
from pathlib import Path
import re


CATALOG = Path(__file__).resolve().parents[1] / "references" / "repair-patterns.md"
FIELDS = ("pattern", "symptoms", "evidence", "likely_root", "durable_repair")

# The installer/wheel bundles the canonical stdlib-only module beside this
# script. Source checkout execution reads that same file from the package.
_scorer_path = Path(__file__).with_name("lexical_retrieval.py")
if not _scorer_path.is_file():
_scorer_path = Path(__file__).resolve().parents[3] / "loopx" / "lexical_retrieval.py"
_spec = importlib.util.spec_from_file_location("repair_lexical_retrieval", _scorer_path)
assert _spec is not None and _spec.loader is not None
_scorer = importlib.util.module_from_spec(_spec)
_spec.loader.exec_module(_scorer)


def read_patterns(path: Path) -> list[dict[str, str]]:
"""Keep complete table cells and prose appendices; never silently drop a row."""
Expand Down Expand Up @@ -48,25 +57,15 @@ def read_patterns(path: Path) -> list[dict[str, str]]:
def search_patterns(
patterns: list[dict[str, str]], query: str, *, offset: int, limit: int
) -> dict[str, object]:
# BM25 (k1=1.2, b=0.75), with Lucene's positive IDF. Whole identifiers
# remain addressable by --id; tokenization also exposes their components.
terms = set(re.findall(r"[^\W_]+", query.casefold()))
documents = [Counter(re.findall(r"[^\W_]+", "\n".join(row.values()).casefold())) for row in patterns]
lengths = [sum(doc.values()) for doc in documents]
average = sum(lengths) / len(lengths) if lengths else 1
frequencies = Counter(term for doc in documents for term in doc)
texts = ["\n".join(row.values()) for row in patterns]
ranking = _scorer.score_bm25(texts, query)
scored = []
for row, document, length in zip(patterns, documents, lengths):
matched_terms = sorted(terms & document.keys())
score = sum(
log1p((len(patterns) - frequencies[term] + 0.5) / (frequencies[term] + 0.5))
* document[term] * 2.2
/ (document[term] + 1.2 * (0.25 + 0.75 * length / (average or 1)))
for term in matched_terms
)
for row, text, hit in zip(patterns, texts, ranking.documents):
score = hit.score
matched_terms = list(hit.matched_terms)
exact_id = row["pattern"].casefold() == query.strip().casefold()
exact_code = any(query.strip().casefold() == code.casefold()
for code in re.findall(r"`([^`\n]+)`", "\n".join(row.values())))
for code in re.findall(r"`([^`\n]+)`", text))
if score > 0 or exact_id or not query:
scored.append((exact_id, exact_code, score, row, matched_terms))
# Stable id tie-break keeps pagination reproducible; scores are not confidence.
Expand All @@ -82,7 +81,7 @@ def search_patterns(
"ok": True,
"query": query,
"match_mode": "bm25_exact_first" if query else "catalog_order",
"unmatched_terms": sorted(terms - frequencies.keys()),
"unmatched_terms": list(ranking.unmatched_terms),
"total_matches": len(matched),
"offset": offset,
"next_offset": end if end < len(matched) else None,
Expand Down
28 changes: 28 additions & 0 deletions tests/test_lexical_retrieval.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,28 @@
"""Neutral scoring and caller-owned relevance boundaries."""

from loopx.lexical_retrieval import lexical_tokens, score_bm25


def test_rare_multiple_terms_beat_repeated_generic_word():
result = score_bm25(["lease recovery proof", "lease " * 20, "unrelated"],
"lease recovery absent")
assert result.documents[0].score > result.documents[1].score > 0
assert result.documents[2].score == 0
assert result.documents[0].matched_terms == ("lease", "recovery")
assert result.unmatched_terms == ("absent",)


def test_zero_empty_and_generator_corpora_keep_input_identity():
assert score_bm25([], "missing").documents == ()
assert score_bm25([], "missing").unmatched_terms == ("missing",)
result = score_bm25((text for text in ["", "cash", "cash"]), "cash")
assert len(result.documents) == 3
assert result.documents[0].score == 0
assert result.documents[1] == result.documents[2]
assert all(hit.score == 0 for hit in score_bm25(["cash"], "").documents)


def test_tokens_are_literal_and_query_repetition_does_not_boost_scores():
assert lexical_tokens("STALE_Proof [现金] C++") == ["stale", "proof", "现金", "c"]
assert score_bm25(["cash"], "cash cash") == score_bm25(["cash"], "cash")
assert score_bm25(["equity:ABC"], "CompanyName").documents[0].score == 0
8 changes: 8 additions & 0 deletions tests/test_repair_pattern_lookup.py
Original file line number Diff line number Diff line change
Expand Up @@ -115,6 +115,13 @@ def test_real_install_delivers_lookup_and_runs_without_loopx_on_path(tmp_path):

assert installed_skill_summary((destination,))["loopx-self-repair"]["required_phrases"]
installed = destination / "loopx-self-repair" / "scripts" / "find_pattern.py"
bundled = installed.with_name("lexical_retrieval.py")
assert bundled.read_bytes() == (ROOT / "loopx/lexical_retrieval.py").read_bytes()
again = subprocess.run([
sys.executable, "-m", "loopx.cli", "--format", "json", "workflow-skills",
"--install", "--skills-dir", str(destination),
], cwd=ROOT, capture_output=True, text=True, check=True)
assert json.loads(again.stdout)["installed"]["loopx-self-repair"] == "unchanged"
result = subprocess.run([
sys.executable, "-I", str(installed), "--query", "closeout recovery",
], cwd=tmp_path, env={"PATH": str(tmp_path)}, capture_output=True, text=True, check=True)
Expand All @@ -130,6 +137,7 @@ def test_real_install_delivers_lookup_and_runs_without_loopx_on_path(tmp_path):
with (ROOT / "pyproject.toml").open("rb") as stream:
data = tomllib.load(stream)["tool"]["setuptools"]["data-files"]
included = {file for files in data.values() for file in files}
assert "loopx/lexical_retrieval.py" in included
for path in SKILL.rglob("*"):
if path.is_file() and "__pycache__" not in path.parts:
assert str(path.relative_to(ROOT)) in included
20 changes: 19 additions & 1 deletion tests/test_workflow_skill_install.py
Original file line number Diff line number Diff line change
Expand Up @@ -378,6 +378,8 @@ def test_frozen_bundle_install_lifecycle(
bundled_skills = bundle / layout
for skill_id in PACKAGED_HOST_SKILL_IDS:
shutil.copytree(canonical / skill_id, bundled_skills / skill_id)
shutil.copy2(Path(install_module.__file__).with_name("lexical_retrieval.py"),
bundled_skills / "loopx-self-repair/scripts/lexical_retrieval.py")
monkeypatch.setattr(sys, "frozen", True, raising=False)
if with_meipass:
monkeypatch.setattr(sys, "_MEIPASS", str(bundle), raising=False)
Expand Down Expand Up @@ -412,7 +414,7 @@ def unexpected_distribution(name: str) -> None:
target / skill_id,
ignored_relative_paths=(SKILL_VERSION_MARKER_FILENAME,),
) == install_module.hash_skill_tree(
canonical / skill_id,
bundled_skills / skill_id,
ignored_relative_paths=(SKILL_VERSION_MARKER_FILENAME,),
)
repeated = workflow_skill_install(skills_dir=target, execute=True)
Expand All @@ -429,6 +431,18 @@ def unexpected_distribution(name: str) -> None:
assert sorted(removed["result"]["removed"]) == sorted(ARK_MANAGED_AGENT_REQUIRED_SKILL_IDS)


def test_frozen_bundle_missing_shared_scorer_does_not_borrow_ambient_code(tmp_path, monkeypatch):
canonical = Path(resolve_workflow_skill_source()["skills_root"])
bundle = tmp_path / "incomplete bundle"
for skill_id in PACKAGED_HOST_SKILL_IDS:
shutil.copytree(canonical / skill_id, bundle / "skills" / skill_id)
monkeypatch.setattr(sys, "frozen", True, raising=False)
monkeypatch.setattr(sys, "_MEIPASS", str(bundle), raising=False)
result = workflow_skill_install(skills_dir=tmp_path / "host skills", execute=True)
assert result["ok"] is False
assert result["source"]["kind"] == "missing"


@pytest.mark.parametrize("partial", [False, True])
def test_frozen_missing_data_does_not_fall_back_to_checkout(
tmp_path: Path, monkeypatch: pytest.MonkeyPatch, partial: bool,
Expand Down Expand Up @@ -469,6 +483,10 @@ def test_frozen_bundle_layout_precedence(
path = tmp_path / layout / skill_id / "SKILL.md"
path.parent.mkdir(parents=True)
path.write_text(f"# {layout}: {skill_id}\n", encoding="utf-8")
if not layout.startswith("share/") or complete_wheel_layout:
scorer = tmp_path / layout / "loopx-self-repair/scripts/lexical_retrieval.py"
scorer.parent.mkdir(parents=True)
scorer.write_text("# bundled scorer\n", encoding="utf-8")
monkeypatch.setattr(sys, "frozen", True, raising=False)
monkeypatch.setattr(sys, "_MEIPASS", str(tmp_path), raising=False)
source = resolve_workflow_skill_source()
Expand Down
Loading