Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 2 additions & 1 deletion apps/reviewer-web/src/view-model.ts
Original file line number Diff line number Diff line change
Expand Up @@ -99,7 +99,8 @@ const integrityFlagDescriptions: Record<string, string> = {
suspicious_bulk_paste: "Detected unusually large pasted content.",
excessive_focus_switching: "Detected excessive app focus switching.",
excessive_idle_time: "Detected excessive idle time during session.",
unmanaged_browser_detected: "Detected unmanaged browser usage."
unmanaged_browser_detected: "Detected unmanaged browser usage.",
low_information_session: "Too few events for reliable scoring; archetype label should be treated as indicative only."
};

function humanizeFlag(flag: string): string {
Expand Down
22 changes: 22 additions & 0 deletions services/analytics-py/assessment_analytics/integrity.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,12 @@
from collections import defaultdict
from typing import Any

# Sessions with fewer than this many total events can be low-information for
# archetype scoring, but we only flag the specific sparse signature that is
# known to inflate Independent Solver confidence: short session + typing-only
# edits + no AI prompt telemetry.
_LOW_INFORMATION_EVENT_THRESHOLD = 10


def evaluate_integrity(
events: list[dict[str, Any]],
Expand Down Expand Up @@ -67,6 +73,22 @@ def evaluate_integrity(
if any(event["event_type"] == "system.browser.unmanaged" for event in events):
flags.append("unmanaged_browser_detected")

total_insert_events = feature_vector["signal_values"].get("total_insert_events", 0)
total_paste_events = feature_vector["signal_values"].get("total_paste_events", 0)
total_prompts_sent = feature_vector["signal_values"].get("total_prompts_sent", 0)
likely_sparse_independent_solver_signature = (
len(events) < _LOW_INFORMATION_EVENT_THRESHOLD
and total_insert_events > 0
and total_paste_events == 0
and total_prompts_sent == 0
)
if likely_sparse_independent_solver_signature:
flags.append("low_information_session")
notes.append(
f"Session has only {len(events)} event(s); archetype scoring may be unreliable "
"due to insufficient behavioral signal."
)

verdict = "clean"
if any(flag in flags for flag in ("missing_required_streams", "tamper_signal_detected", "unmanaged_browser_detected")):
verdict = "invalid"
Expand Down
134 changes: 133 additions & 1 deletion services/analytics-py/tests/test_pipeline.py
Original file line number Diff line number Diff line change
Expand Up @@ -142,6 +142,7 @@ def test_integrity_allows_managed_bootstrap_navigation(self) -> None:
integrity = evaluate_integrity(events, feature_vector, session_context)

self.assertNotIn("unsupported_site_visited", integrity["flags"])
self.assertNotIn("low_information_session", integrity["flags"])
self.assertEqual(integrity["verdict"], "clean")

def test_integrity_flags_explicitly_unsupported_browser_navigation(self) -> None:
Expand Down Expand Up @@ -360,5 +361,136 @@ def test_haci_is_stable_across_scoring_modes(self) -> None:
self.assertEqual(heuristic_result["haci_score"], trained_result["haci_score"])


def _make_short_session_events(event_count: int, typed_chars: int = 0) -> list[dict]:
"""Build a minimal event list for short-session tests.

The first two events are always session.started (desktop) and
session.heartbeat (desktop). Additional events up to *event_count* are
ide.document.changed events that insert *typed_chars* characters each.
All events belong to the same session and are spaced 5 seconds apart.
"""
base_ts = "2026-04-14T06:00:{:02d}Z"
events = [
{
"event_id": "desktop-1",
"session_id": "short-session-1",
"timestamp_utc": base_ts.format(0),
"source": "desktop",
"event_type": "session.started",
"sequence_no": 1,
"artifact_ref": "session",
"payload": {"status": "active"},
"client_version": "0.1.0",
"integrity_hash": "h1",
"policy_context": {},
},
{
"event_id": "desktop-2",
"session_id": "short-session-1",
"timestamp_utc": base_ts.format(5),
"source": "desktop",
"event_type": "session.heartbeat",
"sequence_no": 2,
"artifact_ref": "session",
"payload": {"status": "active"},
"client_version": "0.1.0",
"integrity_hash": "h2",
"policy_context": {},
},
]
for index in range(event_count - 2):
events.append({
"event_id": f"ide-{index + 1}",
"session_id": "short-session-1",
"timestamp_utc": base_ts.format(10 + index * 5),
"source": "ide",
Comment on lines +372 to +406
"event_type": "ide.document.changed",
"sequence_no": index + 1,
"artifact_ref": "file:main.py",
"payload": {
"inserted_chars": typed_chars,
"deleted_chars": 0,
"change_source": "typing",
},
"client_version": "0.1.0",
"integrity_hash": f"h-ide-{index + 1}",
"policy_context": {},
})
return events


class ShortSessionTests(unittest.TestCase):
"""Regression tests for sparse-telemetry / low-information sessions."""

_SESSION_CONTEXT = {
"required_streams": ["desktop", "ide"],
}

# ------------------------------------------------------------------
# Integrity flag tests
# ------------------------------------------------------------------

def test_short_session_without_sparse_signature_is_not_flagged_low_information(self) -> None:
"""Short sessions are not flagged unless they match the known sparse signature."""
events = _make_short_session_events(event_count=4)
feature_vector = extract_feature_vector(events, self._SESSION_CONTEXT)
integrity = evaluate_integrity(events, feature_vector, self._SESSION_CONTEXT)

self.assertNotIn("low_information_session", integrity["flags"])
self.assertEqual(integrity["verdict"], "clean")

def test_sparse_signature_without_ai_or_paste_gets_low_information_flag(self) -> None:
"""Low-information flag is grounded to the known short typing-only sparse pattern."""
events = _make_short_session_events(event_count=4, typed_chars=20)
feature_vector = extract_feature_vector(events, self._SESSION_CONTEXT)
integrity = evaluate_integrity(events, feature_vector, self._SESSION_CONTEXT)

self.assertIn("low_information_session", integrity["flags"])
self.assertEqual(integrity["verdict"], "review")
self.assertTrue(any("insufficient behavioral signal" in note for note in integrity["notes"]))

# ------------------------------------------------------------------
# Scoring bias documentation tests
# ------------------------------------------------------------------

def test_sparse_session_with_typing_produces_high_independent_solver_confidence(self) -> None:
"""Documents the known heuristic bias: a short session with a few typed
chars and no paste/AI activity yields a high-confidence Independent
Solver label due to typing_vs_paste_ratio being computed as raw typed
character count when no paste events are present.

This test CONFIRMS the bias and the low_information_session flag is the
mitigation. The label should be treated as indicative only.
"""
events = _make_short_session_events(event_count=4, typed_chars=20)
result = score_session(events, self._SESSION_CONTEXT)

# Bias confirmed: Independent Solver wins at high confidence.
self.assertEqual(result["heuristic_result"]["predicted_archetype"], "Independent Solver")
self.assertGreater(result["heuristic_result"]["confidence"], 0.50)

# Mitigation confirmed: integrity flags the low-information condition.
self.assertIn("low_information_session", result["integrity"]["flags"])

# Policy confirmed: sparse session requires human review (cannot auto-advance).
self.assertEqual(result["policy_recommendation"], "human-review")
self.assertTrue(result["review_required"])

def test_sparse_session_always_requires_human_review(self) -> None:
"""Even if a sparse session somehow satisfies the confidence threshold,
the low_information_session integrity flag keeps verdict at 'review',
blocking auto-advance policy."""
events = _make_short_session_events(event_count=4, typed_chars=20)
# Use a very permissive policy override so that confidence alone would
# allow auto-advance.
context = {**self._SESSION_CONTEXT, "decision_policy": {"auto_advance_min_confidence": 0.10}}
result = score_session(events, context)

# Despite the permissive threshold, the integrity review verdict and the
# absence of a clean integrity verdict prevent auto-advance.
self.assertNotEqual(result["policy_recommendation"], "auto-advance")
self.assertTrue(result["review_required"])


if __name__ == "__main__":
unittest.main()
unittest.main()
52 changes: 52 additions & 0 deletions tests/web/view-models.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -4,6 +4,7 @@ import {
buildArchetypeProbabilityEntries,
buildArchetypeProbabilityEntriesFromMap,
buildCompletenessSummary,
buildIntegrityFlagLabels,
buildSourceMix,
buildTimelineEntries,
confidenceLabel,
Expand Down Expand Up @@ -365,3 +366,54 @@ test("scoringModesDisagree detects archetype disagreement between dual scoring m
trained_model_result: modelBlindCopier
}), true);
});

test("buildIntegrityFlagLabels renders low_information_session with reviewer-facing description", () => {
const baseScoring = {
session_id: "session-sparse",
model_version: "bootstrap-centroid-v1",
scoring_mode: "heuristic" as const,
haci_score: 20,
haci_band: "low" as const,
predicted_archetype: "Independent Solver" as const,
archetype_probabilities: { "Independent Solver": 0.71 },
confidence: 0.71,
top_features: [],
integrity: {
verdict: "review" as const,
flags: ["low_information_session"],
required_streams_present: ["desktop", "ide"],
missing_streams: [],
notes: ["Session has only 4 event(s); archetype scoring may be unreliable due to insufficient behavioral signal."]
},
policy_recommendation: "human-review" as const,
review_required: true,
feature_vector: {
session_id: "session-sparse",
extraction_version: "0.1.0",
generated_at: "2026-04-12T09:10:00Z",
signal_values: {},
signals: [],
completeness: "partial" as const,
invalidation_reasons: []
}
};

// null input returns empty array
assert.deepEqual(buildIntegrityFlagLabels(null), []);

const labels = buildIntegrityFlagLabels(baseScoring);
assert.equal(labels.length, 1);
// The label must contain the human-readable description and the flag name
assert.match(labels[0], /Too few events for reliable scoring/);
assert.match(labels[0], /low_information_session/);

// Known flags with no descriptions fall back to humanized form
const withUnknownFlag = {
...baseScoring,
integrity: { ...baseScoring.integrity, flags: ["some_future_flag"] }
};
const unknownLabels = buildIntegrityFlagLabels(withUnknownFlag);
assert.equal(unknownLabels.length, 1);
assert.match(unknownLabels[0], /Some Future Flag/);
assert.match(unknownLabels[0], /some_future_flag/);
});