Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
30 changes: 27 additions & 3 deletions db/lab_inventory.py
Original file line number Diff line number Diff line change
Expand Up @@ -31,6 +31,7 @@
CODE_SAMPLE_TITLE_RE = re.compile(r"\bcode\s+sample\b", re.IGNORECASE)
MIN_PROJECT_TEXT_CHARS = 90
PROJECT_FINDER_CHUNK_SOURCE_LIMIT = 5000
PROJECT_FINDER_CHUNK_PREFILTER_LIMIT = 20000
PROJECT_FINDER_INTELLIGENCE_SOURCE_LIMIT = 400

GENERIC_PART_PATTERNS: tuple[tuple[re.Pattern[str], str, str], ...] = (
Expand Down Expand Up @@ -280,15 +281,17 @@ def find(
with self.database.connection() as conn:
chunk_rows = conn.execute(
load_query("project_finder_chunk_candidates.sql"),
(self._term_regex(terms), PROJECT_FINDER_CHUNK_SOURCE_LIMIT),
(self._term_regex(terms), PROJECT_FINDER_CHUNK_PREFILTER_LIMIT),
).fetchall()
intelligence_rows = conn.execute(
load_query("project_finder_intelligence_candidates.sql"),
(terms, PROJECT_FINDER_INTELLIGENCE_SOURCE_LIMIT),
).fetchall()

inventory_index = self._inventory_index(inventory, term_rows)
chunk_rows = self._annotate_chunk_matches(chunk_rows, self._prepared_search_terms(terms))
chunk_rows = self._rank_chunk_rows(
self._annotate_chunk_matches(chunk_rows, self._prepared_search_terms(terms))
)[:PROJECT_FINDER_CHUNK_SOURCE_LIMIT]
candidates = [
self._chunk_candidate(row, inventory_index)
for row in chunk_rows
Expand Down Expand Up @@ -406,12 +409,33 @@ def _annotate_chunk_matches(self, rows: list[dict[str, Any]], terms: list[tuple[
annotated.append(payload)
return annotated

@staticmethod
def _rank_chunk_rows(rows: list[dict[str, Any]]) -> list[dict[str, Any]]:
return sorted(
rows,
key=lambda row: (
int(row.get("matched_count") or 0),
float(row.get("quality_score") or 0.0),
str(row.get("source_path") or ""),
-(int(row.get("page_number") or 0)),
-(int(row.get("chunk_index") or 0)),
),
reverse=True,
)

def _matched_terms_for_text(self, text: str, terms: list[tuple[str, str]]) -> list[str]:
normalized_text = normalize_part_name(text)
text_tokens = set(normalized_text.split())
compact_text = compact_part_key(text)
matched = []
for normalized_term, compact_term in terms:
if normalized_term in normalized_text or (compact_term and compact_term in compact_text):
term_tokens = normalized_term.split()
if len(term_tokens) == 1:
text_match = term_tokens[0] in text_tokens
else:
text_match = normalized_term in normalized_text
compact_match = bool(compact_term and len(compact_term) >= 4 and any(char.isdigit() for char in compact_term) and compact_term in compact_text)
if text_match or compact_match:
matched.append(normalized_term)
return matched

Expand Down
4 changes: 4 additions & 0 deletions frontend/src/components/ProjectCandidateFilters.tsx
Original file line number Diff line number Diff line change
@@ -1,4 +1,5 @@
import type { ProjectCandidateFilter } from "../types";
import { LoadingSpinner } from "./LoadingSpinner";

const filters: Array<{ id: ProjectCandidateFilter; label: string }> = [
{ id: "all", label: "All" },
Expand All @@ -9,10 +10,12 @@ const filters: Array<{ id: ProjectCandidateFilter; label: string }> = [
export function ProjectCandidateFilters({
active,
counts,
loading,
onChange
}: {
active: ProjectCandidateFilter;
counts: Record<ProjectCandidateFilter, number>;
loading?: boolean;
onChange: (filter: ProjectCandidateFilter) => void;
}) {
return (
Expand All @@ -25,6 +28,7 @@ export function ProjectCandidateFilters({
onClick={() => onChange(filter.id)}
>
{filter.label}
{loading && active === filter.id ? <LoadingSpinner className="filter-button-spinner" /> : null}
<span>{counts[filter.id]}</span>
</button>
))}
Expand Down
4 changes: 3 additions & 1 deletion frontend/src/components/ProjectFinderView.tsx
Original file line number Diff line number Diff line change
Expand Up @@ -121,7 +121,9 @@ export function ProjectFinderView({
title="Build candidates"
description={`${formatNumber(candidates.length)} visible of ${formatNumber(activeFilterTotal)} ${filterDescription(candidateFilter)} matches`}
/>
{hasFinderRun ? <ProjectCandidateFilters active={candidateFilter} counts={candidateCounts} onChange={changeCandidateFilter} /> : null}
{hasFinderRun ? (
<ProjectCandidateFilters active={candidateFilter} counts={candidateCounts} loading={finding} onChange={changeCandidateFilter} />
) : null}
</div>
<ProjectCandidateList
candidates={candidates}
Expand Down
6 changes: 6 additions & 0 deletions frontend/src/styles.css
Original file line number Diff line number Diff line change
Expand Up @@ -2099,6 +2099,12 @@ nav {
font-size: 0.82rem;
}

.filter-button-spinner {
width: 13px;
height: 13px;
border-width: 2px;
}

.project-candidate-pagination {
display: flex;
justify-content: center;
Expand Down
22 changes: 22 additions & 0 deletions tests/test_lab_inventory.py
Original file line number Diff line number Diff line change
Expand Up @@ -182,6 +182,28 @@ def test_chunk_match_annotation_uses_prepared_terms(self):
self.assertEqual(annotated[0]["matched_terms"], ["ne555 timer", "10 segment led"])
self.assertEqual(annotated[0]["matched_count"], 2)

def test_chunk_match_annotation_does_not_match_short_terms_inside_words(self):
store = ProjectFinderStore(None, None)
terms = store._prepared_search_terms(["IC", "LED"])

annotated = store._annotate_chunk_matches(
[{"chunk_text": "Electronic parts are discussed here, but no integrated circuit token appears."}],
terms,
)

self.assertEqual(annotated, [])

def test_chunk_rows_rank_by_match_count_before_quality(self):
rows = [
{"source_path": "high-quality.pdf", "matched_count": 1, "quality_score": 0.99, "page_number": 1, "chunk_index": 1},
{"source_path": "better-match.pdf", "matched_count": 3, "quality_score": 0.2, "page_number": 1, "chunk_index": 2},
{"source_path": "middle.pdf", "matched_count": 2, "quality_score": 0.5, "page_number": 1, "chunk_index": 3},
]

ranked = ProjectFinderStore._rank_chunk_rows(rows)

self.assertEqual([row["source_path"] for row in ranked], ["better-match.pdf", "middle.pdf", "high-quality.pdf"])

def test_missing_part_summary_ranks_repeated_gaps(self):
store = ProjectFinderStore(None, None)
summary = store._missing_part_summary(
Expand Down