Perf: parallelize the full-N palette scan across the thread pool
PaletteModel::rebuild splits the bulk score-only scan into per-thread chunks via QtConcurrent::blockingMapped when item count exceeds kParallelThreshold (10000). The scan is a pure map over read-only data (FuzzyRanker::score is a stateless static; m_items not mutated); chunks are concatenated in order and finalizeVisible imposes a deterministic sort, so the result is bit-identical to the serial path regardless of thread timing. Incremental (type-forward) path stays serial — it only touches <=1000 items. First-scan at 397k symbols (mobydick, 16 cores): ~373ms -> ~61ms. Link Qt6::Concurrent into the palette lib. New test parallelScanIsCorrectAndDeterministic (40k items) guards correctness.
This commit is contained in:
parent
1caa8d2221
commit
00f81f7f56
10
docs/PERF.md
10
docs/PERF.md
|
|
@ -104,9 +104,13 @@ Measured on mobydick (397k symbols, RelWithDebInfo, type-forward):
|
|||
| `serv` | 704 ms | **1.4 ms** |
|
||||
| `server` | 198 ms | **0.8 ms** |
|
||||
|
||||
The one remaining O(N) cost is the single first scan at the 2nd char (~370 ms at
|
||||
397k), absorbed by the 80 ms debounce during a typing burst. Parallelising that
|
||||
bulk scan (QtConcurrent across cores) is an optional future ~8× on that one hit.
|
||||
The one remaining O(N) cost is the single first scan at the 2nd char. It is now
|
||||
**parallelised** across the thread pool (`QtConcurrent::blockingMapped` over
|
||||
per-thread chunks in `PaletteModel::rebuild`, above `kParallelThreshold = 10000`
|
||||
items). The scan is a pure map over read-only data, concatenated in chunk order
|
||||
and sorted deterministically, so the result is identical to the serial path
|
||||
regardless of thread timing. Measured first-scan at 397k symbols: **~373 ms →
|
||||
~61 ms** (16 cores). Everything after is incremental (<2 ms).
|
||||
|
||||
## Indexing off the UI thread (done)
|
||||
|
||||
|
|
|
|||
|
|
@ -6,7 +6,7 @@ add_library(palette STATIC
|
|||
frecencystore.cpp
|
||||
frecencystore.h
|
||||
)
|
||||
target_link_libraries(palette PUBLIC fuzzyranker Qt6::Core Qt6::Gui Qt6::Widgets)
|
||||
target_link_libraries(palette PUBLIC fuzzyranker Qt6::Core Qt6::Gui Qt6::Widgets Qt6::Concurrent)
|
||||
target_include_directories(palette PUBLIC ${CMAKE_CURRENT_SOURCE_DIR})
|
||||
|
||||
if(Qt6Test_FOUND)
|
||||
|
|
|
|||
|
|
@ -4,8 +4,11 @@
|
|||
#include "palettemodel.h"
|
||||
|
||||
#include <QPoint>
|
||||
#include <QThread>
|
||||
#include <QtConcurrent/QtConcurrentMap>
|
||||
|
||||
#include <algorithm>
|
||||
#include <vector>
|
||||
|
||||
namespace katecustom
|
||||
{
|
||||
|
|
@ -23,6 +26,10 @@ constexpr int kMaxVisible = 1000;
|
|||
// useless. Below this, the palette shows the (capped) full list unranked; the
|
||||
// first real scan only runs once the query is selective enough to matter.
|
||||
constexpr int kMinQueryChars = 2;
|
||||
|
||||
// Below this many items, the full scan runs serially — thread dispatch overhead
|
||||
// is not worth it. Above it, the scan is split across the thread pool.
|
||||
constexpr int kParallelThreshold = 10000;
|
||||
} // namespace
|
||||
|
||||
PaletteModel::PaletteModel(QObject *parent)
|
||||
|
|
@ -71,15 +78,51 @@ void PaletteModel::setQuery(const QString &query)
|
|||
void PaletteModel::rebuild()
|
||||
{
|
||||
const QList<QString> needles = FuzzyRanker::tokenize(m_query);
|
||||
const int n = m_items.size();
|
||||
m_scored.clear();
|
||||
for (int i = 0; i < m_items.size(); ++i) {
|
||||
const PaletteItem &item = m_items.at(i);
|
||||
// Score-only (no ranges) for the bulk scan — the hot path at large N.
|
||||
const MatchResult r = FuzzyRanker::score(item.label, needles, /*withRanges=*/false);
|
||||
if (r.matched) {
|
||||
m_scored.push_back({i, r.score + item.frecency});
|
||||
|
||||
auto scoreRange = [this, &needles](int lo, int hi) {
|
||||
std::vector<Scored> local;
|
||||
for (int i = lo; i < hi; ++i) {
|
||||
// Score-only (no ranges) for the bulk scan — the hot path at large N.
|
||||
const MatchResult r =
|
||||
FuzzyRanker::score(m_items.at(i).label, needles, /*withRanges=*/false);
|
||||
if (r.matched) {
|
||||
local.push_back({i, r.score + m_items.at(i).frecency});
|
||||
}
|
||||
}
|
||||
return local;
|
||||
};
|
||||
|
||||
if (n < kParallelThreshold) {
|
||||
m_scored = scoreRange(0, n);
|
||||
} else {
|
||||
// Split [0, n) into one chunk per hardware thread and score the chunks
|
||||
// in parallel. The scan is a pure map over read-only data
|
||||
// (FuzzyRanker::score is a stateless static; m_items is not mutated), so
|
||||
// chunks are independent. Each returns its own vector; we concatenate in
|
||||
// chunk order and let finalizeVisible() impose the deterministic sort —
|
||||
// so the result does not depend on thread timing.
|
||||
const int threads = qMax(1, QThread::idealThreadCount());
|
||||
const int chunk = (n + threads - 1) / threads;
|
||||
struct Range { int lo; int hi; };
|
||||
QList<Range> ranges;
|
||||
for (int lo = 0; lo < n; lo += chunk) {
|
||||
ranges.push_back({lo, qMin(lo + chunk, n)});
|
||||
}
|
||||
const QList<std::vector<Scored>> parts = QtConcurrent::blockingMapped(
|
||||
ranges, [&scoreRange](const Range &r) { return scoreRange(r.lo, r.hi); });
|
||||
|
||||
int total = 0;
|
||||
for (const auto &p : parts) {
|
||||
total += static_cast<int>(p.size());
|
||||
}
|
||||
m_scored.reserve(total);
|
||||
for (const auto &p : parts) {
|
||||
m_scored.insert(m_scored.end(), p.begin(), p.end());
|
||||
}
|
||||
}
|
||||
|
||||
finalizeVisible(needles);
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -237,6 +237,44 @@ private Q_SLOTS:
|
|||
QCOMPARE(m.index(0, 0).data(PaletteModel::LabelRole).toString(),
|
||||
QStringLiteral("alpha"));
|
||||
}
|
||||
|
||||
// Above the parallel threshold the bulk scan is split across the thread
|
||||
// pool. The result must be identical and deterministic regardless of
|
||||
// threading: a known needle in a sea of non-matches yields exactly the
|
||||
// planted matches, in deterministic (source) order.
|
||||
void parallelScanIsCorrectAndDeterministic()
|
||||
{
|
||||
QList<PaletteItem> items;
|
||||
const int n = 40000; // well above kParallelThreshold (10000)
|
||||
for (int i = 0; i < n; ++i) {
|
||||
// Every 1000th item contains the needle "zebra"; the rest do not.
|
||||
const bool planted = (i % 1000 == 0);
|
||||
items.push_back({QString::number(i),
|
||||
planted ? QStringLiteral("zebra_%1").arg(i)
|
||||
: QStringLiteral("filler_%1").arg(i),
|
||||
QString(), 0});
|
||||
}
|
||||
PaletteModel m;
|
||||
m.setItems(items);
|
||||
m.setQuery(QStringLiteral("zebra"));
|
||||
|
||||
const int expected = n / 1000; // 40 planted matches
|
||||
QCOMPARE(m.rowCount(), expected);
|
||||
|
||||
// Deterministic: the kept ids match the planted set, and re-running the
|
||||
// same query twice yields the same order.
|
||||
QStringList first;
|
||||
for (int r = 0; r < m.rowCount(); ++r) {
|
||||
first << m.index(r, 0).data(PaletteModel::IdRole).toString();
|
||||
}
|
||||
m.setQuery(QString());
|
||||
m.setQuery(QStringLiteral("zebra"));
|
||||
QStringList second;
|
||||
for (int r = 0; r < m.rowCount(); ++r) {
|
||||
second << m.index(r, 0).data(PaletteModel::IdRole).toString();
|
||||
}
|
||||
QCOMPARE(first, second);
|
||||
}
|
||||
};
|
||||
|
||||
QTEST_MAIN(TestPaletteModel)
|
||||
|
|
|
|||
Loading…
Reference in New Issue