Kalibrierte Scan-Confidence + Cross-Model-Bonus
This commit is contained in:
@@ -82,6 +82,35 @@ describe('decideReviewOutcome', () => {
|
||||
expect(decision.reason).toBe('review-confirmed-primary');
|
||||
});
|
||||
|
||||
it('boosts confidence when two different models agree on the species', () => {
|
||||
// The 55%-forever bug: primary 0.55 + agreeing review 0.54 must NOT stay
|
||||
// at 0.55 — independent cross-model agreement is real evidence.
|
||||
const decision = decideReviewOutcome({
|
||||
primaryResult: ai('Sacred lotus', 'Nelumbo nucifera', 0.55),
|
||||
reviewResult: ai('Sacred lotus', 'Nelumbo nucifera', 0.54),
|
||||
agrees: true,
|
||||
});
|
||||
expect(decision.confidence).toBeCloseTo(0.75, 5);
|
||||
});
|
||||
|
||||
it('caps the agreement-boosted confidence at 0.97', () => {
|
||||
const decision = decideReviewOutcome({
|
||||
primaryResult: ai('Sacred lotus', 'Nelumbo nucifera', 0.95),
|
||||
reviewResult: ai('Sacred lotus', 'Nelumbo nucifera', 0.9),
|
||||
agrees: true,
|
||||
});
|
||||
expect(decision.confidence).toBe(0.97);
|
||||
});
|
||||
|
||||
it('gives no confidence bonus on disagreement', () => {
|
||||
const decision = decideReviewOutcome({
|
||||
primaryResult: ai('Rose', 'Rosa chinensis', 0.5),
|
||||
reviewResult: ai('Sacred lotus', 'Nelumbo nucifera', 0.6),
|
||||
agrees: false,
|
||||
});
|
||||
expect(decision.confidence).toBeUndefined();
|
||||
});
|
||||
|
||||
it('replaces with the agreeing review on a confidence tie (stronger model wins ties)', () => {
|
||||
const decision = decideReviewOutcome({
|
||||
primaryResult: ai('Swiss Cheese Plant', 'Monstera deliciosa', 0.7),
|
||||
|
||||
2
app.json
2
app.json
@@ -2,7 +2,7 @@
|
||||
"expo": {
|
||||
"name": "GreenLens",
|
||||
"slug": "greenlens",
|
||||
"version": "2.2.8",
|
||||
"version": "2.2.9",
|
||||
"orientation": "portrait",
|
||||
"icon": "./assets/icon.png",
|
||||
"userInterfaceStyle": "automatic",
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "greenlens",
|
||||
"version": "2.2.7",
|
||||
"version": "2.2.9",
|
||||
"main": "expo-router/entry",
|
||||
"private": true,
|
||||
"scripts": {
|
||||
|
||||
@@ -822,6 +822,10 @@ app.post('/v1/scan', async (request, response) => {
|
||||
modelUsed = openAiReview.modelUsed || modelUsed;
|
||||
if (grounded.grounded) modelPath.push('catalog-grounded-review');
|
||||
}
|
||||
if (decision.confidence != null) {
|
||||
// Cross-model agreement bonus: both models named the same species.
|
||||
result = { ...result, confidence: decision.confidence };
|
||||
}
|
||||
modelPath.push('openai-review');
|
||||
modelPath.push(decision.reason);
|
||||
} else {
|
||||
|
||||
@@ -243,8 +243,11 @@ const buildIdentifyPrompt = (language, mode) => {
|
||||
: '- "careInfo.temp": temperature range in Celsius (e.g. "18–24 °C"). Must always be a real plant-specific value, never "Unknown".',
|
||||
'- "botanicalName" must use accepted Latin scientific naming and must not be invented or misspelled.',
|
||||
'- If species is uncertain, prefer genus-level naming (for example: "Calathea sp.").',
|
||||
'- "confidence" must be between 0 and 1.',
|
||||
'- Keep confidence <= 0.55 when the image is ambiguous, blurred, or partially visible.',
|
||||
'- "confidence" must be between 0 and 1 and reflect how certain the species identification is. Calibrate it:',
|
||||
' - 0.85-0.95: species clearly recognizable, distinctive features (leaf shape, flower, pattern) plainly visible.',
|
||||
' - 0.65-0.84: species very likely, but some distinguishing features are hidden or similar species exist.',
|
||||
' - 0.40-0.64: image is ambiguous, blurred, partially visible, or several species fit equally well.',
|
||||
' - Below 0.40: mostly guessing; prefer genus-level naming instead.',
|
||||
'- "waterIntervalDays" must be an integer between 1 and 45.',
|
||||
'- Do not include markdown, explanations, or extra keys.',
|
||||
].join('\n');
|
||||
|
||||
@@ -1,5 +1,13 @@
|
||||
const { normalizeText } = require('./scanGrounding');
|
||||
|
||||
const clamp = (value, min, max) => Math.min(max, Math.max(min, value));
|
||||
|
||||
// Two DIFFERENT models independently naming the same species is genuine
|
||||
// evidence beyond either model's self-reported confidence, so agreement
|
||||
// earns a bonus on top of the better single estimate.
|
||||
const REVIEW_AGREEMENT_BONUS = 0.2;
|
||||
const REVIEW_AGREEMENT_CONFIDENCE_CAP = 0.97;
|
||||
|
||||
// Agreement must be judged on the RAW model answers, not the grounded ones:
|
||||
// catalog grounding has a genus-level fallback that can collapse two different
|
||||
// species onto the same catalog entry and fake an agreement.
|
||||
@@ -30,6 +38,11 @@ const decideReviewOutcome = ({ primaryResult, reviewResult, agrees }) => {
|
||||
accept: true,
|
||||
replace: reviewConfidence >= primaryConfidence,
|
||||
reason: 'review-confirmed-primary',
|
||||
confidence: clamp(
|
||||
Math.max(primaryConfidence, reviewConfidence) + REVIEW_AGREEMENT_BONUS,
|
||||
0.05,
|
||||
REVIEW_AGREEMENT_CONFIDENCE_CAP,
|
||||
),
|
||||
};
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user