{
  "schemaVersion": "1.0.0",
  "protocolVersion": "1.0.0",
  "canonicalUrl": "https://www.bling.best/research/food-photo-estimate-benchmark",
  "status": "protocol-only",
  "published": "2026-08-01",
  "modified": "2026-08-01",
  "notice": "This release defines the evaluation before results exist. It publishes no accuracy rate, model ranking, clinical claim, or completed dataset.",
  "scope": {
    "product": "The evaluated bling product build and estimation configuration will be frozen and identified before data analysis.",
    "intendedUse": "Consumer food journaling with editable calorie and macronutrient estimates.",
    "excludedUses": [
      "diagnosis or treatment",
      "medication or insulin dosing",
      "management of a medical condition",
      "individualized clinical nutrition advice"
    ],
    "evaluationUnit": "An independent meal. Multiple photos or estimate stages from the same meal remain clustered under one meal identifier."
  },
  "studyQuestions": [
    "How far are first-pass energy and macronutrient estimates from documented reference values?",
    "How often do estimates fall within predeclared error bands?",
    "Which meal contexts and visually hidden ingredients are associated with larger error?",
    "How much does a structured user correction step change error without implying laboratory precision?"
  ],
  "sampling": {
    "governanceFloor": "A general benchmark release requires a 100-meal logistics pilot followed by at least 1,000 independent evaluation meals collected after the product and protocol are frozen. At least 200 evaluation meals remain in the confirmatory holdout, every reported primary context contains at least 50 meals, and collection spans at least three venues or household sources across multiple menu cycles.",
    "note": "The floor is a publication safeguard, not a claim of statistical power. The pilot is never pooled into the headline evaluation, and achieved confidence-interval precision will be reported.",
    "primaryContexts": [
      "single visible item",
      "home-cooked mixed dish",
      "restaurant or takeout meal",
      "packaged or branded food",
      "soup, beverage, or other low-visibility preparation"
    ],
    "inclusion": [
      "A complete meal or food item photographed before consumption",
      "A documented reference method and edible quantity",
      "A frozen product version and timestamped first-pass result",
      "Consent and de-identification status recorded before any image release"
    ],
    "exclusion": [
      "Missing reference quantity or unverifiable nutrition source",
      "Partial meals unless the consumed fraction is measured",
      "Images containing people or unrelated personal information",
      "Post-hoc removal because an estimate performed poorly"
    ],
    "isolation": "Development, pilot, and evaluation partitions are isolated by diner, meal, recipe lineage, restaurant or venue, and near-duplicate image group. The final evaluation images are new, non-public, and unavailable for prompt tuning before the protocol and system are frozen.",
    "distributions": [
      "A natural-use sample that preserves the observed product traffic distribution",
      "A balanced stress sample spanning cuisine, energy and portion bands, single versus mixed dishes, home versus restaurant or packaged food, beverages, sauces, hidden ingredients, occlusion, light, angle, container, and low-contrast preparations"
    ]
  },
  "referenceStandard": {
    "priority": [
      "Weighed edible ingredients plus measured cooked recipe yield",
      "Exact package Nutrition Facts label and consumed serving weight",
      "Official restaurant nutrition record matched to the ordered item and portion"
    ],
    "foodComposition": "USDA FoodData Central records will include the FDC identifier, data type, access date, and release or update date. A source is chosen to match the food form and preparation, not merely the nearest name.",
    "recipeHandling": "Ingredient nutrient totals are scaled by edible weight and the measured finished-recipe yield. Any retention or yield factor must name its source and version.",
    "weighing": "Start from a tared plate or container. Record each ingredient or locked recipe addition separately, the scale model, calibration date and mass, scale uncertainty, edible weight increment, final reweigh, preparation state, recipe revision, leftovers, and discarded oil or liquid.",
    "uncertainty": "Reference uncertainty, substitutions, leftovers, and unavailable values remain explicit fields; they are not silently converted into exact truth.",
    "adjudication": "Two qualified nutrition reviewers independently match ingredients to composition records; a third reviewer resolves disagreements. High-oil, high-water, sauce-heavy, and brand-variable subsets receive duplicate preparation or laboratory analysis to quantify reference-method error."
  },
  "acquisition": {
    "photoViews": "The primary analysis uses one ordinary consumer photo. Additional views are recorded separately and cannot be mixed into the primary result.",
    "requiredMetadata": [
      "device class",
      "camera view and distance band",
      "lighting condition",
      "plate or container type",
      "whether a scale reference is visible",
      "meal context and preparation visibility"
    ],
    "privacy": "Public images must have explicit publication permission, contain no person or unrelated personal information, and have removable metadata stripped."
  },
  "estimateCapture": {
    "stages": [
      "first-pass photo-only estimate",
      "post-review estimate after the predeclared correction prompts"
    ],
    "correctionPrompts": [
      "food identity",
      "portion or consumed amount",
      "cooking fat or sauce",
      "missing side or beverage",
      "database or branded-food match"
    ],
    "lockedFields": [
      "product and build version",
      "model provider, exact model name, snapshot or version, invocation region, and call timestamp",
      "estimation configuration identifier",
      "system and user prompt hashes, image resize and compression settings, temperature, seed when available, and parser version",
      "generation timestamp",
      "input photo identifiers",
      "foods returned",
      "energy and macronutrient outputs",
      "whether a result was unavailable"
    ],
    "execution": "The primary result is the first response produced through the real product path. Repeated calls are run only on a preregistered subset to quantify stochastic output and failure consistency; no run is selected because it scored best.",
    "reproducibilityLimit": "When a commercial model cannot be pinned to an immutable snapshot, the report must say exact reproduction is not guaranteed and preserve timestamped raw responses plus every available invocation setting."
  },
  "outcomes": {
    "primary": [
      "meal-level absolute energy error in kilocalories",
      "proportion within ±20% of reference energy"
    ],
    "secondary": [
      "signed energy error (bias)",
      "median absolute energy error",
      "root mean squared energy error",
      "aggregate energy wAPE: sum of absolute errors divided by sum of reference energy",
      "proportions within ±10% and ±30%",
      "gross error proportion above ±50%",
      "Bland–Altman 95% limits of agreement, with confidence intervals for bias and both limits",
      "absolute protein, carbohydrate, and fat error in grams",
      "protein, carbohydrate, and fat bias and aggregate wAPE",
      "visible major-component recall",
      "change from first-pass to post-review error",
      "unavailable, refused, unparseable, and unsupported-completion rates"
    ],
    "guardrails": [
      "Recognition accuracy is not reported as nutrition accuracy.",
      "Missing or failed estimates count as failures and are reported separately; they are not dropped.",
      "Macro percentage errors are not calculated when the reference value is too close to zero.",
      "No single universal accuracy percentage is used outside the frozen sample, version, and reference method.",
      "Correlation is not treated as agreement or accuracy.",
      "No threshold is described as an industry or regulatory standard unless a cited authority actually defines it."
    ]
  },
  "analysis": {
    "confidenceIntervals": "Report two-sided 95% confidence intervals using 10,000 meal-cluster bootstrap resamples. All photos and estimate stages from one meal stay in the same resample.",
    "aggregation": "Publish mean, median, interquartile range, and the complete empirical distribution where appropriate; do not rely on one average.",
    "agreement": "Inspect error against reference magnitude before applying fixed Bland–Altman limits. If variance changes materially with meal energy, use a preregistered log-ratio or regression-based limits-of-agreement analysis.",
    "comparison": "Any system or stage comparison uses meal-paired bootstrap difference intervals. Equivalence or non-inferiority requires a margin frozen before labels are opened; a non-significant difference is never called equivalent.",
    "multiplicity": "The two co-primary outcomes are declared before analysis. Confirmatory secondary or subgroup claims use a frozen hierarchical or false-discovery-rate procedure; all other subgroup analyses are labeled exploratory.",
    "stratification": [
      "meal context",
      "ingredient visibility",
      "single item versus mixed dish",
      "home-cooked versus restaurant or packaged",
      "photo condition",
      "cuisine only when the stratum meets the publication floor",
      "region and price band",
      "camera and lighting condition",
      "skin tone only when a hand or person-adjacent crop is retained with explicit consent"
    ],
    "missingness": "Publish a flow count from collected meals to analyzed meals, every exclusion reason, unavailable estimates, and missing reference nutrients.",
    "failureTaxonomy": [
      "food identity error",
      "portion or volume error",
      "hidden oil, sauce, sugar, or preparation error",
      "brand or recipe assumption error",
      "unit or structured-output parse error",
      "unsupported ingredient completion",
      "gross under- or over-estimate",
      "inappropriate medical or disease advice",
      "privacy exposure"
    ]
  },
  "preregistration": {
    "timing": "Register the protocol and statistical analysis plan before unlocking evaluation labels.",
    "registry": "The public release must name the registration record, timestamp, protocol file hash, image-manifest hash, reference-data version, product build, model configuration, prompt hashes, parser version, primary outcomes, margins, exclusions, missing-data rules, subgroup plan, and sample-size precision analysis.",
    "deviations": "Every departure from the frozen registration is enumerated, dated, justified, and kept out of the confirmatory result when it could bias the outcome."
  },
  "releaseGates": [
    {
      "id": "protocol-freeze",
      "requirement": "Publicly preregister the protocol, statistical analysis plan, observation schema, analysis code version, and content hashes before opening evaluation labels."
    },
    {
      "id": "version-lock",
      "requirement": "Freeze the product build, exact model or supplier snapshot, prompts, image pipeline, parser, region, invocation settings, and food-composition release; do not combine materially different versions in one headline result."
    },
    {
      "id": "sample-floor",
      "requirement": "Complete the separate pilot, 1,000-meal evaluation floor, 200-meal confirmatory holdout, source-diversity rule, and every reported stratum floor; otherwise label the release pilot-only with no general benchmark claim."
    },
    {
      "id": "reference-audit",
      "requirement": "Complete traceable weighing, recipe-yield and retention records, independent nutrition review with adjudication, and quantified reference uncertainty."
    },
    {
      "id": "complete-accounting",
      "requirement": "Account for every collected meal, failed estimate, exclusion, correction, and unavailable nutrient."
    },
    {
      "id": "agreement-analysis",
      "requirement": "Report confidence intervals for co-primary outcomes, Bland–Altman bias and limits of agreement, gross errors, failed outputs, and prespecified stress and subgroup results."
    },
    {
      "id": "contamination-check",
      "requirement": "Verify partition isolation and document the evidence that evaluation images, recipe lineages, venues, and near duplicates were unavailable for prompt tuning or result selection."
    },
    {
      "id": "reproducible-release",
      "requirement": "Publish de-identified row-level data where consent permits, the data dictionary, analysis code, checksums, and a versioned change log."
    },
    {
      "id": "privacy-review",
      "requirement": "Complete consent, EXIF and GPS removal, person and sensitive-background screening, retention and deletion controls, access review, and model-supplier data-use review."
    },
    {
      "id": "drift-plan",
      "requirement": "Publish a golden sentinel set, change triggers, monitoring cadence, alert thresholds, and the conditions that pause or retract an accuracy claim."
    },
    {
      "id": "claims-review",
      "requirement": "Keep conclusions within consumer journaling, name failure modes, and reject clinical, universal, or causally unsupported claims."
    }
  ],
  "reportingAlignment": {
    "applicable": "GRRAS and Bland–Altman reporting form the direct agreement-study backbone. TRIPOD-LLM and selected TRIPOD+AI transparency items govern model, prompt, data, confidence-interval, subgroup, failure, and open-science reporting. NIST AI RMF and its Generative AI Profile govern version, supplier, privacy, drift, incident, and retirement controls. USDA FoodData Central and Nutrition5k inform reference provenance and acquisition.",
    "limitation": "This is not a diagnostic-accuracy study, clinical prediction-model study, randomized intervention, or clinical decision-support evaluation. It therefore does not claim STARD-AI, TRIPOD+AI, CONSORT-AI, SPIRIT-AI, or DECIDE-AI compliance."
  },
  "machineReadable": {
    "protocol": "https://www.bling.best/research/food-photo-estimate-benchmark/v1.0.0/protocol.json",
    "observationSchema": "https://www.bling.best/research/food-photo-estimate-benchmark/v1.0.0/schema.json",
    "citation": "https://www.bling.best/research/food-photo-estimate-benchmark/v1.0.0/citation.bib"
  }
}