Files

436 lines
21 KiB
Python
Executable File

#!/usr/bin/env python3
"""Search FoodData Central and import a provenance-preserving nutrition mapping."""
from __future__ import annotations
import argparse
import datetime as dt
import difflib
import json
import os
import re
import time
import urllib.parse
import urllib.request
import urllib.error
from pathlib import Path
import yaml
ROOT = Path(__file__).resolve().parent.parent
API = "https://api.nal.usda.gov/fdc/v1"
NUTRIENTS = {
"1008": "energy_kcal",
"2047": "energy_kcal",
"2048": "energy_kcal",
"1003": "protein_g",
"1004": "total_fat_g",
"1005": "carbohydrate_g",
"1079": "dietary_fiber_g",
"2000": "total_sugars_g",
"1093": "sodium_mg",
"1258": "saturated_fat_g",
"1253": "cholesterol_mg",
"1087": "calcium_mg",
"1089": "iron_mg",
"1092": "potassium_mg",
"1114": "vitamin_d_mcg",
}
def request(path: str, params: dict) -> dict:
key = os.environ.get("USDA_FDC_API_KEY")
if not key:
env_path = ROOT / ".env"
if env_path.exists():
for line in env_path.read_text(encoding="utf-8").splitlines():
name, separator, value = line.partition("=")
if separator and name.strip() == "USDA_FDC_API_KEY":
key = value.strip().strip("'\"")
break
key = key or "DEMO_KEY"
url = f"{API}/{path}?{urllib.parse.urlencode({**params, 'api_key': key})}"
with urllib.request.urlopen(url, timeout=30) as response:
return json.load(response)
def search(query: str, limit: int, data_types: str) -> None:
data = request("foods/search", {"query": query, "dataType": data_types, "pageSize": limit})
for food in data.get("foods", []):
print(f"{food['fdcId']}\t{food['dataType']}\t{food['description']}")
def normalized_text(value: str) -> str:
return " ".join(re.findall(r"[a-z0-9]+", value.lower()))
def match_score(term: str, description: str) -> float:
query = normalized_text(term)
candidate = normalized_text(description)
query_tokens = set(query.split())
candidate_tokens = set(candidate.split())
if not query_tokens:
return 0
recall = len(query_tokens & candidate_tokens) / len(query_tokens)
precision = len(query_tokens & candidate_tokens) / len(candidate_tokens)
score = 0.72 * recall + 0.18 * precision + 0.10 * difflib.SequenceMatcher(None, query, candidate).ratio()
head = normalized_text(description.split(",", 1)[0])
if head == query:
score += 0.2
elif query in candidate and not query_tokens.intersection(set(head.split())):
score -= 0.2
return max(0, min(score, 1))
def load_dataset(path: Path) -> list[dict]:
payload = json.loads(path.read_text(encoding="utf-8"))
return next(iter(payload.values()))
def propose(datasets: list[Path], output: Path, limit: int) -> None:
foods = [food for dataset in datasets for food in load_dataset(dataset) if isinstance(food, dict) and food.get("description")]
foods_by_id = {food["fdcId"]: food for food in foods}
recommendations_path = ROOT / "config" / "usda-review-recommendations.yaml"
recommendations = yaml.safe_load(recommendations_path.read_text()) if recommendations_path.exists() else {}
searchable = [(food, normalized_text(food["description"])) for food in foods]
proposals = []
for ingredient_path in sorted((ROOT / "culinary" / "ingredients").glob("*.yaml")):
ingredient = yaml.safe_load(ingredient_path.read_text(encoding="utf-8"))
if ingredient.get("nutrition_mapping_ids"):
continue
terms = [ingredient["name"], *(alias["name"] for alias in ingredient.get("aliases", []))]
scored: list[tuple[float, dict]] = []
for food, description in searchable:
score = max(match_score(term, food["description"]) for term in terms)
# Foundation records win close ties, without hiding the actual score.
rank = score + (0.015 if food.get("dataType") == "Foundation" else 0)
scored.append((rank, food))
matches = sorted(scored, key=lambda item: (-item[0], item[1]["fdcId"]))[:limit]
recommended_id = recommendations.get(ingredient["id"])
if recommended_id in foods_by_id and all(food["fdcId"] != recommended_id for _, food in matches):
matches.append((match_score(ingredient["name"], foods_by_id[recommended_id]["description"]), foods_by_id[recommended_id]))
proposals.append({
"ingredient_id": ingredient["id"],
"ingredient_name": ingredient["name"],
"candidates": [
{
"fdc_id": food["fdcId"],
"data_type": food["dataType"],
"description": food["description"],
"similarity": round(score - (0.015 if food.get("dataType") == "Foundation" else 0), 4),
}
for score, food in matches
],
})
output.parent.mkdir(parents=True, exist_ok=True)
output.write_text(yaml.safe_dump({"generated_at": dt.datetime.now(dt.timezone.utc).isoformat(), "candidates": proposals}, sort_keys=False), encoding="utf-8")
print(f"Wrote candidates for {len(proposals)} unmapped ingredients to {output}")
def normalized_nutrients(food: dict) -> dict[str, float]:
result: dict[str, float] = {}
for entry in food.get("foodNutrients", []):
nutrient = entry.get("nutrient", {})
nutrient_id = str(nutrient.get("id", entry.get("nutrientId", "")))
key = NUTRIENTS.get(nutrient_id)
amount = entry.get("amount")
if key and isinstance(amount, (int, float)):
# Foundation foods may expose multiple energy calculations. Prefer 2047,
# then 2048, over the older general energy entry 1008.
if key != "energy_kcal" or nutrient_id != "1008" or key not in result:
result[key] = amount
return dict(sorted(result.items()))
PORTION_UNITS = {
"cup": "cup_us", "tablespoon": "tablespoon_us", "tbsp": "tablespoon_us",
"teaspoon": "teaspoon_us", "tsp": "teaspoon_us", "milliliter": "milliliter",
"ml": "milliliter", "fluid ounce": "fluid_ounce_us", "fl oz": "fluid_ounce_us",
}
REJECTED_PORTION_WORDS = {"racc", "serving", "package", "container", "packet", "scoop"}
def normalized_portion(food: dict, portion: dict) -> dict | None:
amount, grams = portion.get("amount"), portion.get("gramWeight")
if not isinstance(amount, (int, float)) or amount <= 0 or not isinstance(grams, (int, float)) or grams <= 0:
return None
measure = portion.get("measureUnit") or {}
measure_name = str(measure.get("name") or measure.get("abbreviation") or "").strip().lower()
modifier = str(portion.get("modifier") or "").strip()
searchable = f"{measure_name} {modifier}".lower()
if any(re.search(rf"\b{word}\b", searchable) for word in REJECTED_PORTION_WORDS):
return None
unit_id, state = PORTION_UNITS.get(measure_name), modifier or None
if not unit_id:
patterns = [
(r"^cups?\b", "cup_us"), (r"^(?:tablespoons?|tbsp)\b", "tablespoon_us"),
(r"^(?:teaspoons?|tsp)\b", "teaspoon_us"), (r"^(?:fluid ounces?|fl oz)\b", "fluid_ounce_us"),
]
for pattern, candidate_unit in patterns:
match = re.match(pattern, modifier, re.IGNORECASE)
if match:
unit_id = candidate_unit
state = modifier[match.end():].lstrip(" ,-") or None
break
if not unit_id:
return None
fdc_id = int(food["fdcId"])
portion_id = str(portion.get("id") or f"{unit_id}_{amount}_{grams}")
return {
"id": f"usda_{fdc_id}_{portion_id}",
"from": {"quantity": float(amount), "unit_id": unit_id},
"to": {"quantity": float(grams), "unit_id": "gram"},
**({"state": state} if state else {}),
"data_points": int(portion.get("dataPoints") or 0),
"source": {
"source_type": "usda_fdc", "external_id": str(fdc_id),
"url": f"https://fdc.nal.usda.gov/food-details/{fdc_id}/measures",
"title": food["description"], "publisher": "USDA Agricultural Research Service",
"retrieved_at": dt.date.today().isoformat(), "reviewed": False,
"notes": f"FoodData Central portion ID {portion_id}.",
},
}
def mapped_usda_foods() -> list[tuple[str, int]]:
result = []
for path in sorted((ROOT / "culinary" / "source_mappings").glob("*.yaml")):
mapping = yaml.safe_load(path.read_text(encoding="utf-8"))
source = mapping.get("source", {})
subject = mapping.get("subject", {})
if mapping.get("mapping_type") == "nutrition" and mapping.get("status") == "reviewed" and source.get("source_type") == "usda_fdc" and subject.get("type") == "ingredient":
result.append((subject["id"], int(source["external_id"])))
return result
def propose_portions(output: Path, delay: float) -> None:
cache_dir = ROOT / "generated" / "usda-food-details"
cache_dir.mkdir(parents=True, exist_ok=True)
proposals = []
mappings = mapped_usda_foods()
for index, (ingredient_id, fdc_id) in enumerate(mappings):
cache_path = cache_dir / f"{fdc_id}.json"
if cache_path.exists():
food = json.loads(cache_path.read_text(encoding="utf-8"))
else:
try:
food = request(f"food/{fdc_id}", {})
except urllib.error.HTTPError as error:
if error.code == 404:
print(f"Skipping unavailable FDC {fdc_id} mapped to {ingredient_id}")
continue
raise
cache_path.write_text(json.dumps(food), encoding="utf-8")
if index + 1 < len(mappings):
time.sleep(delay)
portions = [candidate for portion in food.get("foodPortions", []) if (candidate := normalized_portion(food, portion))]
if portions:
ingredient = yaml.safe_load((ROOT / "culinary" / "ingredients" / f"{ingredient_id}.yaml").read_text(encoding="utf-8"))
proposals.append({"ingredient_id": ingredient_id, "ingredient_name": ingredient["name"], "fdc_id": fdc_id, "food_description": food["description"], "portions": portions})
output.parent.mkdir(parents=True, exist_ok=True)
output.write_text(yaml.safe_dump({"generated_at": dt.datetime.now(dt.timezone.utc).isoformat(), "proposals": proposals}, sort_keys=False), encoding="utf-8")
print(f"Wrote {sum(len(item['portions']) for item in proposals)} portion candidates for {len(proposals)} ingredients to {output}")
def apply_portions(selected: set[str]) -> None:
candidates_path = ROOT / "generated" / "usda-portion-candidates.yaml"
candidates = yaml.safe_load(candidates_path.read_text(encoding="utf-8"))["proposals"]
applied = 0
for proposal in candidates:
ingredient_path = ROOT / "culinary" / "ingredients" / f"{proposal['ingredient_id']}.yaml"
ingredient = yaml.safe_load(ingredient_path.read_text(encoding="utf-8"))
conversions = [item for item in ingredient.get("measure_conversions", []) if not item["id"].startswith("usda_") or item["id"] in selected]
ingredient["measure_conversions"] = conversions
existing = {item["id"]: index for index, item in enumerate(conversions)}
for portion in proposal["portions"]:
if portion["id"] in selected:
portion = {key: value for key, value in portion.items() if key != "data_points"}
portion["source"]["reviewed"] = True
portion["source"]["reviewed_at"] = dt.date.today().isoformat()
if portion["id"] in existing:
conversions[existing[portion["id"]]] = portion
else:
conversions.append(portion); applied += 1
if conversions:
ingredient_path.write_text(yaml.safe_dump(ingredient, sort_keys=False), encoding="utf-8")
print(f"Applied {applied} reviewed USDA portion equivalencies")
def apply_portion_decisions(path: Path) -> None:
payload = json.loads(path.read_text(encoding="utf-8"))
apply_portions(set(payload.get("accepted", [])))
def auto_apply_portions() -> None:
candidates_path = ROOT / "generated" / "usda-portion-candidates.yaml"
proposals = yaml.safe_load(candidates_path.read_text(encoding="utf-8"))["proposals"]
selected: set[str] = set()
for proposal in proposals:
# Keep preparation states distinct. For duplicate observations of the same
# unit and state, prefer USDA's largest underlying sample count.
best: dict[str, dict] = {}
for portion in proposal["portions"]:
key = portion["from"]["unit_id"]
current = best.get(key)
if current is None or portion.get("data_points", 0) > current.get("data_points", 0):
best[key] = portion
selected.update(portion["id"] for portion in best.values())
apply_portions(selected)
def names_from_usda() -> None:
names: dict[str, str] = {}
for path in sorted((ROOT / "culinary" / "source_mappings").glob("*.yaml")):
mapping = yaml.safe_load(path.read_text(encoding="utf-8"))
source, subject = mapping.get("source", {}), mapping.get("subject", {})
if mapping.get("mapping_type") == "nutrition" and mapping.get("status") == "reviewed" and source.get("source_type") == "usda_fdc" and subject.get("type") == "ingredient":
names[subject["id"]] = source["title"]
renamed = 0
duplicates: dict[str, list[str]] = {}
for ingredient_id, usda_name in sorted(names.items()):
ingredient_path = ROOT / "culinary" / "ingredients" / f"{ingredient_id}.yaml"
ingredient = yaml.safe_load(ingredient_path.read_text(encoding="utf-8"))
for alias in ingredient.get("aliases", []):
if alias.get("kind") == "legacy":
alias["kind"] = "search"
old_name = ingredient["name"]
if old_name != usda_name:
aliases = ingredient.setdefault("aliases", [])
alias_names = {alias["name"].casefold() for alias in aliases}
if old_name.casefold() != usda_name.casefold() and old_name.casefold() not in alias_names:
aliases.append({"name": old_name, "kind": "search"})
ingredient["name"] = usda_name
renamed += 1
ingredient_path.write_text(yaml.safe_dump(ingredient, sort_keys=False), encoding="utf-8")
duplicates.setdefault(usda_name.casefold(), []).append(ingredient_id)
collisions = {name: ids for name, ids in duplicates.items() if len(ids) > 1}
print(f"Renamed {renamed} ingredients from {len(names)} reviewed USDA mappings; preserved old names as aliases")
for name, ids in sorted(collisions.items()):
print(f"Duplicate USDA name {name!r}: {', '.join(ids)}")
def import_food(ingredient_id: str, fdc_id: int, reviewed: bool, dataset: Path | None = None) -> None:
ingredient_path = ROOT / "culinary" / "ingredients" / f"{ingredient_id}.yaml"
if not ingredient_path.exists():
raise SystemExit(f"Unknown ingredient: {ingredient_id}")
if dataset:
payload = json.loads(dataset.read_text(encoding="utf-8"))
foods = next(iter(payload.values()))
food = next((item for item in foods if item.get("fdcId") == fdc_id), None)
if not food:
raise SystemExit(f"FDC ID {fdc_id} not found in {dataset}")
else:
food = request(f"food/{fdc_id}", {})
data_type = food.get("dataType")
allowed = {"Foundation", "SR Legacy", "FNDDS", "Branded", "Experimental"}
if data_type not in allowed:
raise SystemExit(f"Unsupported USDA data type: {data_type!r}")
mapping_id = f"usda_fdc_{ingredient_id}_{fdc_id}"
today = dt.date.today().isoformat()
mapping = {
"schema_version": 2,
"id": mapping_id,
"subject": {"type": "ingredient", "id": ingredient_id},
"mapping_type": "nutrition",
"status": "reviewed" if reviewed else "candidate",
"source": {
"source_type": "usda_fdc",
"external_id": str(fdc_id),
"url": f"https://fdc.nal.usda.gov/food-details/{fdc_id}/nutrients",
"title": food["description"],
"publisher": "USDA Agricultural Research Service",
"retrieved_at": today,
"reviewed": reviewed,
**({"reviewed_at": today} if reviewed else {}),
},
"usda": {
"fdc_id": fdc_id,
"data_type": data_type,
"description": food["description"],
**({"brand_owner": food["brandOwner"]} if food.get("brandOwner") else {}),
**({"gtin_upc": food["gtinUpc"]} if food.get("gtinUpc") else {}),
},
"nutrition_per_100g": normalized_nutrients(food),
}
mapping_dir = ROOT / "culinary" / "source_mappings"
mapping_dir.mkdir(exist_ok=True)
mapping_path = mapping_dir / f"{mapping_id}.yaml"
mapping_path.write_text(yaml.safe_dump(mapping, sort_keys=False), encoding="utf-8")
ingredient = yaml.safe_load(ingredient_path.read_text(encoding="utf-8"))
ids = ingredient.setdefault("nutrition_mapping_ids", [])
if mapping_id not in ids:
ids.append(mapping_id)
ingredient_path.write_text(yaml.safe_dump(ingredient, sort_keys=False), encoding="utf-8")
print(f"Imported {mapping_id} as {mapping['status']}: {food['description']}")
def apply_decisions(path: Path, reviewed: bool, delay: float, datasets: list[Path]) -> None:
payload = json.loads(path.read_text(encoding="utf-8"))
decisions = payload.get("decisions", {})
if not isinstance(decisions, dict):
raise SystemExit("Decision file must contain a decisions object")
dataset_foods = {}
for dataset in datasets:
for food in load_dataset(dataset):
if isinstance(food, dict) and food.get("fdcId"):
dataset_foods[food["fdcId"]] = dataset
selected = [(ingredient_id, fdc_id) for ingredient_id, fdc_id in decisions.items() if isinstance(fdc_id, int)]
for index, (ingredient_id, fdc_id) in enumerate(selected):
import_food(ingredient_id, fdc_id, reviewed, dataset_foods.get(fdc_id))
if index + 1 < len(selected) and fdc_id not in dataset_foods:
time.sleep(delay)
skipped = sum(fdc_id is None for fdc_id in decisions.values())
print(f"Applied {len(selected)} selections; left {skipped} ingredients unmapped")
def main() -> None:
parser = argparse.ArgumentParser(description=__doc__)
sub = parser.add_subparsers(dest="command", required=True)
search_parser = sub.add_parser("search")
search_parser.add_argument("query")
search_parser.add_argument("--limit", type=int, default=10)
search_parser.add_argument("--data-types", default="Foundation,SR Legacy,FNDDS")
import_parser = sub.add_parser("import")
import_parser.add_argument("ingredient_id")
import_parser.add_argument("fdc_id", type=int)
import_parser.add_argument("--reviewed", action="store_true")
import_parser.add_argument("--dataset", type=Path, help="Official USDA JSON download (works without an API key)")
propose_parser = sub.add_parser("propose")
propose_parser.add_argument("--dataset", type=Path, action="append", required=True)
propose_parser.add_argument("--output", type=Path, default=ROOT / "generated" / "usda-candidates.yaml")
propose_parser.add_argument("--limit", type=int, default=5)
apply_parser = sub.add_parser("apply")
apply_parser.add_argument("decisions", type=Path)
apply_parser.add_argument("--reviewed", action="store_true")
apply_parser.add_argument("--delay", type=float, default=0.25, help="Seconds between API requests")
apply_parser.add_argument("--dataset", type=Path, action="append", default=[])
portions_parser = sub.add_parser("portions", help="Fetch USDA portion-weight candidates for reviewed mappings")
portions_parser.add_argument("--output", type=Path, default=ROOT / "generated" / "usda-portion-candidates.yaml")
portions_parser.add_argument("--delay", type=float, default=0.1)
portions_apply_parser = sub.add_parser("portions-apply", help="Apply reviewed portion decisions")
portions_apply_parser.add_argument("decisions", type=Path)
sub.add_parser("portions-auto", help="Trust USDA portions and prefer the largest observation count")
sub.add_parser("names-from-usda", help="Use reviewed USDA descriptions as canonical ingredient names")
args = parser.parse_args()
if args.command == "search":
search(args.query, args.limit, args.data_types)
elif args.command == "import":
import_food(args.ingredient_id, args.fdc_id, args.reviewed, args.dataset)
elif args.command == "propose":
propose(args.dataset, args.output, args.limit)
elif args.command == "apply":
apply_decisions(args.decisions, args.reviewed, args.delay, args.dataset)
elif args.command == "portions":
propose_portions(args.output, args.delay)
elif args.command == "portions-apply":
apply_portion_decisions(args.decisions)
elif args.command == "portions-auto":
auto_apply_portions()
else:
names_from_usda()
if __name__ == "__main__":
main()