"""LLM-backed fallback candidate generation for hard-to-resolve locations.""" from __future__ import annotations import json import re from dataclasses import dataclass from typing import Any, Iterable from app.core.countries import COUNTRY_ENTRIES, normalize_country from app.schemas.ai import SituationalAnalysisRequest from app.services.ai_client import AIProviderClient from app.services.ai_tools.evidence_store import normalize_search_evidence from app.services.ai_tools.web_search import WebSearchClient, WebSearchError from app.services.location.models import LocationCandidate, LocationQuery from app.services.location.resolvers.nominatim import build_default_nominatim_geocoder from app.services.location.text import ( coerce_str, normalize_country_text, normalize_text, parse_float, ) VALID_LLM_PRECISIONS = {"precise", "site", "city"} DEFAULT_MIN_CONFIDENCE = 0.55 MODEL_CONFIDENCE_WEIGHT = 0.25 _geocode_llm_city = build_default_nominatim_geocoder() _LLM_LOCATION_NAME_KEYS = ( "matched_location_name", "display_name", "location_name", "location", "place", "city", ) _NAME_HINT_STOPWORDS = { "ai", "cloud", "cluster", "compute", "computer", "gpu", "hpc", "mercury", "phase", "super", "supercomputer", } LLM_PRECISION_ALIASES = { "precise": "precise", "exact": "precise", "coordinate": "precise", "coordinates": "precise", "site": "site", "site level": "site", "site-level": "site", "site_level": "site", "facility": "site", "facility level": "site", "city": "city", "city level": "city", "city-level": "city", "city_level": "city", } @dataclass(frozen=True) class LocationLLMFallbackResult: candidates: list[LocationCandidate] attempted_queries: list[str] failure_reason: str | None = None @dataclass(frozen=True) class LocationSearchEvidenceResult: evidence: list[dict[str, Any]] attempted_queries: list[str] failure_reason: str | None = None @dataclass(frozen=True) class LocationEvidenceScore: score: float model_confidence: float source_quality: float entity_match: float geography_match: float precision_quality: float conflict_penalty: float weak_evidence_penalty: float name_location_hint: float summary: str def _first_json_object(text: str) -> dict[str, Any] | None: stripped = text.strip() if not stripped: return None if stripped.startswith("```"): stripped = re.sub(r"^```(?:json)?\s*", "", stripped, flags=re.IGNORECASE) stripped = re.sub(r"\s*```$", "", stripped) try: data = json.loads(stripped) return data if isinstance(data, dict) else None except json.JSONDecodeError: pass start = stripped.find("{") end = stripped.rfind("}") if start < 0 or end <= start: return None try: data = json.loads(stripped[start : end + 1]) except json.JSONDecodeError: return None return data if isinstance(data, dict) else None def _compact_evidence(value: Any) -> str: if isinstance(value, list): parts = [_evidence_label(item) for item in value if _evidence_label(item)] return "; ".join(parts[:3]) return coerce_str(value) def _evidence_items(value: Any) -> list[dict[str, Any]]: if isinstance(value, list): raw_items = value elif value in (None, ""): raw_items = [] else: raw_items = [value] items: list[dict[str, Any]] = [] for item in raw_items: if isinstance(item, dict): items.append(dict(item)) else: text = coerce_str(item) if text: items.append({"text": text}) return items def _evidence_label(item: Any) -> str: if isinstance(item, dict): source = coerce_str(item.get("source") or item.get("title") or item.get("name")) url = coerce_str(item.get("url")) text = coerce_str(item.get("text") or item.get("quote") or item.get("summary")) if source and url: return f"{source} ({url})" if source: return source if url: return url return text return coerce_str(item) def _normalize_llm_precision(value: Any) -> str: text = coerce_str(value).lower() return LLM_PRECISION_ALIASES.get(text, text) def _detect_country_in_text(text: str) -> str: normalized_text = normalize_text(text) if not normalized_text: return "" for canonical, aliases in COUNTRY_ENTRIES: variants = [canonical, *aliases] for variant in variants: normalized_variant = normalize_text(variant) if normalized_variant and normalized_variant in normalized_text: return canonical return "" def _extract_city_from_text(text: str, *, country: str | None = None) -> str: patterns = [ r"\(([^()]{2,80})\)", r"\blocated\s+(?:in|at)\s+([^,.;()\n]{2,80})(?:,\s*([^.;()\n]{2,80}))?", r"\bbased\s+in\s+([^,.;()\n]{2,80})(?:,\s*([^.;()\n]{2,80}))?", r"\b位[于於]\s*(?:[^,。;;\n]{0,40}?的\s*)?([^,。;;()\n]{2,40})", ] normalized_country = normalize_text(country) for pattern in patterns: match = re.search(pattern, text, flags=re.IGNORECASE) if not match: continue for group in match.groups(): candidate = coerce_str(group) if not candidate: continue candidate = re.sub(r"^(?:the\s+city\s+of|city\s+of)\s+", "", candidate, flags=re.I) candidate = candidate.strip(" -–—::,,。.;;") if not candidate: continue if normalized_country and normalize_text(candidate) == normalized_country: continue if normalize_country(candidate): continue return candidate return "" def _payload_from_free_text(text: str, *, query: LocationQuery) -> dict[str, Any] | None: """Build a conservative payload when the model answered in prose. This is deliberately small: it only extracts a country and a city/place-like phrase. The normal scoring and geocoding gates still decide whether the result can become a candidate. """ if not coerce_str(text): return None country = _detect_country_in_text(text) or normalize_country_text(query.country) city = _extract_city_from_text(text, country=country) if not city or not country: return None evidence_text = " ".join(coerce_str(text).split())[:500] return { "precision": "city", "confidence": 0.55, "city": city, "country": country, "matched_location_name": f"{city}, {country}", "evidence": [ { "source": "LLM prose location factcheck", "source_type": "generic", "entity_match": bool( normalize_text(query.name) and normalize_text(query.name) in normalize_text(text) ), "text": evidence_text, } ], "reasoning_summary": "Location extracted from a non-JSON LLM answer.", "parse_strategy": "free_text_location_extraction", } def _query_name_city_terms(query: LocationQuery) -> list[str]: values = [ query.name, *query.aliases, (query.extra or {}).get("site"), ] terms: list[str] = [] seen: set[str] = set() for value in values: text = coerce_str(value) if not text: continue for raw_token in re.findall(r"[A-Za-z][A-Za-z.'-]{2,}|[\u4e00-\u9fff]{2,}", text): token = raw_token.strip(" .'-") key = normalize_text(token) if not key or key in seen or key in _NAME_HINT_STOPWORDS: continue seen.add(key) terms.append(token.title() if token.isupper() else token) return terms[:5] def _payload_from_query_name_geocode(query: LocationQuery) -> dict[str, Any] | None: """Use entity-name city hints only after LLM parsing fails. The hint is accepted only when the derived term geocodes to a city-like result in the query country. This keeps names such as "MUSICA Phase 1" from becoming arbitrary coordinates while allowing "TAIPEI-1" -> Taipei. """ country = normalize_country_text(query.country) if not country: return None for term in _query_name_city_terms(query): geocode_query = f"{term}, {country}" try: result = _geocode_llm_city(geocode_query) except Exception: continue if not isinstance(result, dict): continue latitude = parse_float(result.get("lat")) longitude = parse_float(result.get("lon")) if latitude in (None, 0.0) or longitude in (None, 0.0): continue address = result.get("address") if isinstance(result.get("address"), dict) else {} city = ( address.get("city") or address.get("town") or address.get("village") or address.get("municipality") or address.get("suburb") ) result_country = normalize_country_text(address.get("country") or country) if not city or normalize_text(result_country) != normalize_text(country): continue if normalize_text(term) not in normalize_text(city) and normalize_text(term) not in normalize_text(result.get("display_name")): continue return { "latitude": latitude, "longitude": longitude, "precision": "city", "confidence": 0.50, "city": city, "region": address.get("state") or address.get("region"), "country": result_country, "matched_location_name": result.get("display_name") or geocode_query, "evidence": [ { "source": "Entity name city hint", "source_type": "generic", "entity_match": True, "text": ( f"Derived city term '{term}' from entity name " f"'{coerce_str(query.name)}' and verified it by geocoding." ), } ], "reasoning_summary": "City derived from entity name after LLM parsing failed.", "parse_strategy": "query_name_city_hint", "coordinate_source": "nominatim_city_fallback", } return None def _extract_llm_coordinates(payload: dict[str, Any]) -> tuple[float | None, float | None]: latitude = parse_float( payload.get("latitude") if payload.get("latitude") not in (None, "") else payload.get("lat") ) longitude = parse_float( payload.get("longitude") if payload.get("longitude") not in (None, "") else ( payload.get("lon") if payload.get("lon") not in (None, "") else payload.get("lng") ) ) if latitude not in (None, 0.0) and longitude not in (None, 0.0): return latitude, longitude coordinates = payload.get("coordinates") or payload.get("coordinate") if isinstance(coordinates, dict): latitude = parse_float( coordinates.get("latitude") if coordinates.get("latitude") not in (None, "") else coordinates.get("lat") ) longitude = parse_float( coordinates.get("longitude") if coordinates.get("longitude") not in (None, "") else ( coordinates.get("lon") if coordinates.get("lon") not in (None, "") else coordinates.get("lng") ) ) elif isinstance(coordinates, (list, tuple)) and len(coordinates) >= 2: first = parse_float(coordinates[0]) second = parse_float(coordinates[1]) if first is not None and second is not None: # GeoJSON-style [lon, lat] is the common interchange format. longitude, latitude = first, second return latitude, longitude def _fill_city_coordinates_from_geocoder( payload: dict[str, Any], *, query: LocationQuery, ) -> tuple[dict[str, Any], str | None]: city = coerce_str(payload.get("city") or query.city) country = coerce_str(payload.get("country") or query.country) geocode_queries: list[str] = [] def add_geocode_query(value: str) -> None: cleaned = coerce_str(value) if cleaned and cleaned not in geocode_queries: geocode_queries.append(cleaned) if city and country: add_geocode_query(f"{city}, {country}") for key in _LLM_LOCATION_NAME_KEYS: value = payload.get(key) if not isinstance(value, str): continue if country and country.lower() not in value.lower(): add_geocode_query(f"{value}, {country}") add_geocode_query(value) if not geocode_queries: return payload, None failures: list[str] = [] geocode_query = "" result: dict[str, Any] | None = None for candidate_query in geocode_queries: geocode_query = candidate_query try: maybe_result = _geocode_llm_city(geocode_query) except Exception as exc: failures.append(f"{geocode_query}: {exc}") continue if not isinstance(maybe_result, dict): failures.append(f"{geocode_query}: no result") continue latitude = parse_float(maybe_result.get("lat")) longitude = parse_float(maybe_result.get("lon")) if latitude in (None, 0.0) or longitude in (None, 0.0): failures.append(f"{geocode_query}: invalid coordinates") continue result = maybe_result break if result is None: detail = "; ".join(failures[:3]) or "no usable geocode query" return payload, f"city geocode fallback found no usable result ({detail})" latitude = parse_float(result.get("lat")) longitude = parse_float(result.get("lon")) if latitude in (None, 0.0) or longitude in (None, 0.0): return payload, f"city geocode fallback returned invalid coordinates for '{geocode_query}'" address = result.get("address") if isinstance(result.get("address"), dict) else {} city = ( city or address.get("city") or address.get("town") or address.get("village") or address.get("municipality") or address.get("suburb") ) country = country or address.get("country") try: precision = _normalize_llm_precision(payload.get("precision")) or "city" except Exception: precision = "city" filled = { **payload, "latitude": latitude, "longitude": longitude, "precision": precision, "city": payload.get("city") or city, "region": payload.get("region") or address.get("state") or address.get("region"), "country": payload.get("country") or address.get("country") or country, "matched_location_name": ( payload.get("matched_location_name") or result.get("display_name") or geocode_query ), "coordinate_source": "nominatim_city_fallback", } return filled, None def _truthy_evidence_field(item: dict[str, Any], *keys: str) -> bool: for key in keys: value = item.get(key) if isinstance(value, bool): if value: return True elif coerce_str(value).lower() in {"true", "yes", "exact", "strong"}: return True return False def _source_quality_score(evidence_items: list[dict[str, Any]]) -> float: best = 0.0 for item in evidence_items: source_type = normalize_text( item.get("source_type") or item.get("type") or item.get("source_kind") or "" ) source_text = normalize_text( " ".join( [ coerce_str(item.get("source")), coerce_str(item.get("url")), coerce_str(item.get("text")), coerce_str(item.get("summary")), ] ) ) combined = f"{source_type} {source_text}" if any(token in combined for token in ("official", "government", "gov", "edu", "university")): best = max(best, 0.35) elif any(token in combined for token in ("database", "registry", "wikipedia", "news", "press")): best = max(best, 0.25) elif combined.strip(): best = max(best, 0.15) return best def _entity_match_score(payload: dict[str, Any], query: LocationQuery, evidence_items: list[dict[str, Any]]) -> float: if any( _truthy_evidence_field(item, "entity_match", "matches_entity", "name_match") for item in evidence_items ): return 0.25 names = [ query.name, *query.aliases, (query.extra or {}).get("site"), (query.extra or {}).get("operator"), (query.extra or {}).get("organization"), ] needles = [normalize_text(name) for name in names if normalize_text(name)] haystack = normalize_text( " ".join( [ coerce_str(payload.get("matched_location_name")), coerce_str(payload.get("reasoning_summary")), *[_evidence_label(item) for item in evidence_items], ] ) ) if needles and any(needle in haystack for needle in needles): return 0.25 return 0.0 def _geography_match_score(payload: dict[str, Any], query: LocationQuery) -> float: city = normalize_text(payload.get("city") or query.city) country = normalize_text(normalize_country_text(payload.get("country") or query.country)) context_country = normalize_text(normalize_country_text(query.country)) if city and country and (not context_country or country == context_country): return 0.20 if country and (not context_country or country == context_country): return 0.05 return 0.0 def _precision_quality_score(precision: str) -> float: return { "precise": 0.15, "site": 0.12, "city": 0.08, }.get(precision, 0.0) def _name_location_hint_score(payload: dict[str, Any], query: LocationQuery) -> float: query_name = normalize_text(query.name) city = normalize_text(payload.get("city") or query.city) matched_name = normalize_text(payload.get("matched_location_name")) if not query_name or not city: return 0.0 if city in query_name or query_name in city: return 0.07 if matched_name and (city in matched_name) and any(part in query_name for part in city.split()): return 0.04 return 0.0 def _ambiguity_text(payload: dict[str, Any], evidence_items: list[dict[str, Any]]) -> str: return normalize_text( " ".join( [ coerce_str(payload.get("ambiguity")), coerce_str(payload.get("conflicts")), coerce_str(payload.get("reasoning_summary")), *[_evidence_label(item) for item in evidence_items], ] ) ) def _conflict_penalty(payload: dict[str, Any], evidence_items: list[dict[str, Any]]) -> float: penalty = 0.0 ambiguity_text = _ambiguity_text(payload, evidence_items) if any(token in ambiguity_text for token in ("conflict", "contradict", "inconsistent")): penalty += 0.35 if any( _truthy_evidence_field(item, "has_conflict", "conflicting") for item in evidence_items ): penalty += 0.35 return min(penalty, 0.45) def _weak_evidence_penalty( payload: dict[str, Any], evidence_items: list[dict[str, Any]], *, entity_match: float, geography_match: float, conflict_penalty: float, ) -> float: ambiguity_text = _ambiguity_text(payload, evidence_items) penalty = 0.0 if any(token in ambiguity_text for token in ("ambiguous", "unclear", "weak", "guess")): penalty += 0.20 if any(_truthy_evidence_field(item, "ambiguous") for item in evidence_items): penalty += 0.15 if conflict_penalty == 0.0 and entity_match > 0 and geography_match >= 0.20: return min(penalty, 0.15) return min(penalty, 0.30) def _score_llm_location_payload( payload: dict[str, Any], *, query: LocationQuery, precision: str, ) -> LocationEvidenceScore: model_confidence = parse_float(payload.get("confidence")) model_confidence = min(max(model_confidence if model_confidence is not None else 0.0, 0.0), 1.0) evidence_items = _evidence_items(payload.get("evidence")) source_quality = _source_quality_score(evidence_items) entity_match = _entity_match_score(payload, query, evidence_items) geography_match = _geography_match_score(payload, query) precision_quality = _precision_quality_score(precision) conflict_penalty = _conflict_penalty(payload, evidence_items) weak_evidence_penalty = _weak_evidence_penalty( payload, evidence_items, entity_match=entity_match, geography_match=geography_match, conflict_penalty=conflict_penalty, ) name_location_hint = _name_location_hint_score(payload, query) score = ( model_confidence * MODEL_CONFIDENCE_WEIGHT + source_quality + entity_match + geography_match + precision_quality + name_location_hint - conflict_penalty - weak_evidence_penalty ) score = min(max(score, 0.0), 1.0) summary = ( f"combined={score:.2f}; model={model_confidence:.2f}; " f"source={source_quality:.2f}; entity={entity_match:.2f}; " f"geo={geography_match:.2f}; precision={precision_quality:.2f}; " f"conflict={conflict_penalty:.2f}; weak={weak_evidence_penalty:.2f}; " f"name_hint={name_location_hint:.2f}" ) return LocationEvidenceScore( score=score, model_confidence=model_confidence, source_quality=source_quality, entity_match=entity_match, geography_match=geography_match, precision_quality=precision_quality, conflict_penalty=conflict_penalty, weak_evidence_penalty=weak_evidence_penalty, name_location_hint=name_location_hint, summary=summary, ) def _candidate_from_payload( payload: dict[str, Any], *, query: LocationQuery, entity_type: str, min_confidence: float, ) -> tuple[LocationCandidate | None, str | None]: latitude, longitude = _extract_llm_coordinates(payload) if latitude in (None, 0.0) or longitude in (None, 0.0): return None, "missing, invalid, or zero latitude/longitude" precision = _normalize_llm_precision(payload.get("precision")) if precision not in VALID_LLM_PRECISIONS: return None, f"precision '{payload.get('precision')}' is not precise/site/city" city = coerce_str(payload.get("city")) or query.city or None country = ( normalize_country_text(payload.get("country")) or normalize_country_text(query.country) or query.country ) evidence_score = _score_llm_location_payload(payload, query=query, precision=precision) if evidence_score.score < min_confidence: return None, ( f"combined evidence score {evidence_score.score:.2f} is below minimum " f"{min_confidence}; {evidence_score.summary}" ) confidence = evidence_score.score matched_location_name = ( coerce_str(payload.get("matched_location_name")) or coerce_str(payload.get("display_name")) or coerce_str(query.name) or "LLM factcheck location" ) evidence = _compact_evidence(payload.get("evidence")) reasoning_summary = coerce_str(payload.get("reasoning_summary")) source_note_parts = ["LLM location factcheck fallback"] if payload.get("coordinate_source") == "nominatim_city_fallback": source_note_parts.append("coordinates: Nominatim city fallback") if evidence: source_note_parts.append(f"evidence: {evidence}") if reasoning_summary: source_note_parts.append(f"summary: {reasoning_summary}") source_note_parts.append(f"score: {evidence_score.summary}") extra = query.extra or {} matched_fields = tuple( field for field in ("name", "site", "operator", "organization", "city", "country") if ( (field in {"name", "city", "country"} and getattr(query, field, None)) or coerce_str(extra.get(field)) ) ) or ("llm_factcheck",) return LocationCandidate( latitude=float(latitude), longitude=float(longitude), display_name=matched_location_name, precision=precision, confidence=confidence, query=f"llm_factcheck:{entity_type}:{coerce_str(query.name) or 'unknown'}", source="llm_location_factcheck", source_note="; ".join(source_note_parts), matched_fields=matched_fields, needs_confirmation=True, city=city, region=coerce_str(payload.get("region")) or query.region or None, country=country or None, matched_location_name=matched_location_name, location_verified_at=None, suggested_registry_entry={ "canonical_name": matched_location_name, "aliases": list( { value for value in [ coerce_str(query.name), *[coerce_str(alias) for alias in query.aliases], coerce_str(extra.get("operator")), coerce_str(extra.get("site")), ] if value } ), "operator": coerce_str(extra.get("operator")) or None, "site": coerce_str(extra.get("site")) or None, "country": country or None, "city": city, "region": coerce_str(payload.get("region")) or query.region or None, "latitude": float(latitude), "longitude": float(longitude), "precision": precision, "confidence": confidence, "source_note": "; ".join(source_note_parts), "llm_model_confidence": evidence_score.model_confidence, "llm_combined_confidence": evidence_score.score, "llm_score_breakdown": { "source_quality": evidence_score.source_quality, "entity_match": evidence_score.entity_match, "geography_match": evidence_score.geography_match, "precision_quality": evidence_score.precision_quality, "conflict_penalty": evidence_score.conflict_penalty, "weak_evidence_penalty": evidence_score.weak_evidence_penalty, "name_location_hint": evidence_score.name_location_hint, }, }, raw_payload={ "llm_payload": payload, "search_evidence": payload.get("search_evidence") or [], }, ), None def _normalize_llm_payload(payload: dict[str, Any]) -> dict[str, Any]: for key in ("candidate", "location", "result"): nested = payload.get(key) if isinstance(nested, dict): return nested return payload def _query_context(query: LocationQuery) -> dict[str, Any]: extra = dict(query.extra or {}) return { "name": query.name, "aliases": list(query.aliases), "city": query.city, "region": query.region, "country": query.country, "source_latitude": query.source_latitude, "source_longitude": query.source_longitude, "extra": extra, } def _observations(query: LocationQuery, attempted_queries: Iterable[str]) -> list[str]: extra = query.extra or {} fields = [ ("name", query.name), ("aliases", ", ".join(query.aliases)), ("site", extra.get("site")), ("operator", extra.get("operator")), ("organization", extra.get("organization")), ("city", query.city), ("region", query.region), ("country", query.country), ("source", extra.get("source")), ("source_id", extra.get("source_id")), ("collector", extra.get("collector")), ] observations = [ f"{label}: {value}" for label, value in fields if coerce_str(value) ] attempts = [coerce_str(item) for item in attempted_queries if coerce_str(item)] if attempts: observations.append("previous resolver attempts: " + " | ".join(attempts[:12])) return observations def _location_search_query(query: LocationQuery, entity_type: str) -> str: extra = query.extra or {} parts = [ coerce_str(query.name), coerce_str(extra.get("site")), coerce_str(extra.get("operator")), coerce_str(extra.get("organization")), coerce_str(query.city), coerce_str(query.country), "physical location", ] if entity_type == "bgp_collector": parts.append("route collector city") elif entity_type == "compute_center": parts.append("datacenter supercomputer facility city") return " ".join(part for part in parts if part) async def collect_location_search_evidence( *, web_search_client: WebSearchClient, query: LocationQuery, entity_type: str, max_results: int = 5, ) -> LocationSearchEvidenceResult: search_query = _location_search_query(query, entity_type) attempt = f"web_search:{entity_type}:{search_query}" try: evidence = await web_search_client.search(search_query, max_results=max_results) except WebSearchError as exc: return LocationSearchEvidenceResult( evidence=[], attempted_queries=[attempt], failure_reason=f"WebSearch location evidence failed: {exc}", ) except Exception as exc: return LocationSearchEvidenceResult( evidence=[], attempted_queries=[attempt], failure_reason=f"WebSearch location evidence unavailable: {exc}", ) normalized = normalize_search_evidence(evidence, limit=max_results) if not normalized: return LocationSearchEvidenceResult( evidence=[], attempted_queries=[attempt], failure_reason="WebSearch returned no usable location evidence.", ) return LocationSearchEvidenceResult( evidence=normalized, attempted_queries=[attempt], failure_reason=None, ) async def _repair_location_payload_from_text( *, provider_client: AIProviderClient, raw_text: str, query: LocationQuery, entity_type: str, ) -> dict[str, Any] | None: """Second-pass structure repair for models that answer in prose. The first LLM call owns the factcheck. This call is intentionally framed as extraction/normalization only; it should not introduce new facts. """ if not coerce_str(raw_text): return None request = SituationalAnalysisRequest( title=f"Normalize location factcheck for {entity_type}", objective=( "Convert the supplied location factcheck text into exactly one strict " "JSON object. Extract only facts present in the text or original query." ), context={ "entity_type": entity_type, "location_query": _query_context(query), "raw_location_factcheck_text": raw_text[:4000], "required_json_schema": { "latitude": "number|null", "longitude": "number|null", "precision": "precise|site|city", "confidence": "number from 0 to 1", "city": "string|null", "region": "string|null", "country": "string|null", "matched_location_name": "string", "evidence": "array of objects with source/source_type/entity_match/text/url when present", "ambiguity": "string|null", "reasoning_summary": "short string", }, }, observations=[], constraints=[ "Return only strict JSON. Do not wrap it in markdown.", "Do not add new evidence or locations that are not present in the supplied text.", "If exact coordinates are absent but a city and country are present, set latitude and longitude to null and precision to city.", "Use confidence 0.55-0.70 for credible city-level text; use lower confidence for weak or ambiguous text.", ], ) try: response = await provider_client.analyze(request) except Exception: return None payload = _first_json_object(response.content) return _normalize_llm_payload(payload) if isinstance(payload, dict) else None async def collect_llm_location_fallback_candidate( *, provider_client: AIProviderClient, query: LocationQuery, entity_type: str, attempted_queries: Iterable[str] = (), search_evidence: list[dict[str, Any]] | None = None, min_confidence: float = DEFAULT_MIN_CONFIDENCE, ) -> LocationLLMFallbackResult: """Ask the configured LLM for one fact-checked location candidate. The result is intentionally conservative: invalid, low-confidence, or non-city-level responses are treated as no candidate. Callers should only use this in user-triggered collection flows. """ attempt = f"llm_factcheck:{entity_type}:{coerce_str(query.name) or 'unknown'}" if search_evidence is not None and not search_evidence: return LocationLLMFallbackResult( candidates=[], attempted_queries=[attempt], failure_reason="LLM location factcheck skipped: no WebSearch evidence.", ) request = SituationalAnalysisRequest( title=f"Location factcheck fallback for {entity_type}", objective=( "Return exactly one JSON object for the most likely physical location. " "Use only fact-checkable public knowledge; return null fields rather " "than guessing when evidence is weak." ), context={ "entity_type": entity_type, "location_query": _query_context(query), "search_evidence": search_evidence or [], "required_json_schema": { "latitude": "number", "longitude": "number", "precision": "precise|site|city", "confidence": "number from 0 to 1", "city": "string|null", "region": "string|null", "country": "string|null", "matched_location_name": "string", "evidence": "array of short source/evidence phrases", "evidence[].source_type": "official|government|academic|database|news|generic", "evidence[].entity_match": "boolean when the evidence names the queried entity", "ambiguity": "string|null describing same-name conflicts or contradictory sources", "reasoning_summary": "short string", }, }, observations=_observations(query, attempted_queries), constraints=[ "Return only strict JSON. Do not wrap it in markdown.", "Do not return country-level, regional-only, or unknown precision.", "Do not invent coordinates. Use lower confidence when evidence is incomplete.", "Calibrate model confidence using this rubric: 0.85-1.0 for exact facility coordinates backed by an authoritative source; 0.70-0.84 for a confirmed facility/campus with strong public evidence; 0.55-0.69 for a confirmed city-level location backed by credible sources but without exact facility coordinates; 0.35-0.54 for weak or ambiguous city evidence; below 0.35 when the location is mostly a guess.", "Return evidence as objects when possible, including source, url, source_type, and entity_match.", "Include source names or URLs in evidence when known. The backend will recompute the final confidence from model confidence plus evidence quality.", "If search_evidence is provided, use only that evidence as factual support.", "Prefer the facility/site if known; otherwise use the best supported city.", ], ) try: response = await provider_client.analyze(request) except Exception as exc: return LocationLLMFallbackResult( candidates=[], attempted_queries=[attempt], failure_reason=f"LLM location factcheck failed: {exc}", ) payload = _first_json_object(response.content) if payload is None: payload = await _repair_location_payload_from_text( provider_client=provider_client, raw_text=response.content, query=query, entity_type=entity_type, ) if payload is None: payload = _payload_from_free_text(response.content, query=query) if payload is None: payload = _payload_from_query_name_geocode(query) if payload is None: return LocationLLMFallbackResult( candidates=[], attempted_queries=[attempt], failure_reason=( "LLM location factcheck did not return a parseable city-level " "location fact." ), ) payload = _normalize_llm_payload(payload) if search_evidence: payload["search_evidence"] = search_evidence existing_evidence = _evidence_items(payload.get("evidence")) payload["evidence"] = [ *existing_evidence, *[ { "source": item.get("title") or item.get("url"), "url": item.get("url"), "text": item.get("snippet") or item.get("content"), "source_type": "web_search", "entity_match": True, } for item in search_evidence if isinstance(item, dict) ], ] latitude, longitude = _extract_llm_coordinates(payload) city_geocode_failure = None if latitude in (None, 0.0) or longitude in (None, 0.0): payload, city_geocode_failure = _fill_city_coordinates_from_geocoder( payload, query=query, ) candidate, rejection_reason = _candidate_from_payload( payload, query=query, entity_type=entity_type, min_confidence=min_confidence, ) if candidate is None: if city_geocode_failure and rejection_reason == "missing, invalid, or zero latitude/longitude": rejection_reason = f"{rejection_reason}; {city_geocode_failure}" return LocationLLMFallbackResult( candidates=[], attempted_queries=[attempt], failure_reason=( "LLM location factcheck returned no acceptable city-level candidate" + (f": {rejection_reason}." if rejection_reason else ".") ), ) return LocationLLMFallbackResult( candidates=[candidate], attempted_queries=[attempt], failure_reason=None, )