NANI-Nithin commited on
Commit
3b738a4
·
1 Parent(s): e25ee08

feat: city_context: Add real-world city context from Wikipedia to generation prompt

Browse files
app/prompts/game_generation.txt CHANGED
@@ -12,6 +12,11 @@ Generate a location-based game in strict JSON format.
12
  ## Retrieved Examples
13
  {retrieved_examples}
14
 
 
 
 
 
 
15
  ## Safety
16
  - NO entering buildings or private property
17
  - NO proximity to water, traffic, or rail lines
 
12
  ## Retrieved Examples
13
  {retrieved_examples}
14
 
15
+ ## City Context (real-world reference data)
16
+ {city_context}
17
+
18
+ **Instructions:** Use the task patterns and game structures from the Retrieved Examples, but adapt them to fit the City Context above — replace Paris-specific locations with real landmarks, districts, and parks from the target city.
19
+
20
  ## Safety
21
  - NO entering buildings or private property
22
  - NO proximity to water, traffic, or rail lines
app/services/city_context.py ADDED
@@ -0,0 +1,216 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """City context enrichment via Wikipedia API.
2
+
3
+ Fetches real landmarks, districts, parks, and geographic data for any city
4
+ in the world — making the AI generation location-aware without needing to
5
+ expand the local dataset.
6
+ """
7
+
8
+ import json
9
+ import re
10
+ from typing import Optional
11
+ from urllib.request import urlopen, Request
12
+ from urllib.error import URLError
13
+ from urllib.parse import urlencode
14
+
15
+
16
+ # ── Simple in-memory cache ───────────────────────────────────────────────
17
+ _cache: dict[str, dict] = {}
18
+
19
+ USER_AGENT = "CityQuestAI/1.0 (hackathon; mailto:cityquest@example.com)"
20
+
21
+
22
+ def _wikipedia_api(action: str, params: dict) -> Optional[dict]:
23
+ """Call the Wikipedia API and return parsed JSON."""
24
+ params["action"] = action
25
+ params["format"] = "json"
26
+ query = urlencode(params)
27
+ url = f"https://en.wikipedia.org/w/api.php?{query}"
28
+
29
+ try:
30
+ req = Request(url, headers={"User-Agent": USER_AGENT})
31
+ with urlopen(req, timeout=10) as resp:
32
+ return json.loads(resp.read().decode("utf-8"))
33
+ except (URLError, OSError, json.JSONDecodeError) as e:
34
+ print(f"[city] Wikipedia API error: {e}")
35
+ return None
36
+
37
+
38
+ def _fetch_city_context(city: str) -> Optional[dict]:
39
+ """Fetch city information from Wikipedia.
40
+
41
+ Returns a dict with:
42
+ - ``landmarks``: list of notable landmarks
43
+ - ``districts``: list of neighbourhoods/districts
44
+ - ``parks``: list of parks/green spaces
45
+ - ``description``: short city description (2-3 sentences)
46
+ """
47
+ result = {
48
+ "landmarks": [],
49
+ "districts": [],
50
+ "parks": [],
51
+ "description": "",
52
+ }
53
+
54
+ # Handle common disambiguation cases
55
+ city_variants = [city]
56
+ if city.lower() == "new york":
57
+ city_variants.append("New York City")
58
+ elif city.lower() == "nyc":
59
+ city_variants = ["New York City"]
60
+ elif city.lower() in ("la", "los angeles"):
61
+ city_variants = ["Los Angeles"]
62
+
63
+ for variant in city_variants:
64
+ # Single API call — get the full page text
65
+ data = _wikipedia_api("query", {
66
+ "titles": variant,
67
+ "prop": "extracts",
68
+ "explaintext": "1",
69
+ "redirects": "1",
70
+ })
71
+ if not data:
72
+ continue
73
+
74
+ pages = data.get("query", {}).get("pages", {})
75
+ if not pages:
76
+ continue
77
+
78
+ page_id = next(iter(pages.keys()))
79
+ if page_id == "-1":
80
+ print(f"[city] No Wikipedia page found for '{variant}'")
81
+ continue
82
+
83
+ extract = pages[page_id].get("extract", "")
84
+ if not extract:
85
+ continue
86
+
87
+ # Description = first paragraph
88
+ paragraphs = [p.strip() for p in extract.split("\n\n") if p.strip()]
89
+ if paragraphs:
90
+ result["description"] = paragraphs[0][:600].strip()
91
+
92
+ # ── Landmark extraction ────────────────────────────────────────
93
+ # Wikipedia uses bullet lists (*) for landmarks/sights/tourist attractions
94
+ # Also look for sentences with landmark keywords in the first 50 paragraphs
95
+ landmark_candidates = set()
96
+ for para in paragraphs[:50]:
97
+ # Bullet-point lists are the best signal
98
+ if para.startswith("*"):
99
+ for line in para.split("\n"):
100
+ line = line.strip()
101
+ if line.startswith("*"):
102
+ clean = re.sub(r"^[*]\s*", "", line)
103
+ clean = re.sub(r"\[.*?\]", "", clean)
104
+ clean = re.sub(r"\s*\(.*?\)\s*", " ", clean).strip()
105
+ if 10 < len(clean) < 300:
106
+ landmark_candidates.add(clean)
107
+
108
+ # Plain sentences mentioning known attraction types
109
+ elif any(word in para.lower() for word in [
110
+ "landmark", "attraction", "museum", "cathedral", "square",
111
+ "monument", "palace", "garden", "market", "tower",
112
+ "bridge", "theatre", "gallery", "castle", "temple",
113
+ "shrine", "stadium", "park", "beach", "harbour",
114
+ ]):
115
+ for sent in re.split(r"(?<=[.!?])\s+", para):
116
+ if any(word in sent.lower() for word in [
117
+ "square", "museum", "cathedral", "church", "bridge",
118
+ "tower", "palace", "garden", "market", "theatre",
119
+ "gallery", "monument", "castle", "station", "stadium",
120
+ "temple", "shrine", "mosque", "plaza", "fountain",
121
+ "statue", "memorial", "opera", "aquarium", "zoo",
122
+ "beach", "waterfront", "basilica", "fort", "hall",
123
+ ]) and len(sent) > 20 and len(sent) < 300:
124
+ clean = re.sub(r"\[.*?\]", "", sent).strip()
125
+ landmark_candidates.add(clean)
126
+
127
+ # ── District/neighbourhood extraction ──────────────────────────
128
+ districts = set()
129
+ for para in paragraphs[:40]:
130
+ lower = para.lower()
131
+ if any(word in lower for word in ["district", "neighbourhood", "neighborhood", "quarter"]):
132
+ # Extract capitalized proper nouns near these keywords
133
+ for match in re.finditer(
134
+ r"\b([A-Z][a-zéèêëàâîïôûùçüöäñ]+(?:\s[A-Z][a-zéèêëàâîïôûùçüöäñ]+)*)\s+(district|neighbourhood|neighborhood|quarter)",
135
+ para,
136
+ ):
137
+ name = match.group(1).strip()
138
+ if name.lower() != city.lower() and 3 < len(name) < 40:
139
+ districts.add(name)
140
+
141
+ # ── Park extraction ────────────────────────────────────────────
142
+ parks = set()
143
+ for match in re.finditer(
144
+ r"\b([A-Z][a-zéèêëàâîïôûùçüöäñ]+(?:\s[A-Z][a-zéèêëàâîïôûùçüöäñ]+)*)\s+(Park|Gardens|Garden|Forest|Reserve)\b",
145
+ extract,
146
+ ):
147
+ name = match.group(0).strip()
148
+ if 5 < len(name) < 50:
149
+ parks.add(name)
150
+
151
+ # Assign results to the output dict
152
+ result["landmarks"] = sorted(landmark_candidates, key=len)[:10]
153
+ result["districts"] = sorted(districts, key=len)[:8]
154
+ result["parks"] = sorted(parks, key=len)[:5]
155
+
156
+ break # first successful variant wins
157
+
158
+ return result
159
+
160
+
161
+ def get_city_context(city: str) -> dict:
162
+ """Get city context, using a simple in-memory cache.
163
+
164
+ Args:
165
+ city: City name (e.g. "Tokyo", "New York", "Berlin").
166
+
167
+ Returns:
168
+ Dict with keys ``landmarks``, ``districts``, ``parks``, ``description``.
169
+ If the fetch fails, returns empty lists and a generic description.
170
+ """
171
+ key = city.strip().lower()
172
+ if key not in _cache:
173
+ print(f"[city] Fetching context for '{city}' …")
174
+ ctx = _fetch_city_context(city)
175
+ if ctx is None:
176
+ ctx = {
177
+ "landmarks": [],
178
+ "districts": [],
179
+ "parks": [],
180
+ "description": f"{city} is a major city.",
181
+ }
182
+ _cache[key] = ctx
183
+ print(f"[city] Got {len(ctx['landmarks'])} landmarks, "
184
+ f"{len(ctx['districts'])} districts, {len(ctx['parks'])} parks")
185
+ return _cache[key]
186
+
187
+
188
+ def build_city_section(city: str) -> str:
189
+ """Build a human-readable 'City Context' section for a prompt.
190
+
191
+ Args:
192
+ city: City name.
193
+
194
+ Returns:
195
+ A multi-line string to inject into the game generation prompt.
196
+ """
197
+ ctx = get_city_context(city)
198
+
199
+ lines = [f"### {city}"]
200
+ if ctx.get("description"):
201
+ lines.append(ctx["description"])
202
+
203
+ if ctx.get("districts"):
204
+ lines.append("\n**Notable districts/neighbourhoods:**")
205
+ lines.append(", ".join(ctx["districts"][:6]))
206
+
207
+ if ctx.get("parks"):
208
+ lines.append("\n**Parks & green spaces (great for outdoor games):**")
209
+ lines.append(", ".join(ctx["parks"][:4]))
210
+
211
+ if ctx.get("landmarks"):
212
+ lines.append("\n**Notable landmarks/monuments:**")
213
+ for lm in ctx["landmarks"][:6]:
214
+ lines.append(f"- {lm[:120]}")
215
+
216
+ return "\n".join(lines)
app/services/generator.py CHANGED
@@ -105,6 +105,17 @@ def build_generation_prompt(config: dict, retrieved_examples: list[dict]) -> str
105
  for task in ex.get('task_patterns', [])[:2]:
106
  examples_str += f" • {task.get('task_id')}: {task.get('points')} pts ({task.get('proof_type')})\n"
107
 
 
 
 
 
 
 
 
 
 
 
 
108
  # Load prompt template
109
  template_path = Path("app/prompts/game_generation.txt")
110
  if template_path.exists():
@@ -123,7 +134,7 @@ def build_generation_prompt(config: dict, retrieved_examples: list[dict]) -> str
123
 
124
  # Build prompt with all context
125
  prompt = template.format(
126
- city=config.get('city', 'Paris'),
127
  area=config.get('area', 'downtown'),
128
  game_type=config.get('game_type', 'scavenger_hunt'),
129
  duration_minutes=config.get('duration_minutes', 45),
@@ -132,6 +143,7 @@ def build_generation_prompt(config: dict, retrieved_examples: list[dict]) -> str
132
  age_group=config.get('age_group', 'adults'),
133
  location_type=config.get('location_type', 'mixed'),
134
  retrieved_examples=examples_str,
 
135
  output_schema=schema_str,
136
  )
137
 
 
105
  for task in ex.get('task_patterns', [])[:2]:
106
  examples_str += f" • {task.get('task_id')}: {task.get('points')} pts ({task.get('proof_type')})\n"
107
 
108
+ # ── Live city context via Wikipedia ─────────────────────────────────
109
+ # Inject real landmarks, districts, and parks so the model generates
110
+ # location-accurate content even for cities not in the dataset.
111
+ city = config.get('city', 'Paris')
112
+ city_context_str = ""
113
+ try:
114
+ from app.services.city_context import build_city_section
115
+ city_context_str = build_city_section(city)
116
+ except Exception as e:
117
+ print(f"[prompt] Wikipedia city context unavailable: {e}")
118
+
119
  # Load prompt template
120
  template_path = Path("app/prompts/game_generation.txt")
121
  if template_path.exists():
 
134
 
135
  # Build prompt with all context
136
  prompt = template.format(
137
+ city=city,
138
  area=config.get('area', 'downtown'),
139
  game_type=config.get('game_type', 'scavenger_hunt'),
140
  duration_minutes=config.get('duration_minutes', 45),
 
143
  age_group=config.get('age_group', 'adults'),
144
  location_type=config.get('location_type', 'mixed'),
145
  retrieved_examples=examples_str,
146
+ city_context=city_context_str,
147
  output_schema=schema_str,
148
  )
149