feat: add Jordan places migration scripts and enhance map controller with style-load safety and camera controls
This commit is contained in:
@@ -56,6 +56,7 @@ class NavigationCubit extends Cubit<NavigationState> {
|
||||
Timer? _movementInterpolationTimer;
|
||||
LatLng? _currentDisplayPosition;
|
||||
double _currentDisplayHeading = 0.0;
|
||||
double _currentCameraBearing = 0.0;
|
||||
LatLng? _animStartPosition;
|
||||
LatLng? _animTargetPosition;
|
||||
double _animStartHeading = 0.0;
|
||||
@@ -64,6 +65,13 @@ class NavigationCubit extends Cubit<NavigationState> {
|
||||
static const int _animTotalSteps = 25; // 25 steps * 40ms = 1000ms duration
|
||||
static const Duration _animTickDuration = Duration(milliseconds: 40);
|
||||
|
||||
void updateCameraBearing(double bearing) {
|
||||
_currentCameraBearing = bearing;
|
||||
if (_currentDisplayPosition != null) {
|
||||
_updateCarMarker(_currentDisplayPosition!, _currentDisplayHeading);
|
||||
}
|
||||
}
|
||||
|
||||
Future<void> _init() async {
|
||||
print("🚀 [NavigationCubit] Initializing NavigationCubit...");
|
||||
CarPlatformBridge.ensureInitialized();
|
||||
@@ -493,20 +501,48 @@ class NavigationCubit extends Cubit<NavigationState> {
|
||||
|
||||
void _onNewLocationFix(
|
||||
LatLng newLoc, double heading, double speedKmH, double alt) {
|
||||
double resolvedHeading = heading;
|
||||
|
||||
// Calculate heading from real movement if GPS course is missing / 0.0
|
||||
if (resolvedHeading <= 0.0 || resolvedHeading.isNaN) {
|
||||
if (_currentDisplayPosition != null) {
|
||||
final dist = locationService.calculateDistance(_currentDisplayPosition!, newLoc);
|
||||
if (dist >= 1.5) {
|
||||
final computed = locationService.calculateBearing(_currentDisplayPosition!, newLoc);
|
||||
resolvedHeading = (computed % 360.0 + 360.0) % 360.0;
|
||||
} else {
|
||||
resolvedHeading = _currentDisplayHeading;
|
||||
}
|
||||
} else {
|
||||
resolvedHeading = _currentDisplayHeading;
|
||||
}
|
||||
}
|
||||
|
||||
// In active navigation, if stopped or crawling, lock heading to route segment ahead
|
||||
if (state.isNavigating && speedKmH < 4.0 && state.currentRoute != null) {
|
||||
final coords = state.currentRoute!.coordinates;
|
||||
if (_lastTraveledIndexInFullRoute + 1 < coords.length) {
|
||||
resolvedHeading = _calculateBearing(
|
||||
coords[_lastTraveledIndexInFullRoute],
|
||||
coords[_lastTraveledIndexInFullRoute + 1],
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
emit(state.copyWith(
|
||||
myLocation: newLoc,
|
||||
altitude: alt,
|
||||
heading: heading,
|
||||
heading: resolvedHeading,
|
||||
speed: speedKmH,
|
||||
));
|
||||
|
||||
// Smooth movement interpolation for the vehicle marker and 3D camera
|
||||
if (_currentDisplayPosition == null) {
|
||||
_currentDisplayPosition = newLoc;
|
||||
_currentDisplayHeading = heading;
|
||||
_updateCarMarker(newLoc, heading);
|
||||
_currentDisplayHeading = resolvedHeading;
|
||||
_updateCarMarker(newLoc, resolvedHeading);
|
||||
} else {
|
||||
_startMovementInterpolation(newLoc, heading);
|
||||
_startMovementInterpolation(newLoc, resolvedHeading);
|
||||
}
|
||||
|
||||
// Proactively move camera on first acquired GPS lock (only when style is loaded)
|
||||
@@ -514,9 +550,10 @@ class NavigationCubit extends Cubit<NavigationState> {
|
||||
mapController != null &&
|
||||
_isMapStyleLoaded) {
|
||||
_hasInitiallyCenteredCamera = true;
|
||||
_currentCameraBearing = resolvedHeading;
|
||||
mapController!.animateCamera(
|
||||
CameraUpdate.newCameraPosition(
|
||||
CameraPosition(target: newLoc, zoom: 16.5, bearing: heading),
|
||||
CameraPosition(target: newLoc, zoom: 16.5, bearing: resolvedHeading),
|
||||
),
|
||||
);
|
||||
}
|
||||
@@ -527,11 +564,11 @@ class NavigationCubit extends Cubit<NavigationState> {
|
||||
latitude: newLoc.latitude,
|
||||
longitude: newLoc.longitude,
|
||||
speedKmH: speedKmH,
|
||||
heading: heading,
|
||||
heading: resolvedHeading,
|
||||
elevation: alt,
|
||||
remainingDistance: state.remainingDistance,
|
||||
);
|
||||
_processActiveNavigationTick(newLoc, speedKmH, heading);
|
||||
_processActiveNavigationTick(newLoc, speedKmH, resolvedHeading);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -624,6 +661,8 @@ class NavigationCubit extends Cubit<NavigationState> {
|
||||
}
|
||||
}
|
||||
|
||||
_currentCameraBearing = effectiveBearing;
|
||||
|
||||
// 3. Shift camera target ahead along heading vector
|
||||
// This positions the car in the bottom ~28% of the viewport with 45° tilt!
|
||||
final cameraTarget = _computeOffset(pos, lookAheadMeters, effectiveBearing);
|
||||
@@ -672,13 +711,21 @@ class NavigationCubit extends Cubit<NavigationState> {
|
||||
);
|
||||
}
|
||||
|
||||
Future<void> _updateCarMarker(LatLng position, double bearing) async {
|
||||
Future<void> _updateCarMarker(LatLng position, double worldHeading) async {
|
||||
if (mapController == null || !_isMapStyleLoaded) return;
|
||||
try {
|
||||
// MapLibre symbols have viewport-aligned rotation.
|
||||
// (worldHeading - _currentCameraBearing) keeps the vehicle oriented correctly:
|
||||
// In active navigation with locked camera, (worldHeading - _currentCameraBearing) = 0° (facing UP/forward along the road).
|
||||
// During turns, it tilts smoothly into the corner as the car turns.
|
||||
// In free map exploration, it points towards true geographic direction relative to rotated map.
|
||||
final double screenRotation =
|
||||
(worldHeading - _currentCameraBearing + 360.0) % 360.0;
|
||||
|
||||
await mapController!.setUserMarker(Marker(
|
||||
markerId: const MarkerId('current_user_car'),
|
||||
position: position,
|
||||
rotation: bearing,
|
||||
rotation: screenRotation,
|
||||
anchor: const Offset(0.5, 0.5),
|
||||
flat: true,
|
||||
icon: InlqBitmap.fromStyleImage('current_vehicle_icon',
|
||||
@@ -709,6 +756,7 @@ class NavigationCubit extends Cubit<NavigationState> {
|
||||
state.distanceToNextStep,
|
||||
);
|
||||
} else {
|
||||
_currentCameraBearing = state.heading;
|
||||
mapController!.animateCamera(
|
||||
CameraUpdate.newCameraPosition(
|
||||
CameraPosition(
|
||||
|
||||
@@ -1705,6 +1705,7 @@ class _MapViewState extends State<MapView> {
|
||||
zoomControlsEnabled: false,
|
||||
compassEnabled: false,
|
||||
onCameraMove: (pos) {
|
||||
cubit.updateCameraBearing(pos.bearing);
|
||||
if (state.isSelectingLocationOnMap) {
|
||||
cubit.updatePickedLocation(pos.target);
|
||||
}
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -47,6 +47,7 @@ class IntaleqMapController {
|
||||
/// [setUserMarker]. Reset to null on style reload (native symbol is
|
||||
/// destroyed) so the next update re-adds it.
|
||||
mgl.Symbol? _userSymbol;
|
||||
Marker? _pendingUserMarker;
|
||||
bool _userSymbolBusy = false;
|
||||
bool _isStyleLoaded = false;
|
||||
CameraUpdate? _pendingCameraUpdate;
|
||||
@@ -369,15 +370,21 @@ class IntaleqMapController {
|
||||
/// dropped, and the next tick supplies fresh coordinates). The puck is
|
||||
/// re-created automatically after a style reload.
|
||||
Future<void> setUserMarker(Marker marker) async {
|
||||
if (!_isStyleLoaded || _userSymbolBusy) return;
|
||||
if (!_isStyleLoaded) return;
|
||||
_pendingUserMarker = marker;
|
||||
if (_userSymbolBusy) return;
|
||||
_userSymbolBusy = true;
|
||||
try {
|
||||
await _loadBitmapIfNeeded(marker.icon);
|
||||
while (_pendingUserMarker != null) {
|
||||
final m = _pendingUserMarker!;
|
||||
_pendingUserMarker = null;
|
||||
await _loadBitmapIfNeeded(m.icon);
|
||||
final symbol = _userSymbol;
|
||||
if (symbol == null) {
|
||||
_userSymbol = await _raw.addSymbol(marker.toSymbolOptions());
|
||||
_userSymbol = await _raw.addSymbol(m.toSymbolOptions());
|
||||
} else {
|
||||
await _raw.updateSymbol(symbol, marker.toSymbolOptions());
|
||||
await _raw.updateSymbol(symbol, m.toSymbolOptions());
|
||||
}
|
||||
}
|
||||
} catch (_) {
|
||||
} finally {
|
||||
|
||||
@@ -0,0 +1,194 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Generate SQL migration file to ingest Jordan (Amman, Zarqa, Irbid) checkpoints CSV into places_jordan
|
||||
and refresh the unified_search_index.
|
||||
"""
|
||||
import csv
|
||||
import sys
|
||||
import os
|
||||
import math
|
||||
from collections import defaultdict
|
||||
|
||||
def haversine(lat1, lon1, lat2, lon2):
|
||||
R = 6371000
|
||||
phi1, phi2 = math.radians(lat1), math.radians(lat2)
|
||||
dphi = math.radians(lat2 - lat1)
|
||||
dlam = math.radians(lon2 - lon1)
|
||||
a = math.sin(dphi/2)**2 + math.cos(phi1)*math.cos(phi2)*math.sin(dlam/2)**2
|
||||
return 2 * R * math.atan2(math.sqrt(a), math.sqrt(1 - a))
|
||||
|
||||
def normalize_text(t):
|
||||
if not t: return ''
|
||||
s = t.strip()
|
||||
s = s.replace('أ', 'ا').replace('إ', 'ا').replace('آ', 'ا').replace('ة', 'ه').replace('ى', 'ي')
|
||||
return ' '.join(s.split()).lower()
|
||||
|
||||
def clean_str(val):
|
||||
if not val:
|
||||
return 'NULL'
|
||||
s = str(val).strip().replace("'", "''")
|
||||
return f"'{s}'"
|
||||
|
||||
def calculate_popularity(row):
|
||||
reviews = 0
|
||||
try:
|
||||
reviews = int(float(row.get('reviews_count') or 0))
|
||||
except Exception:
|
||||
reviews = 0
|
||||
|
||||
rating = 0.0
|
||||
try:
|
||||
rating = float(row.get('rating') or 0)
|
||||
except Exception:
|
||||
rating = 0.0
|
||||
|
||||
score = int(reviews * 0.2 + rating * 5)
|
||||
cat = row.get('category_queried', '')
|
||||
if any(k in cat for k in ['مستشفى', 'جامعة', 'كلية', 'مركز صحي', 'طوارئ', 'دفاع مدني']):
|
||||
score = max(score, 90)
|
||||
elif any(k in cat for k in ['صيدلية', 'بنك', 'صراف آلي', 'سوبر ماركت', 'محطة']):
|
||||
score = max(score, 60)
|
||||
elif any(k in cat for k in ['مدرسة', 'روضة', 'عيادة', 'بلدية', 'دائرة']):
|
||||
score = max(score, 50)
|
||||
|
||||
return min(100, max(10, score))
|
||||
|
||||
def main():
|
||||
csv_path = 'data/checkpoint_35242_records.csv'
|
||||
out_sql_path = 'infrastructure/sql/18_ingest_amman_zarqa_checkpoint.sql'
|
||||
|
||||
if not os.path.exists(csv_path):
|
||||
print(f"Error: {csv_path} not found.")
|
||||
sys.exit(1)
|
||||
|
||||
print(f"Reading {csv_path}...")
|
||||
records = []
|
||||
with open(csv_path, 'r', encoding='utf-8-sig') as f:
|
||||
reader = csv.DictReader(f)
|
||||
for row in reader:
|
||||
name = row.get('name', '').strip()
|
||||
if not name: continue
|
||||
try:
|
||||
lat = float(row['latitude'])
|
||||
lng = float(row['longitude'])
|
||||
except Exception:
|
||||
continue
|
||||
|
||||
if not (29.15 <= lat <= 33.45 and 34.85 <= lng <= 39.35):
|
||||
continue
|
||||
|
||||
category = row.get('category_queried', '').strip() or 'مكان عام'
|
||||
sector = row.get('sector', '').strip()
|
||||
gov = row.get('governorate', '').strip() or 'الأردن'
|
||||
area = row.get('area', '').strip()
|
||||
maps_url = row.get('maps_url', '').strip()
|
||||
|
||||
address = f"{area}, {gov}, الأردن" if area else f"{gov}, الأردن"
|
||||
desc = sector if sector else 'نقطة اهتمام في الأردن'
|
||||
pop_score = calculate_popularity(row)
|
||||
|
||||
records.append({
|
||||
'name': name,
|
||||
'norm_name': normalize_text(name),
|
||||
'lat': lat,
|
||||
'lng': lng,
|
||||
'category': category,
|
||||
'city': gov,
|
||||
'neighbourhood': area,
|
||||
'address': address,
|
||||
'description': desc,
|
||||
'popularity_score': pop_score,
|
||||
'maps_url': maps_url
|
||||
})
|
||||
|
||||
# Exact dedup
|
||||
seen_exact = {}
|
||||
seen_urls = {}
|
||||
for r in records:
|
||||
k = (r['norm_name'], round(r['lat'], 5), round(r['lng'], 5))
|
||||
u = r['maps_url']
|
||||
if k in seen_exact or (u and u in seen_urls):
|
||||
continue
|
||||
seen_exact[k] = r
|
||||
if u: seen_urls[u] = r
|
||||
|
||||
step1 = list(seen_exact.values())
|
||||
|
||||
# Spatial near-duplicate dedup (<50m)
|
||||
by_norm = defaultdict(list)
|
||||
for r in step1:
|
||||
by_norm[r['norm_name']].append(r)
|
||||
|
||||
final_records = []
|
||||
for norm_name, items in by_norm.items():
|
||||
if len(items) == 1:
|
||||
final_records.append(items[0])
|
||||
else:
|
||||
items.sort(key=lambda x: x['popularity_score'], reverse=True)
|
||||
kept = []
|
||||
for candidate in items:
|
||||
is_dup = False
|
||||
for existing in kept:
|
||||
if haversine(candidate['lat'], candidate['lng'], existing['lat'], existing['lng']) < 50:
|
||||
is_dup = True
|
||||
break
|
||||
if not is_dup:
|
||||
kept.append(candidate)
|
||||
final_records.extend(kept)
|
||||
|
||||
print(f"Total valid unique records for SQL ingestion: {len(final_records)}")
|
||||
|
||||
source_tag = 'checkpoint_35242_amman_zarqa'
|
||||
with open(out_sql_path, 'w', encoding='utf-8') as out:
|
||||
out.write("-- ============================================================================\n")
|
||||
out.write("-- MIGRATION 18: INGEST AMMAN & ZARQA CHECKPOINT DATASET (35,000+ PLACES)\n")
|
||||
out.write("-- Ingests into places_jordan and refreshes unified_search_index\n")
|
||||
out.write("-- ============================================================================\n\n")
|
||||
|
||||
out.write("BEGIN;\n\n")
|
||||
out.write(f"-- Remove previous imports with same source to ensure idempotency\n")
|
||||
out.write(f"DELETE FROM places_jordan WHERE source = '{source_tag}';\n\n")
|
||||
out.write("ALTER TABLE places_jordan DISABLE TRIGGER trg_sync_place_location_jordan;\n\n")
|
||||
|
||||
batch_size = 1000
|
||||
for i in range(0, len(final_records), batch_size):
|
||||
batch = final_records[i:i + batch_size]
|
||||
out.write("INSERT INTO places_jordan (\n")
|
||||
out.write(" name, name_ar, latitude, longitude, category, city, neighbourhood,\n")
|
||||
out.write(" address, description, popularity_score, source, location\n")
|
||||
out.write(") VALUES\n")
|
||||
|
||||
value_lines = []
|
||||
for r in batch:
|
||||
c_name = clean_str(r['name'])
|
||||
c_cat = clean_str(r['category'])
|
||||
c_city = clean_str(r['city'])
|
||||
c_area = clean_str(r['neighbourhood'])
|
||||
c_addr = clean_str(r['address'])
|
||||
c_desc = clean_str(r['description'])
|
||||
lat = r['lat']
|
||||
lng = r['lng']
|
||||
pop = r['popularity_score']
|
||||
|
||||
line = (f" ({c_name}, {c_name}, {lat:.7f}, {lng:.7f}, {c_cat}, {c_city}, {c_area}, "
|
||||
f"{c_addr}, {c_desc}, {pop}, '{source_tag}', "
|
||||
f"ST_SetSRID(ST_MakePoint({lng:.7f}, {lat:.7f}), 4326))")
|
||||
value_lines.append(line)
|
||||
|
||||
out.write(",\n".join(value_lines))
|
||||
out.write(";\n\n")
|
||||
|
||||
out.write("ALTER TABLE places_jordan ENABLE TRIGGER trg_sync_place_location_jordan;\n\n")
|
||||
out.write(f"UPDATE places_jordan SET governorate_id = 137 WHERE source = '{source_tag}' AND city = 'عمان';\n")
|
||||
out.write(f"UPDATE places_jordan SET governorate_id = 5 WHERE source = '{source_tag}' AND city = 'الزرقاء';\n")
|
||||
out.write(f"UPDATE places_jordan SET governorate_id = 184 WHERE source = '{source_tag}' AND city = 'إربد';\n\n")
|
||||
out.write("COMMIT;\n\n")
|
||||
|
||||
out.write("-- Refresh Unified Search Index to instantly index all new Jordanian places\n")
|
||||
out.write("REFRESH MATERIALIZED VIEW CONCURRENTLY unified_search_index;\n\n")
|
||||
out.write("ANALYZE places_jordan;\n")
|
||||
|
||||
print(f"✅ Generated SQL file: {out_sql_path} ({os.path.getsize(out_sql_path) / (1024*1024):.2f} MB)")
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
@@ -0,0 +1,358 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Migrate and deduplicate Jordan (Amman, Zarqa, Irbid) Places Dataset into places_jordan table on PostgreSQL.
|
||||
Handles Arabic normalization, spatial bounding box filtering (Jordan bounds),
|
||||
exact deduplication, spatial near-duplicate filtering (< 50m), popularity scoring,
|
||||
and concurrent refresh of unified_search_index.
|
||||
"""
|
||||
|
||||
import os
|
||||
import sys
|
||||
import csv
|
||||
import gzip
|
||||
import math
|
||||
import time
|
||||
import argparse
|
||||
from collections import defaultdict, Counter
|
||||
|
||||
def haversine(lat1, lon1, lat2, lon2):
|
||||
R = 6371000 # meters
|
||||
phi1 = math.radians(lat1)
|
||||
phi2 = math.radians(lat2)
|
||||
delta_phi = math.radians(lat2 - lat1)
|
||||
delta_lambda = math.radians(lon2 - lon1)
|
||||
a = math.sin(delta_phi / 2)**2 + math.cos(phi1) * math.cos(phi2) * math.sin(delta_lambda / 2)**2
|
||||
return 2 * R * math.atan2(math.sqrt(a), math.sqrt(1 - a))
|
||||
|
||||
def normalize_text(t):
|
||||
if not t:
|
||||
return ''
|
||||
s = t.strip()
|
||||
# Normalize Arabic alef, teh marbuta, etc.
|
||||
s = s.replace('أ', 'ا').replace('إ', 'ا').replace('آ', 'ا').replace('ة', 'ه').replace('ى', 'ي')
|
||||
return ' '.join(s.split()).lower()
|
||||
|
||||
def clean_sql_str(val):
|
||||
if val is None:
|
||||
return 'NULL'
|
||||
s = str(val).strip().replace("'", "''")
|
||||
return f"'{s}'"
|
||||
|
||||
def calculate_popularity(row):
|
||||
# Reviews and rating
|
||||
reviews = 0
|
||||
try:
|
||||
reviews = int(float(row.get('reviews_count') or 0))
|
||||
except Exception:
|
||||
reviews = 0
|
||||
|
||||
rating = 0.0
|
||||
try:
|
||||
rating = float(row.get('rating') or 0)
|
||||
except Exception:
|
||||
rating = 0.0
|
||||
|
||||
score = int(reviews * 0.2 + rating * 5)
|
||||
|
||||
# Heuristic boost for critical POIs
|
||||
cat = row.get('category_queried', '')
|
||||
if any(k in cat for k in ['مستشفى', 'جامعة', 'كلية', 'مركز صحي', 'طوارئ', 'دفاع مدني']):
|
||||
score = max(score, 90)
|
||||
elif any(k in cat for k in ['صيدلية', 'بنك', 'صراف آلي', 'سوبر ماركت', 'محطة']):
|
||||
score = max(score, 60)
|
||||
elif any(k in cat for k in ['مدرسة', 'روضة', 'عيادة', 'بلدية', 'دائرة']):
|
||||
score = max(score, 50)
|
||||
|
||||
return min(100, max(10, score))
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description="Migrate Jordan places dataset with deduplication.")
|
||||
parser.add_argument("--file", default="", help="Path to CSV file.")
|
||||
parser.add_argument("--db-host", default="127.0.0.1", help="Database host")
|
||||
parser.add_argument("--db-port", type=int, default=5432, help="Database port")
|
||||
parser.add_argument("--db-user", default="mapuser", help="Database user")
|
||||
parser.add_argument("--db-pass", default="mappass", help="Database password")
|
||||
parser.add_argument("--db-name", default="mapdb", help="Database name")
|
||||
parser.add_argument("--dry-run", action="store_true", help="Perform validation and deduplication only without inserting.")
|
||||
args = parser.parse_args()
|
||||
|
||||
# Find file
|
||||
file_path = args.file
|
||||
if not file_path:
|
||||
candidates = [
|
||||
"data/checkpoint_35242_records.csv",
|
||||
"/home/hamzadoctor/app/data/checkpoint_35242_records.csv",
|
||||
"checkpoint_35242_records.csv"
|
||||
]
|
||||
for c in candidates:
|
||||
if os.path.exists(c):
|
||||
file_path = c
|
||||
break
|
||||
|
||||
if not file_path or not os.path.exists(file_path):
|
||||
print(f"❌ Error: Dataset file not found: {file_path}")
|
||||
sys.exit(1)
|
||||
|
||||
print(f"📂 Processing dataset: {file_path}")
|
||||
open_fn = gzip.open if file_path.endswith('.gz') else open
|
||||
mode = 'rt' if file_path.endswith('.gz') else 'r'
|
||||
|
||||
total_read = 0
|
||||
empty_name = 0
|
||||
out_of_bounds = 0
|
||||
valid_records = []
|
||||
gov_counter = Counter()
|
||||
|
||||
t0 = time.time()
|
||||
with open_fn(file_path, mode, encoding='utf-8-sig') as f:
|
||||
reader = csv.DictReader(f)
|
||||
for row in reader:
|
||||
total_read += 1
|
||||
name = row.get('name', '').strip()
|
||||
if not name:
|
||||
empty_name += 1
|
||||
continue
|
||||
|
||||
try:
|
||||
lat = float(row['latitude'])
|
||||
lng = float(row['longitude'])
|
||||
except Exception:
|
||||
out_of_bounds += 1
|
||||
continue
|
||||
|
||||
# Jordan bounding box check (29.15 to 33.45 Lat, 34.85 to 39.35 Lng)
|
||||
if not (29.15 <= lat <= 33.45 and 34.85 <= lng <= 39.35):
|
||||
out_of_bounds += 1
|
||||
continue
|
||||
|
||||
category = row.get('category_queried', '').strip() or 'مكان عام'
|
||||
sector = row.get('sector', '').strip()
|
||||
gov = row.get('governorate', '').strip() or 'الأردن'
|
||||
area = row.get('area', '').strip()
|
||||
maps_url = row.get('maps_url', '').strip()
|
||||
|
||||
address = f"{area}, {gov}, الأردن" if area else f"{gov}, الأردن"
|
||||
desc = sector if sector else 'نقطة اهتمام في الأردن'
|
||||
pop_score = calculate_popularity(row)
|
||||
|
||||
gov_counter[gov] += 1
|
||||
|
||||
valid_records.append({
|
||||
'name': name,
|
||||
'norm_name': normalize_text(name),
|
||||
'lat': lat,
|
||||
'lng': lng,
|
||||
'category': category,
|
||||
'city': gov,
|
||||
'neighbourhood': area,
|
||||
'address': address,
|
||||
'description': desc,
|
||||
'popularity_score': pop_score,
|
||||
'maps_url': maps_url,
|
||||
'sector': sector
|
||||
})
|
||||
|
||||
print(f"📊 Initial parse complete in {time.time() - t0:.2f}s:")
|
||||
print(f" - Total rows in CSV: {total_read:,}")
|
||||
print(f" - Out of Jordan bounds / invalid coords: {out_of_bounds:,}")
|
||||
print(f" - Valid in Jordan bounds: {len(valid_records):,}")
|
||||
print(f" - Governorate distribution: {dict(gov_counter)}")
|
||||
|
||||
# Deduplication Step 1: Exact (norm_name, round(lat, 5), round(lng, 5)) and maps_url
|
||||
seen_exact = {}
|
||||
seen_urls = {}
|
||||
exact_dups = 0
|
||||
|
||||
for r in valid_records:
|
||||
k = (r['norm_name'], round(r['lat'], 5), round(r['lng'], 5))
|
||||
url = r['maps_url']
|
||||
|
||||
if k in seen_exact:
|
||||
exact_dups += 1
|
||||
if r['popularity_score'] > seen_exact[k]['popularity_score']:
|
||||
seen_exact[k] = r
|
||||
continue
|
||||
|
||||
if url and url in seen_urls:
|
||||
exact_dups += 1
|
||||
if r['popularity_score'] > seen_urls[url]['popularity_score']:
|
||||
seen_urls[url] = r
|
||||
continue
|
||||
|
||||
seen_exact[k] = r
|
||||
if url:
|
||||
seen_urls[url] = r
|
||||
|
||||
dedup_step1 = list(seen_exact.values())
|
||||
print(f"🔍 Step 1 Deduplication (Exact coords / URL):")
|
||||
print(f" - Removed {exact_dups} duplicate records.")
|
||||
print(f" - Remaining: {len(dedup_step1):,}")
|
||||
|
||||
# Deduplication Step 2: Spatial near-duplicates (< 50m with identical normalized name)
|
||||
by_norm_name = defaultdict(list)
|
||||
for r in dedup_step1:
|
||||
by_norm_name[r['norm_name']].append(r)
|
||||
|
||||
final_records = []
|
||||
near_dups_filtered = 0
|
||||
|
||||
for norm_name, items in by_norm_name.items():
|
||||
if len(items) == 1:
|
||||
final_records.append(items[0])
|
||||
else:
|
||||
items.sort(key=lambda x: x['popularity_score'], reverse=True)
|
||||
kept = []
|
||||
for candidate in items:
|
||||
is_dup = False
|
||||
for existing in kept:
|
||||
d = haversine(candidate['lat'], candidate['lng'], existing['lat'], existing['lng'])
|
||||
if d < 50:
|
||||
is_dup = True
|
||||
near_dups_filtered += 1
|
||||
break
|
||||
if not is_dup:
|
||||
kept.append(candidate)
|
||||
final_records.extend(kept)
|
||||
|
||||
print(f"🎯 Step 2 Deduplication (Spatial near-duplicates < 50m):")
|
||||
print(f" - Filtered out {near_dups_filtered} near-duplicates.")
|
||||
print(f" - Final unique clean places to migrate: {len(final_records):,}")
|
||||
|
||||
final_govs = Counter([r['city'] for r in final_records])
|
||||
print(f" - Final Governorates: {dict(final_govs)}")
|
||||
|
||||
if args.dry_run:
|
||||
print("💡 Dry run complete. No database changes made.")
|
||||
return
|
||||
|
||||
# Database Migration
|
||||
import pg8000.native
|
||||
print(f"\n🔌 Connecting to PostgreSQL at {args.db_host}:{args.db_port} ({args.db_name})...")
|
||||
con = pg8000.native.Connection(
|
||||
user=args.db_user,
|
||||
password=args.db_pass,
|
||||
host=args.db_host,
|
||||
port=args.db_port,
|
||||
database=args.db_name
|
||||
)
|
||||
|
||||
t_db = time.time()
|
||||
source_tag = 'checkpoint_35242_amman_zarqa'
|
||||
|
||||
print(f"🗑️ Removing previous '{source_tag}' imports from places_jordan...")
|
||||
con.run("BEGIN;")
|
||||
con.run(f"DELETE FROM places_jordan WHERE source = '{source_tag}';")
|
||||
|
||||
# Temporarily disable trigger for high-speed bulk ingestion
|
||||
print("⚡ Disabling location trigger for high-speed ingestion...")
|
||||
con.run("ALTER TABLE places_jordan DISABLE TRIGGER trg_sync_place_location_jordan;")
|
||||
|
||||
batch_size = 2000
|
||||
total_inserted = 0
|
||||
print(f"🚀 Inserting {len(final_records):,} places in batches of {batch_size}...")
|
||||
|
||||
for i in range(0, len(final_records), batch_size):
|
||||
batch = final_records[i:i + batch_size]
|
||||
values = []
|
||||
for r in batch:
|
||||
c_name = clean_sql_str(r['name'])
|
||||
c_cat = clean_sql_str(r['category'])
|
||||
c_city = clean_sql_str(r['city'])
|
||||
c_area = clean_sql_str(r['neighbourhood'])
|
||||
c_addr = clean_sql_str(r['address'])
|
||||
c_desc = clean_sql_str(r['description'])
|
||||
lat = r['lat']
|
||||
lng = r['lng']
|
||||
pop = r['popularity_score']
|
||||
|
||||
line = (
|
||||
f"({c_name}, {c_name}, {lat:.7f}, {lng:.7f}, {c_cat}, {c_city}, {c_area}, "
|
||||
f"{c_addr}, {c_desc}, {pop}, '{source_tag}', "
|
||||
f"ST_SetSRID(ST_MakePoint({lng:.7f}, {lat:.7f}), 4326))"
|
||||
)
|
||||
values.append(line)
|
||||
|
||||
sql = (
|
||||
"INSERT INTO places_jordan ("
|
||||
" name, name_ar, latitude, longitude, category, city, neighbourhood,"
|
||||
" address, description, popularity_score, source, location"
|
||||
") VALUES " + ",\n".join(values) + ";"
|
||||
)
|
||||
con.run(sql)
|
||||
total_inserted += len(batch)
|
||||
print(f" -> Inserted {total_inserted:,} / {len(final_records):,} places...")
|
||||
|
||||
# Re-enable trigger
|
||||
print("⚡ Re-enabling location trigger...")
|
||||
con.run("ALTER TABLE places_jordan ENABLE TRIGGER trg_sync_place_location_jordan;")
|
||||
|
||||
# Batch update governorate_id for Amman and Zarqa based on city name
|
||||
print("🏛️ Updating administrative links for new places...")
|
||||
con.run(f"""
|
||||
UPDATE places_jordan
|
||||
SET governorate_id = 137
|
||||
WHERE source = '{source_tag}' AND city = 'عمان';
|
||||
""")
|
||||
con.run(f"""
|
||||
UPDATE places_jordan
|
||||
SET governorate_id = 5
|
||||
WHERE source = '{source_tag}' AND city = 'الزرقاء';
|
||||
""")
|
||||
con.run(f"""
|
||||
UPDATE places_jordan
|
||||
SET governorate_id = 184
|
||||
WHERE source = '{source_tag}' AND city = 'إربد';
|
||||
""")
|
||||
|
||||
con.run("COMMIT;")
|
||||
print(f"✅ Ingestion committed successfully in {time.time() - t_db:.2f}s!")
|
||||
|
||||
# Refresh materialized view
|
||||
print("🔄 Refreshing materialized view unified_search_index concurrently...")
|
||||
t_mv = time.time()
|
||||
con.run("REFRESH MATERIALIZED VIEW CONCURRENTLY unified_search_index;")
|
||||
print(f"✅ unified_search_index refreshed in {time.time() - t_mv:.2f}s!")
|
||||
|
||||
# Analyze table
|
||||
print("⚡ Running ANALYZE on places_jordan...")
|
||||
con.run("ANALYZE places_jordan;")
|
||||
|
||||
# Verification
|
||||
print("\n🔍 Verification & Statistics:")
|
||||
count_res = con.run("SELECT count(*) FROM places_jordan;")[0][0]
|
||||
new_count = con.run(f"SELECT count(*) FROM places_jordan WHERE source = '{source_tag}';")[0][0]
|
||||
gov_stats = con.run(f"""
|
||||
SELECT city, count(*)
|
||||
FROM places_jordan
|
||||
WHERE source = '{source_tag}'
|
||||
GROUP BY city
|
||||
ORDER BY count(*) DESC;
|
||||
""")
|
||||
print(f" - Total places in places_jordan: {count_res:,}")
|
||||
print(f" - Newly added from Amman & Zarqa checkpoint: {new_count:,}")
|
||||
print(" - Governorates breakdown:")
|
||||
for gov, cnt in gov_stats:
|
||||
print(f" * {gov}: {cnt:,} places")
|
||||
|
||||
# Sample Geocoding search test
|
||||
test_queries = ['صيدلية في عمان', 'الجبيهة', 'الزرقاء الجديدة', 'وسط البلد عمان', 'الشميساني']
|
||||
print("\n🧪 Testing search on unified_search_index:")
|
||||
for q in test_queries:
|
||||
results = con.run(f"""
|
||||
SELECT name, category, city, latitude, longitude
|
||||
FROM places_jordan
|
||||
WHERE (name ILIKE '%{q}%' OR neighbourhood ILIKE '%{q}%' OR address ILIKE '%{q}%')
|
||||
AND source = '{source_tag}'
|
||||
LIMIT 2;
|
||||
""")
|
||||
if results:
|
||||
first = results[0]
|
||||
print(f" ✓ Query '{q}': Found '{first[0]}' ({first[1]} - {first[2]}) @ {first[3]}, {first[4]}")
|
||||
else:
|
||||
print(f" - Query '{q}': No direct checkpoint match, checking overall...")
|
||||
|
||||
con.close()
|
||||
print("\n🎉 Jordan dataset migration completed successfully!")
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
Reference in New Issue
Block a user