Update: 2026-07-06 17:00:43
This commit is contained in:
@@ -0,0 +1,84 @@
|
||||
import { RideSample, PricingTier } from './types';
|
||||
import { kMeans } from '../utils/math';
|
||||
|
||||
const TIER_LABELS: Array<'economy' | 'standard' | 'premium'> = [
|
||||
'economy',
|
||||
'standard',
|
||||
'premium',
|
||||
];
|
||||
|
||||
/**
|
||||
* Cluster rides into pricing tiers based on price_per_km using K-Means.
|
||||
* Returns sorted tiers (economy < standard < premium).
|
||||
*/
|
||||
export function clusterTiers(
|
||||
samples: RideSample[],
|
||||
k: number = 3
|
||||
): PricingTier[] {
|
||||
if (samples.length < k) {
|
||||
return [{
|
||||
label: 'standard',
|
||||
samples,
|
||||
ppkRange: [0, Infinity],
|
||||
regression: null,
|
||||
}];
|
||||
}
|
||||
|
||||
const ppkValues = samples.map(s => s.ppk);
|
||||
const assignments = kMeans(ppkValues, k);
|
||||
|
||||
// Calculate centroids for sorting
|
||||
const centroids = new Array(k).fill(0).map((_, c) => {
|
||||
const cluster = samples.filter((_, i) => assignments[i] === c);
|
||||
return cluster.length > 0
|
||||
? cluster.reduce((sum, s) => sum + s.ppk, 0) / cluster.length
|
||||
: 0;
|
||||
});
|
||||
|
||||
// Sort clusters by centroid (ascending)
|
||||
const sortedClusterIndices = centroids
|
||||
.map((c, i) => ({ centroid: c, index: i }))
|
||||
.filter(c => !isNaN(c.centroid) && c.centroid > 0)
|
||||
.sort((a, b) => a.centroid - b.centroid);
|
||||
|
||||
const tiers: PricingTier[] = sortedClusterIndices.map((cluster, idx) => {
|
||||
const clusterSamples = samples.filter((_, i) => assignments[i] === cluster.index);
|
||||
const clusterPPKs = clusterSamples.map(s => s.ppk);
|
||||
|
||||
return {
|
||||
label: TIER_LABELS[idx] || 'unknown',
|
||||
samples: clusterSamples,
|
||||
ppkRange: [
|
||||
Math.min(...clusterPPKs),
|
||||
Math.max(...clusterPPKs),
|
||||
],
|
||||
regression: null,
|
||||
};
|
||||
});
|
||||
|
||||
return tiers;
|
||||
}
|
||||
|
||||
/**
|
||||
* Assign zones to routes based on coordinate grid.
|
||||
* Grid size ~2.5km (0.025 degrees).
|
||||
*/
|
||||
export function assignZone(lat: number, lng: number): string {
|
||||
const gridLat = Math.round(lat / 0.025) * 0.025;
|
||||
const gridLng = Math.round(lng / 0.025) * 0.025;
|
||||
return `${gridLat.toFixed(3)},${gridLng.toFixed(3)}`;
|
||||
}
|
||||
|
||||
/**
|
||||
* Classify zone type based on distance from city center (Amman: 31.95, 35.90).
|
||||
*/
|
||||
export function classifyZoneType(lat: number, lng: number): string {
|
||||
const dlat = lat - 31.95;
|
||||
const dlng = lng - 35.90;
|
||||
const dist = Math.sqrt(dlat * dlat + dlng * dlng);
|
||||
|
||||
if (dist < 0.025) return 'centre';
|
||||
if (dist < 0.050) return 'mid';
|
||||
if (dist < 0.100) return 'suburb';
|
||||
return 'outskirts';
|
||||
}
|
||||
@@ -0,0 +1,136 @@
|
||||
import { RideSample, AnalysisReport } from './types';
|
||||
import { removeOutliers, groupByRoute, extractBasePrices } from './outliers';
|
||||
import { clusterTiers } from './clustering';
|
||||
import { analyzeAllTiers } from './regression';
|
||||
import { detectSurge, aggregateSurgeHours } from './surge';
|
||||
import { analyzeByZone, analyzeByZoneType } from './zone';
|
||||
|
||||
export interface EngineOptions {
|
||||
competitorName?: string;
|
||||
countryCode?: string;
|
||||
cleanOutliers?: boolean;
|
||||
surgeThreshold?: number;
|
||||
tierCount?: number;
|
||||
}
|
||||
|
||||
/**
|
||||
* Main pricing analysis engine.
|
||||
* Orchestrates the full pipeline: fetch → clean → cluster → regress → surge → zone.
|
||||
*/
|
||||
export async function runAnalysis(
|
||||
samples: RideSample[],
|
||||
options: EngineOptions = {}
|
||||
): Promise<AnalysisReport> {
|
||||
const {
|
||||
cleanOutliers = true,
|
||||
surgeThreshold = 0.12,
|
||||
tierCount = 3,
|
||||
} = options;
|
||||
|
||||
if (samples.length < 5) {
|
||||
throw new Error(`Insufficient samples (${samples.length}). Need at least 5.`);
|
||||
}
|
||||
|
||||
const firstSample = samples[0];
|
||||
|
||||
// Step 1: Remove statistical outliers (MAD on PPK)
|
||||
const cleanSamples = cleanOutliers ? removeOutliers(samples) : samples;
|
||||
|
||||
// Step 2: Group by route and extract base (non-surge) prices
|
||||
const routeGroups = groupByRoute(cleanSamples);
|
||||
const baseSamples = extractBasePrices(routeGroups, surgeThreshold);
|
||||
|
||||
// Step 3: Cluster into pricing tiers by PPK
|
||||
const rawTiers = clusterTiers(cleanSamples, tierCount);
|
||||
|
||||
// Step 4: Run regression on each tier
|
||||
const analyzedTiers = analyzeAllTiers(rawTiers);
|
||||
|
||||
// Step 5: Detect surge patterns
|
||||
const surgePatterns = detectSurge(cleanSamples, surgeThreshold);
|
||||
const surgeHours = aggregateSurgeHours(surgePatterns);
|
||||
|
||||
// Step 6: Zone analysis
|
||||
const zones = analyzeByZone(cleanSamples);
|
||||
const zoneTypes = analyzeByZoneType(cleanSamples);
|
||||
|
||||
// Build report
|
||||
const report: AnalysisReport = {
|
||||
competitorName: firstSample.competitorName,
|
||||
countryCode: firstSample.countryCode,
|
||||
tiers: analyzedTiers,
|
||||
surgePatterns,
|
||||
zones,
|
||||
totalSamples: samples.length,
|
||||
analyzedAt: new Date().toISOString(),
|
||||
};
|
||||
|
||||
// Print summary
|
||||
printSummary(report, baseSamples, surgeHours, zoneTypes);
|
||||
|
||||
return report;
|
||||
}
|
||||
|
||||
function printSummary(
|
||||
report: AnalysisReport,
|
||||
baseSamples: RideSample[],
|
||||
surgeHours: ReturnType<typeof aggregateSurgeHours>,
|
||||
zoneTypes: ReturnType<typeof analyzeByZoneType>
|
||||
): void {
|
||||
const sep = '═══════════════════════════════════════════════════════';
|
||||
console.log(`\n${sep}`);
|
||||
console.log(` 📊 Pricing Analysis Report — ${report.competitorName} (${report.countryCode})`);
|
||||
console.log(` ${report.totalSamples} total samples, ${baseSamples.length} base-price samples`);
|
||||
console.log(` Analyzed at: ${report.analyzedAt}`);
|
||||
console.log(sep);
|
||||
|
||||
// Tiers
|
||||
console.log(`\n📦 PRICING TIERS:`);
|
||||
for (const tier of report.tiers) {
|
||||
const reg = tier.regression;
|
||||
if (reg) {
|
||||
const tierIcon = tier.label === 'economy' ? '💰' : tier.label === 'standard' ? '🚗' : '💎';
|
||||
console.log(` ${tierIcon} ${tier.label.toUpperCase()}:`);
|
||||
console.log(` Base Fare: ${reg.baseFare.toFixed(3)} ${report.countryCode === 'JO' ? 'JOD' : report.countryCode === 'SY' ? 'SYP' : 'CUR'}`);
|
||||
console.log(` Per KM: ${reg.kmRate.toFixed(3)}`);
|
||||
console.log(` Per Min: ${reg.minRate.toFixed(3)}`);
|
||||
console.log(` Min Fare: ${reg.minFare.toFixed(3)} ${reg.hasMinFare ? '✅ active' : ''}`);
|
||||
console.log(` RMSE: ${reg.rmse.toFixed(4)}`);
|
||||
console.log(` R²: ${reg.rSquared.toFixed(4)}`);
|
||||
console.log(` Samples: ${reg.sampleCount}`);
|
||||
console.log(` PPK range: ${tier.ppkRange[0].toFixed(3)} – ${tier.ppkRange[1].toFixed(3)}`);
|
||||
} else {
|
||||
console.log(` 📄 ${tier.label.toUpperCase()}: ${tier.samples.length} samples (insufficient for regression)`);
|
||||
}
|
||||
console.log('');
|
||||
}
|
||||
|
||||
// Surge
|
||||
if (surgeHours.length > 0) {
|
||||
console.log(`⚡ SURGE PATTERNS (by hour-of-day):`);
|
||||
for (const sh of surgeHours) {
|
||||
console.log(` Hour ${sh.hour.toString().padStart(2, '0')}:00 → avg ${sh.avgMultiplier.toFixed(3)}x (${sh.routeCount} routes)`);
|
||||
}
|
||||
} else {
|
||||
console.log(`\nℹ️ No significant surge patterns detected.`);
|
||||
}
|
||||
|
||||
// Zones
|
||||
if (zoneTypes.length > 0) {
|
||||
console.log(`\n📍 ZONE TYPE ANALYSIS:`);
|
||||
for (const zt of zoneTypes) {
|
||||
console.log(` ${zt.zoneType.padEnd(12)} → avg ${zt.avgPpk.toFixed(3)}/km (${zt.sampleCount} rides)`);
|
||||
}
|
||||
}
|
||||
|
||||
// Surge route details
|
||||
if (report.surgePatterns.length > 0) {
|
||||
console.log(`\n🔍 TOP SURGE ROUTES:`);
|
||||
for (const sr of report.surgePatterns.slice(0, 5)) {
|
||||
console.log(` ${sr.distanceKm.toFixed(1)}km → base ${sr.basePrice.toFixed(2)}, peak ${(sr.basePrice * sr.maxMultiplier).toFixed(2)} JOD (${sr.maxMultiplier.toFixed(3)}x)`);
|
||||
}
|
||||
}
|
||||
|
||||
console.log(sep);
|
||||
console.log('');
|
||||
}
|
||||
@@ -0,0 +1,72 @@
|
||||
import { RideSample } from './types';
|
||||
import { findInliersMAD } from '../utils/math';
|
||||
|
||||
/**
|
||||
* Remove outlier rides using MAD on price_per_km.
|
||||
* Also removes rides where price is clearly a surge outlier
|
||||
* by comparing same-route prices.
|
||||
*/
|
||||
export function removeOutliers(
|
||||
samples: RideSample[],
|
||||
ppkThreshold: number = 3.5
|
||||
): RideSample[] {
|
||||
if (samples.length < 10) return samples;
|
||||
|
||||
const ppkValues = samples.map(s => s.ppk);
|
||||
const inlierIndices = new Set(findInliersMAD(ppkValues, ppkThreshold));
|
||||
|
||||
// Also remove rides with price_per_km > 3x the median
|
||||
const sortedPPK = [...ppkValues].sort((a, b) => a - b);
|
||||
const medianPPK = sortedPPK[Math.floor(sortedPPK.length / 2)];
|
||||
const upperBound = medianPPK * 3;
|
||||
|
||||
return samples.filter((s, i) =>
|
||||
inlierIndices.has(i) && s.ppk <= upperBound && s.ppk > 0
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* Group samples by unique route (start/end coordinates rounded to 4 decimals).
|
||||
*/
|
||||
export function groupByRoute(samples: RideSample[]): Map<string, RideSample[]> {
|
||||
const groups = new Map<string, RideSample[]>();
|
||||
for (const s of samples) {
|
||||
const key = `${s.startLat.toFixed(4)},${s.startLng.toFixed(4)}->${s.endLat.toFixed(4)},${s.endLng.toFixed(4)}`;
|
||||
if (!groups.has(key)) groups.set(key, []);
|
||||
groups.get(key)!.push(s);
|
||||
}
|
||||
return groups;
|
||||
}
|
||||
|
||||
/**
|
||||
* For each route, keep only the lowest price (non-surge baseline)
|
||||
* if the price variation exceeds threshold.
|
||||
*/
|
||||
export function extractBasePrices(
|
||||
groups: Map<string, RideSample[]>,
|
||||
surgeThreshold: number = 0.15
|
||||
): RideSample[] {
|
||||
const base: RideSample[] = [];
|
||||
|
||||
for (const [, rides] of groups) {
|
||||
if (rides.length === 1) {
|
||||
base.push(rides[0]);
|
||||
continue;
|
||||
}
|
||||
|
||||
const prices = rides.map(r => r.price);
|
||||
const minPrice = Math.min(...prices);
|
||||
const maxPrice = Math.max(...prices);
|
||||
|
||||
// If variation is small, use all rides
|
||||
if (maxPrice - minPrice <= surgeThreshold) {
|
||||
base.push(...rides);
|
||||
} else {
|
||||
// Only keep rides within 5% of minimum price
|
||||
const baseRides = rides.filter(r => r.price <= minPrice * 1.05);
|
||||
base.push(...baseRides);
|
||||
}
|
||||
}
|
||||
|
||||
return base;
|
||||
}
|
||||
@@ -0,0 +1,120 @@
|
||||
import { PricingTier, RegressionResult, RideSample } from './types';
|
||||
import {
|
||||
robustMultipleLinearRegression,
|
||||
calcRMSE,
|
||||
calcRSquared,
|
||||
detectMinimumFare,
|
||||
} from '../utils/math';
|
||||
import { mean } from 'simple-statistics';
|
||||
|
||||
/**
|
||||
* Run multiple linear regression on each pricing tier.
|
||||
* Detects minimum fare and computes RMSE/R².
|
||||
*/
|
||||
export function analyzeTier(tier: PricingTier): PricingTier {
|
||||
const samples = tier.samples;
|
||||
if (samples.length < 5) {
|
||||
tier.regression = null;
|
||||
return tier;
|
||||
}
|
||||
|
||||
// Primary model: price = baseFare + kmRate * dist + minRate * dur
|
||||
// We use robust regression to strip out surge outliers and find the floor price
|
||||
const mlrResult = robustMultipleLinearRegression(
|
||||
samples.map(s => ({
|
||||
distance_km: s.distance_km,
|
||||
duration_min: s.duration_min,
|
||||
price: s.price,
|
||||
}))
|
||||
);
|
||||
|
||||
if (!mlrResult) {
|
||||
tier.regression = null;
|
||||
return tier;
|
||||
}
|
||||
|
||||
// Predict and compute RMSE/R²
|
||||
const actualPrices = samples.map(s => s.price);
|
||||
const predictedPrices = samples.map(s =>
|
||||
mlrResult.baseFare +
|
||||
mlrResult.kmRate * s.distance_km +
|
||||
mlrResult.minRate * s.duration_min
|
||||
);
|
||||
|
||||
const rmse = calcRMSE(actualPrices, predictedPrices);
|
||||
const rSquared = calcRSquared(actualPrices, predictedPrices);
|
||||
|
||||
// Detect minimum fare
|
||||
const minFare = detectMinimumFare(
|
||||
samples.map(s => s.distance_km),
|
||||
samples.map(s => s.price),
|
||||
mlrResult.kmRate
|
||||
);
|
||||
|
||||
// If minFare is detected and the short-ride residuals improve,
|
||||
// apply minFare-adjusted model
|
||||
let hasMinFare = false;
|
||||
let adjustedRMSE = rmse;
|
||||
let adjustedRSquared = rSquared;
|
||||
|
||||
if (minFare && minFare > 0) {
|
||||
const adjustedPredicted = samples.map(s => {
|
||||
const raw = mlrResult.baseFare + mlrResult.kmRate * s.distance_km + mlrResult.minRate * s.duration_min;
|
||||
return Math.max(raw, minFare);
|
||||
});
|
||||
const adjRmse = calcRMSE(actualPrices, adjustedPredicted);
|
||||
const adjRsq = calcRSquared(actualPrices, adjustedPredicted);
|
||||
|
||||
// If minimum fare improves the fit, use it
|
||||
if (adjRmse < rmse) {
|
||||
hasMinFare = true;
|
||||
adjustedRMSE = adjRmse;
|
||||
adjustedRSquared = adjRsq;
|
||||
}
|
||||
}
|
||||
|
||||
tier.regression = {
|
||||
baseFare: mlrResult.baseFare,
|
||||
kmRate: mlrResult.kmRate,
|
||||
minRate: mlrResult.minRate,
|
||||
minFare: minFare || 0,
|
||||
rmse: adjustedRMSE,
|
||||
rSquared: adjustedRSquared,
|
||||
sampleCount: samples.length,
|
||||
hasMinFare,
|
||||
};
|
||||
|
||||
return tier;
|
||||
}
|
||||
|
||||
/**
|
||||
* Run regression on all tiers.
|
||||
*/
|
||||
export function analyzeAllTiers(tiers: PricingTier[]): PricingTier[] {
|
||||
return tiers.map(tier => analyzeTier(tier));
|
||||
}
|
||||
|
||||
/**
|
||||
* Simple distance-only regression for comparison.
|
||||
* price = kmRate * dist
|
||||
*/
|
||||
export function distanceOnlyRegression(
|
||||
samples: RideSample[]
|
||||
): { kmRate: number; rmse: number } | null {
|
||||
if (samples.length < 3) return null;
|
||||
|
||||
const distances = samples.map(s => s.distance_km);
|
||||
const prices = samples.map(s => s.price);
|
||||
|
||||
// Simple average of price/km
|
||||
const ratios = distances.map((d, i) => d > 0 ? prices[i] / d : 0)
|
||||
.filter(r => r > 0 && isFinite(r));
|
||||
|
||||
if (ratios.length < 3) return null;
|
||||
|
||||
const kmRate = mean(ratios);
|
||||
const predicted = distances.map(d => kmRate * d);
|
||||
const rmse = calcRMSE(prices, predicted);
|
||||
|
||||
return { kmRate, rmse };
|
||||
}
|
||||
@@ -0,0 +1,96 @@
|
||||
import { RideSample, SurgeResult } from './types';
|
||||
import { groupByRoute } from './outliers';
|
||||
|
||||
/**
|
||||
* Detect surge pricing by analyzing price variation per route across time.
|
||||
* For routes with multiple samples, identifies base price (minimum)
|
||||
* and surge multipliers per hour-of-day (aggregated across all days).
|
||||
*/
|
||||
export function detectSurge(
|
||||
samples: RideSample[],
|
||||
surgeThreshold: number = 0.12
|
||||
): SurgeResult[] {
|
||||
const routes = groupByRoute(samples);
|
||||
const results: SurgeResult[] = [];
|
||||
|
||||
for (const [routeKey, rides] of routes) {
|
||||
if (rides.length < 3) continue;
|
||||
|
||||
const prices = rides.map(r => r.price);
|
||||
const minPrice = Math.min(...prices);
|
||||
const maxPrice = Math.max(...prices);
|
||||
|
||||
// Only analyze routes with meaningful variation
|
||||
if (maxPrice - minPrice <= surgeThreshold) continue;
|
||||
|
||||
// Find the time of the base price
|
||||
const baseRide = rides.find(r => r.price === minPrice);
|
||||
|
||||
// Aggregate surge by hour-of-day across ALL days
|
||||
const surgeByHour = new Map<number, number[]>();
|
||||
for (const r of rides) {
|
||||
const hour = r.scrapedAt.getHours();
|
||||
if (!surgeByHour.has(hour)) surgeByHour.set(hour, []);
|
||||
surgeByHour.get(hour)!.push(r.price);
|
||||
}
|
||||
|
||||
const surgePrices: SurgeResult['surgePrices'] = [];
|
||||
let maxMultiplier = 1;
|
||||
|
||||
// Sort hours and compute average multiplier per hour
|
||||
for (const [hour, hourPrices] of [...surgeByHour.entries()].sort((a, b) => a[0] - b[0])) {
|
||||
const avgTimePrice = hourPrices.reduce((a, b) => a + b, 0) / hourPrices.length;
|
||||
const multiplier = minPrice > 0 ? avgTimePrice / minPrice : 1;
|
||||
if (multiplier > maxMultiplier) maxMultiplier = multiplier;
|
||||
|
||||
surgePrices.push({
|
||||
time: `${hour.toString().padStart(2, '0')}:00`,
|
||||
price: Math.round(avgTimePrice * 100) / 100,
|
||||
multiplier: Math.round(multiplier * 1000) / 1000,
|
||||
});
|
||||
}
|
||||
|
||||
if (maxMultiplier > 1.05) {
|
||||
results.push({
|
||||
routeKey,
|
||||
distanceKm: rides[0].distance_km,
|
||||
basePrice: minPrice,
|
||||
baseTime: baseRide ? baseRide.scrapedAt.toISOString() : '',
|
||||
surgePrices,
|
||||
maxMultiplier: Math.round(maxMultiplier * 1000) / 1000,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
return results;
|
||||
}
|
||||
|
||||
/**
|
||||
* Aggregate surge patterns across all routes to find global peak hours.
|
||||
* Groups by hour-of-day (0-23) across all detected routes.
|
||||
*/
|
||||
export function aggregateSurgeHours(
|
||||
surgeResults: SurgeResult[]
|
||||
): Array<{ hour: number; avgMultiplier: number; routeCount: number }> {
|
||||
const hourlyData = new Map<number, number[]>();
|
||||
|
||||
for (const sr of surgeResults) {
|
||||
for (const sp of sr.surgePrices) {
|
||||
const hour = parseInt(sp.time.split(':')[0]);
|
||||
if (!isNaN(hour)) {
|
||||
if (!hourlyData.has(hour)) hourlyData.set(hour, []);
|
||||
hourlyData.get(hour)!.push(sp.multiplier);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return Array.from(hourlyData.entries())
|
||||
.map(([hour, multipliers]) => ({
|
||||
hour,
|
||||
avgMultiplier: Math.round(
|
||||
(multipliers.reduce((a, b) => a + b, 0) / multipliers.length) * 1000
|
||||
) / 1000,
|
||||
routeCount: multipliers.length,
|
||||
}))
|
||||
.sort((a, b) => a.hour - b.hour);
|
||||
}
|
||||
@@ -0,0 +1,89 @@
|
||||
export interface ScrapedRide {
|
||||
id: number;
|
||||
task_id: string;
|
||||
app_name: string;
|
||||
competitor_name: string;
|
||||
start_lat: number;
|
||||
start_lng: number;
|
||||
end_lat: number;
|
||||
end_lng: number;
|
||||
price_amount: number;
|
||||
price_per_km: number;
|
||||
distance_km: number;
|
||||
duration_min: number;
|
||||
currency: string;
|
||||
country_code: string;
|
||||
scraped_at: string;
|
||||
created_at: string;
|
||||
}
|
||||
|
||||
export interface RideSample {
|
||||
distance_km: number;
|
||||
duration_min: number;
|
||||
price: number;
|
||||
ppk: number;
|
||||
startLat: number;
|
||||
startLng: number;
|
||||
endLat: number;
|
||||
endLng: number;
|
||||
scrapedAt: Date;
|
||||
competitorName: string;
|
||||
countryCode: string;
|
||||
}
|
||||
|
||||
export interface RouteGroup {
|
||||
key: string;
|
||||
rides: RideSample[];
|
||||
minPrice: number;
|
||||
maxPrice: number;
|
||||
avgPrice: number;
|
||||
distanceKm: number;
|
||||
durationMin: number;
|
||||
surgeMultiplier: number | null;
|
||||
}
|
||||
|
||||
export interface PricingTier {
|
||||
label: 'economy' | 'standard' | 'premium' | 'unknown';
|
||||
samples: RideSample[];
|
||||
ppkRange: [number, number];
|
||||
regression: RegressionResult | null;
|
||||
}
|
||||
|
||||
export interface RegressionResult {
|
||||
baseFare: number;
|
||||
kmRate: number;
|
||||
minRate: number;
|
||||
minFare: number;
|
||||
rmse: number;
|
||||
rSquared: number;
|
||||
sampleCount: number;
|
||||
hasMinFare: boolean;
|
||||
}
|
||||
|
||||
export interface SurgeResult {
|
||||
routeKey: string;
|
||||
distanceKm: number;
|
||||
basePrice: number;
|
||||
baseTime: string;
|
||||
surgePrices: Array<{ time: string; price: number; multiplier: number }>;
|
||||
maxMultiplier: number;
|
||||
}
|
||||
|
||||
export interface ZoneAnalysis {
|
||||
zoneKey: string;
|
||||
centerLat: number;
|
||||
centerLng: number;
|
||||
samples: RideSample[];
|
||||
avgPpk: number;
|
||||
tierDistribution: Record<string, number>;
|
||||
}
|
||||
|
||||
export interface AnalysisReport {
|
||||
competitorName: string;
|
||||
countryCode: string;
|
||||
tiers: PricingTier[];
|
||||
surgePatterns: SurgeResult[];
|
||||
zones: ZoneAnalysis[];
|
||||
totalSamples: number;
|
||||
analyzedAt: string;
|
||||
}
|
||||
@@ -0,0 +1,80 @@
|
||||
import { RideSample, ZoneAnalysis } from './types';
|
||||
import { assignZone, classifyZoneType } from './clustering';
|
||||
|
||||
/**
|
||||
* Analyze pricing by geographical zone (2.5km grid).
|
||||
* Groups samples into zones and computes per-zone statistics.
|
||||
*/
|
||||
export function analyzeByZone(samples: RideSample[]): ZoneAnalysis[] {
|
||||
const zoneMap = new Map<string, RideSample[]>();
|
||||
|
||||
for (const s of samples) {
|
||||
// Use start location for zone assignment
|
||||
const zone = assignZone(s.startLat, s.startLng);
|
||||
if (!zoneMap.has(zone)) zoneMap.set(zone, []);
|
||||
zoneMap.get(zone)!.push(s);
|
||||
}
|
||||
|
||||
const results: ZoneAnalysis[] = [];
|
||||
|
||||
for (const [zoneKey, zoneSamples] of zoneMap) {
|
||||
if (zoneSamples.length < 3) continue;
|
||||
|
||||
const ppkValues = zoneSamples.map(s => s.ppk);
|
||||
const avgPpk = Math.round(
|
||||
(ppkValues.reduce((a, b) => a + b, 0) / ppkValues.length) * 1000
|
||||
) / 1000;
|
||||
|
||||
// Count by tier — thresholds depend on currency scale
|
||||
const tierCounts: Record<string, number> = {};
|
||||
const sample = zoneSamples[0];
|
||||
const isHighDenom = sample.countryCode === 'SY' || sample.countryCode === 'IQ';
|
||||
const econThreshold = isHighDenom ? 15 : 0.35;
|
||||
const stdThreshold = isHighDenom ? 40 : 0.55;
|
||||
|
||||
for (const s of zoneSamples) {
|
||||
const tier =
|
||||
s.ppk < econThreshold ? 'economy' :
|
||||
s.ppk < stdThreshold ? 'standard' : 'premium';
|
||||
tierCounts[tier] = (tierCounts[tier] || 0) + 1;
|
||||
}
|
||||
|
||||
const [latStr, lngStr] = zoneKey.split(',');
|
||||
results.push({
|
||||
zoneKey,
|
||||
centerLat: parseFloat(latStr),
|
||||
centerLng: parseFloat(lngStr),
|
||||
samples: zoneSamples,
|
||||
avgPpk,
|
||||
tierDistribution: tierCounts,
|
||||
});
|
||||
}
|
||||
|
||||
return results.sort((a, b) => a.avgPpk - b.avgPpk);
|
||||
}
|
||||
|
||||
/**
|
||||
* Analyze pricing by zone type (centre, mid, suburb, outskirts).
|
||||
*/
|
||||
export function analyzeByZoneType(
|
||||
samples: RideSample[]
|
||||
): Array<{ zoneType: string; avgPpk: number; sampleCount: number; avgPrice: number }> {
|
||||
const typeMap = new Map<string, number[]>();
|
||||
|
||||
for (const s of samples) {
|
||||
const zoneType = classifyZoneType(s.startLat, s.startLng);
|
||||
if (!typeMap.has(zoneType)) typeMap.set(zoneType, []);
|
||||
typeMap.get(zoneType)!.push(s.ppk);
|
||||
}
|
||||
|
||||
return Array.from(typeMap.entries())
|
||||
.map(([zoneType, ppks]) => ({
|
||||
zoneType,
|
||||
avgPpk: Math.round(
|
||||
(ppks.reduce((a, b) => a + b, 0) / ppks.length) * 1000
|
||||
) / 1000,
|
||||
sampleCount: ppks.length,
|
||||
avgPrice: 0, // calculated below if needed
|
||||
}))
|
||||
.sort((a, b) => a.avgPpk - b.avgPpk);
|
||||
}
|
||||
@@ -0,0 +1,142 @@
|
||||
import mysql, { RowDataPacket, ResultSetHeader } from 'mysql2/promise';
|
||||
import dotenv from 'dotenv';
|
||||
import path from 'path';
|
||||
|
||||
dotenv.config({ path: path.resolve(__dirname, '../../.env') });
|
||||
|
||||
let mysqlPool: mysql.Pool | null = null;
|
||||
|
||||
export async function getMySQL(): Promise<mysql.Pool> {
|
||||
if (!mysqlPool) {
|
||||
mysqlPool = mysql.createPool({
|
||||
host: process.env.DB_HOST || '127.0.0.1',
|
||||
port: parseInt(process.env.DB_PORT || '3306'),
|
||||
database: process.env.DB_NAME || 'siro',
|
||||
user: process.env.DB_USER || 'root',
|
||||
password: process.env.DB_PASS || '',
|
||||
waitForConnections: true,
|
||||
connectionLimit: 5,
|
||||
queueLimit: 0,
|
||||
});
|
||||
}
|
||||
return mysqlPool;
|
||||
}
|
||||
|
||||
export async function fetchSamples(
|
||||
pool: mysql.Pool,
|
||||
competitorName?: string,
|
||||
countryCode?: string,
|
||||
hoursBack?: number
|
||||
): Promise<RowDataPacket[]> {
|
||||
const conditions: string[] = ['distance_km > 0', 'duration_min > 0', 'price_amount > 0'];
|
||||
const params: (string | number)[] = [];
|
||||
|
||||
if (competitorName) {
|
||||
conditions.push('competitor_name = ?');
|
||||
params.push(competitorName);
|
||||
}
|
||||
if (countryCode) {
|
||||
conditions.push('country_code = ?');
|
||||
params.push(countryCode);
|
||||
}
|
||||
if (hoursBack) {
|
||||
conditions.push('scraped_at >= DATE_SUB(NOW(), INTERVAL ? HOUR)');
|
||||
params.push(hoursBack);
|
||||
}
|
||||
|
||||
const sql = `SELECT * FROM scraped_competitor_prices WHERE ${conditions.join(' AND ')} ORDER BY id DESC LIMIT 10000`;
|
||||
const [rows] = await pool.query<RowDataPacket[]>(sql, params);
|
||||
return rows;
|
||||
}
|
||||
|
||||
export async function saveFormulas(
|
||||
pool: mysql.Pool,
|
||||
formulas: Array<{
|
||||
competitorName: string;
|
||||
countryCode: string;
|
||||
tier: string;
|
||||
baseFare: number;
|
||||
kmRate: number;
|
||||
minRate: number;
|
||||
minFare: number;
|
||||
rmse: number;
|
||||
rSquared: number;
|
||||
sampleCount: number;
|
||||
surgeMultiplier: number;
|
||||
peakHours: string;
|
||||
}>
|
||||
): Promise<void> {
|
||||
if (formulas.length === 0) return;
|
||||
|
||||
// Batch INSERT with ON DUPLICATE KEY UPDATE
|
||||
const values = formulas.map(f => `(?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, NOW())`).join(',');
|
||||
const flatParams: (string | number)[] = [];
|
||||
|
||||
for (const f of formulas) {
|
||||
flatParams.push(
|
||||
f.competitorName, f.countryCode, f.tier,
|
||||
f.baseFare, f.kmRate, f.minRate, f.minFare,
|
||||
f.rmse, f.rSquared, f.surgeMultiplier,
|
||||
f.sampleCount, f.peakHours
|
||||
);
|
||||
}
|
||||
|
||||
const sql = `INSERT INTO competitor_secret_formulas
|
||||
(competitor_name, country_code, tier, base_fare, price_per_km, price_per_min, min_fare, rmse, r_squared, surge_multiplier, sample_size, peak_hours, last_updated)
|
||||
VALUES ${values}
|
||||
ON DUPLICATE KEY UPDATE
|
||||
base_fare = VALUES(base_fare),
|
||||
price_per_km = VALUES(price_per_km),
|
||||
price_per_min = VALUES(price_per_min),
|
||||
min_fare = VALUES(min_fare),
|
||||
rmse = VALUES(rmse),
|
||||
r_squared = VALUES(r_squared),
|
||||
surge_multiplier = VALUES(surge_multiplier),
|
||||
sample_size = VALUES(sample_size),
|
||||
peak_hours = VALUES(peak_hours),
|
||||
last_updated = NOW()`;
|
||||
|
||||
await pool.execute(sql, flatParams);
|
||||
}
|
||||
|
||||
export async function saveSurgeInsights(
|
||||
pool: mysql.Pool,
|
||||
insights: Array<{
|
||||
competitorName: string;
|
||||
countryCode: string;
|
||||
surgeMultiplier: number;
|
||||
peakStartHour: number;
|
||||
peakEndHour: number;
|
||||
sampleCount: number;
|
||||
}>
|
||||
): Promise<void> {
|
||||
if (insights.length === 0) return;
|
||||
|
||||
const values = insights.map(() => `(?, ?, ?, ?, ?, ?, NOW())`).join(',');
|
||||
const flatParams: (string | number)[] = [];
|
||||
|
||||
for (const ins of insights) {
|
||||
flatParams.push(
|
||||
ins.competitorName, ins.countryCode,
|
||||
ins.surgeMultiplier, ins.peakStartHour,
|
||||
ins.peakEndHour, ins.sampleCount
|
||||
);
|
||||
}
|
||||
|
||||
const sql = `INSERT INTO competitor_surge_insights
|
||||
(competitor_name, country_code, surge_multiplier, peak_start_hour, peak_end_hour, sample_count, detected_at)
|
||||
VALUES ${values}
|
||||
ON DUPLICATE KEY UPDATE
|
||||
surge_multiplier = VALUES(surge_multiplier),
|
||||
sample_count = VALUES(sample_count),
|
||||
detected_at = NOW()`;
|
||||
|
||||
await pool.execute(sql, flatParams);
|
||||
}
|
||||
|
||||
export async function closeConnections(): Promise<void> {
|
||||
if (mysqlPool) {
|
||||
await mysqlPool.end();
|
||||
mysqlPool = null;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,218 @@
|
||||
/**
|
||||
* Siro Pricing Engine CLI
|
||||
*
|
||||
* Usage:
|
||||
* npm run analyze Full analysis all competitors
|
||||
* npm run analyze:taxif TaxiF only
|
||||
* npm run analyze -- --competitor=com.taxif.passenger --country=JO
|
||||
* npm run dev -- --mode=surge Surge-only analysis
|
||||
*
|
||||
* Cron integration: see crontab examples in package.json scripts
|
||||
*/
|
||||
|
||||
import { getMySQL, fetchSamples, saveFormulas, saveSurgeInsights, closeConnections } from './db/connection';
|
||||
import { runAnalysis } from './analysis/engine';
|
||||
import { Pool, RowDataPacket } from 'mysql2/promise';
|
||||
|
||||
interface CLIOptions {
|
||||
mode: 'full' | 'report';
|
||||
competitor?: string;
|
||||
country?: string;
|
||||
hoursBack?: number;
|
||||
}
|
||||
|
||||
function parseArgs(): CLIOptions {
|
||||
const args = process.argv.slice(2);
|
||||
const opts: CLIOptions = { mode: 'full' };
|
||||
|
||||
for (const arg of args) {
|
||||
if (arg.startsWith('--mode=')) {
|
||||
const mode = arg.split('=')[1];
|
||||
if (mode === 'full' || mode === 'report') {
|
||||
opts.mode = mode;
|
||||
}
|
||||
} else if (arg.startsWith('--competitor=')) {
|
||||
opts.competitor = arg.split('=')[1];
|
||||
} else if (arg.startsWith('--country=')) {
|
||||
opts.country = arg.split('=')[1];
|
||||
} else if (arg.startsWith('--hours=')) {
|
||||
opts.hoursBack = parseInt(arg.split('=')[1]);
|
||||
}
|
||||
}
|
||||
|
||||
return opts;
|
||||
}
|
||||
|
||||
interface CompetitorEntry {
|
||||
competitor_name: string;
|
||||
country_code: string;
|
||||
}
|
||||
|
||||
async function main(): Promise<void> {
|
||||
const opts = parseArgs();
|
||||
const startTime = Date.now();
|
||||
|
||||
console.log(`🚀 Siro Pricing Engine v1.0`);
|
||||
console.log(` Mode: ${opts.mode}`);
|
||||
if (opts.competitor) console.log(` Competitor: ${opts.competitor}`);
|
||||
if (opts.country) console.log(` Country: ${opts.country}`);
|
||||
console.log('');
|
||||
|
||||
try {
|
||||
const pool = await getMySQL();
|
||||
|
||||
const competitors = await fetchCompetitors(pool, opts);
|
||||
|
||||
if (competitors.length === 0) {
|
||||
console.log('❌ No competitors found with sufficient data.');
|
||||
return;
|
||||
}
|
||||
|
||||
// Process competitors in parallel for speed
|
||||
const results = await Promise.allSettled(
|
||||
competitors.map(comp => processCompetitor(pool, comp, opts))
|
||||
);
|
||||
|
||||
const succeeded = results.filter(r => r.status === 'fulfilled').length;
|
||||
const failed = results.filter(r => r.status === 'rejected').length;
|
||||
|
||||
const elapsed = ((Date.now() - startTime) / 1000).toFixed(1);
|
||||
console.log(`\n✨ Analysis complete in ${elapsed}s (${succeeded} succeeded, ${failed} failed)`);
|
||||
|
||||
if (failed > 0) {
|
||||
console.log('\n❌ Failures:');
|
||||
results.forEach((r, i) => {
|
||||
if (r.status === 'rejected') {
|
||||
console.log(` ${competitors[i].competitor_name} (${competitors[i].country_code}): ${r.reason}`);
|
||||
}
|
||||
});
|
||||
}
|
||||
} catch (err) {
|
||||
console.error('❌ Fatal error:', err);
|
||||
process.exit(1);
|
||||
} finally {
|
||||
await closeConnections();
|
||||
}
|
||||
}
|
||||
|
||||
async function processCompetitor(
|
||||
pool: Pool,
|
||||
comp: CompetitorEntry,
|
||||
opts: CLIOptions
|
||||
): Promise<void> {
|
||||
console.log(`\n📥 Fetching data for ${comp.competitor_name} (${comp.country_code})...`);
|
||||
const rows = await fetchSamples(pool, comp.competitor_name, comp.country_code, opts.hoursBack);
|
||||
|
||||
if (rows.length < 10) {
|
||||
console.log(` ⏩ Only ${rows.length} samples — skipping (need 10+)`);
|
||||
return;
|
||||
}
|
||||
|
||||
const samples = rows.map((row: RowDataPacket) => ({
|
||||
distance_km: parseFloat(row.distance_km),
|
||||
duration_min: parseFloat(row.duration_min),
|
||||
price: parseFloat(row.price_amount),
|
||||
ppk: parseFloat(row.price_per_km),
|
||||
startLat: parseFloat(row.start_lat),
|
||||
startLng: parseFloat(row.start_lng),
|
||||
endLat: parseFloat(row.end_lat),
|
||||
endLng: parseFloat(row.end_lng),
|
||||
scrapedAt: new Date(row.scraped_at),
|
||||
competitorName: row.competitor_name,
|
||||
countryCode: row.country_code,
|
||||
}));
|
||||
|
||||
const report = await runAnalysis(samples, {
|
||||
competitorName: comp.competitor_name,
|
||||
countryCode: comp.country_code,
|
||||
cleanOutliers: true,
|
||||
surgeThreshold: 0.12,
|
||||
tierCount: 3,
|
||||
});
|
||||
|
||||
// Save tier formulas
|
||||
const formulas = report.tiers
|
||||
.filter(t => t.regression !== null && t.regression!.sampleCount >= 5)
|
||||
.map(tier => ({
|
||||
competitorName: comp.competitor_name,
|
||||
countryCode: comp.country_code,
|
||||
tier: tier.label,
|
||||
baseFare: tier.regression!.baseFare,
|
||||
kmRate: tier.regression!.kmRate,
|
||||
minRate: tier.regression!.minRate,
|
||||
minFare: tier.regression!.minFare,
|
||||
rmse: tier.regression!.rmse,
|
||||
rSquared: tier.regression!.rSquared,
|
||||
sampleCount: tier.regression!.sampleCount,
|
||||
surgeMultiplier: 1.0,
|
||||
peakHours: '[]',
|
||||
}));
|
||||
|
||||
if (formulas.length > 0) {
|
||||
await saveFormulas(pool, formulas);
|
||||
console.log(` ✅ Saved ${formulas.length} tier formulas`);
|
||||
}
|
||||
|
||||
// Save surge insights — use the average multiplier across all detected routes
|
||||
if (opts.mode !== 'report' && report.surgePatterns.length > 0) {
|
||||
const avgMultiplier = report.surgePatterns
|
||||
.reduce((sum, sr) => sum + sr.maxMultiplier, 0) / report.surgePatterns.length;
|
||||
|
||||
// Find peak hour range from aggregate pattern
|
||||
const allHours = report.surgePatterns.flatMap(sr =>
|
||||
sr.surgePrices
|
||||
.filter(sp => sp.multiplier > 1.05)
|
||||
.map(sp => parseInt(sp.time.split(':')[0]))
|
||||
);
|
||||
|
||||
const peakStart = allHours.length > 0 ? Math.min(...allHours) : 0;
|
||||
const peakEnd = allHours.length > 0 ? Math.max(...allHours) : 23;
|
||||
|
||||
const surgeInsights = [{
|
||||
competitorName: comp.competitor_name,
|
||||
countryCode: comp.country_code,
|
||||
surgeMultiplier: Math.round(avgMultiplier * 1000) / 1000,
|
||||
peakStartHour: peakStart,
|
||||
peakEndHour: peakEnd,
|
||||
sampleCount: report.surgePatterns.length,
|
||||
}];
|
||||
|
||||
await saveSurgeInsights(pool, surgeInsights);
|
||||
console.log(` ✅ Saved surge insight: avg ${(avgMultiplier).toFixed(3)}x, hours ${peakStart}:00-${peakEnd}:00`);
|
||||
}
|
||||
}
|
||||
|
||||
async function fetchCompetitors(
|
||||
pool: Pool,
|
||||
opts: CLIOptions
|
||||
): Promise<CompetitorEntry[]> {
|
||||
if (opts.competitor) {
|
||||
const countryClause = opts.country ? 'AND country_code = ?' : '';
|
||||
const params: (string | number)[] = opts.country
|
||||
? [opts.competitor, opts.country]
|
||||
: [opts.competitor];
|
||||
|
||||
const [rows] = await pool.query<RowDataPacket[]>(
|
||||
`SELECT DISTINCT competitor_name, country_code
|
||||
FROM scraped_competitor_prices
|
||||
WHERE competitor_name = ?
|
||||
AND distance_km > 0 AND duration_min > 0 AND price_amount > 0
|
||||
${countryClause}
|
||||
LIMIT 10`,
|
||||
params
|
||||
);
|
||||
return rows as CompetitorEntry[];
|
||||
}
|
||||
|
||||
const [rows] = await pool.query<RowDataPacket[]>(
|
||||
`SELECT competitor_name, country_code, COUNT(*) as cnt
|
||||
FROM scraped_competitor_prices
|
||||
WHERE distance_km > 0 AND duration_min > 0 AND price_amount > 0
|
||||
GROUP BY competitor_name, country_code
|
||||
HAVING cnt >= 10
|
||||
ORDER BY cnt DESC`
|
||||
);
|
||||
return rows as CompetitorEntry[];
|
||||
}
|
||||
|
||||
main();
|
||||
@@ -0,0 +1,326 @@
|
||||
/**
|
||||
* Matrix and statistical utilities for pricing analysis.
|
||||
* Pure math — no external dependencies except simple-statistics.
|
||||
*/
|
||||
|
||||
import { median, mean, standardDeviation } from 'simple-statistics';
|
||||
|
||||
/**
|
||||
* Compute Pearson correlation between two arrays.
|
||||
*/
|
||||
function pearsonCorr(x: number[], y: number[]): number {
|
||||
const n = Math.min(x.length, y.length);
|
||||
if (n < 3) return 0;
|
||||
const mx = x.reduce((a, b) => a + b, 0) / n;
|
||||
const my = y.reduce((a, b) => a + b, 0) / n;
|
||||
let num = 0, dx2 = 0, dy2 = 0;
|
||||
for (let i = 0; i < n; i++) {
|
||||
const dx = x[i] - mx;
|
||||
const dy = y[i] - my;
|
||||
num += dx * dy;
|
||||
dx2 += dx * dx;
|
||||
dy2 += dy * dy;
|
||||
}
|
||||
const denom = Math.sqrt(dx2 * dy2);
|
||||
return denom === 0 ? 0 : num / denom;
|
||||
}
|
||||
|
||||
/**
|
||||
* Multiple linear regression via Gaussian elimination with ridge regularization.
|
||||
* Solves: price = baseFare + kmRate*distance + minRate*duration
|
||||
*
|
||||
* Uses L2 ridge (lambda=0.1) when distance≈duration are collinear.
|
||||
* Falls back to distance-only model if necessary.
|
||||
*/
|
||||
export function multipleLinearRegression(
|
||||
samples: Array<{ distance_km: number; duration_min: number; price: number }>
|
||||
): { baseFare: number; kmRate: number; minRate: number } | null {
|
||||
const n = samples.length;
|
||||
if (n < 3) return null;
|
||||
|
||||
// Check collinearity: if distance and duration are highly correlated
|
||||
const dists = samples.map(s => s.distance_km);
|
||||
const durs = samples.map(s => s.duration_min);
|
||||
const corr = pearsonCorr(dists, durs);
|
||||
|
||||
const lambda = Math.abs(corr) > 0.85 ? 0.5 : 0.01; // ridge penalty
|
||||
|
||||
let sumX1 = 0, sumX2 = 0, sumY = 0;
|
||||
let sumX1Sq = 0, sumX2Sq = 0, sumX1X2 = 0;
|
||||
let sumX1Y = 0, sumX2Y = 0;
|
||||
|
||||
for (const s of samples) {
|
||||
const x1 = s.distance_km;
|
||||
const x2 = s.duration_min;
|
||||
const y = s.price;
|
||||
|
||||
sumX1 += x1; sumX2 += x2; sumY += y;
|
||||
sumX1Sq += x1 * x1; sumX2Sq += x2 * x2; sumX1X2 += x1 * x2;
|
||||
sumX1Y += x1 * y; sumX2Y += x2 * y;
|
||||
}
|
||||
|
||||
// Ridge: add lambda to diagonal of X^T X (except intercept)
|
||||
const A = [
|
||||
[n, sumX1, sumX2],
|
||||
[sumX1, sumX1Sq + lambda, sumX1X2],
|
||||
[sumX2, sumX1X2, sumX2Sq + lambda],
|
||||
];
|
||||
|
||||
const B = [sumY, sumX1Y, sumX2Y];
|
||||
|
||||
try {
|
||||
const beta = gaussianElimination(A, B);
|
||||
const baseFare = Math.max(0, beta[0]);
|
||||
let kmRate = Math.max(0, beta[1]);
|
||||
let minRate = Math.max(0, beta[2]);
|
||||
|
||||
// If minRate is essentially zero after ridge, keep it minimal
|
||||
if (minRate < 0.001) minRate = 0;
|
||||
|
||||
// If both non-intercept terms are zero, try distance-only model
|
||||
if (kmRate === 0 && minRate === 0) {
|
||||
const k = sumX1Y / (sumX1Sq + lambda);
|
||||
if (k > 0) {
|
||||
kmRate = k;
|
||||
}
|
||||
}
|
||||
|
||||
return { baseFare, kmRate, minRate };
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Robust iterative regression to find floor pricing (exclude surge outliers).
|
||||
* Uses a fixed JOD/SYP threshold per iteration instead of tightening RMSE.
|
||||
*/
|
||||
export function robustMultipleLinearRegression(
|
||||
samples: Array<{ distance_km: number; duration_min: number; price: number }>,
|
||||
maxIterations: number = 4
|
||||
): { baseFare: number; kmRate: number; minRate: number } | null {
|
||||
let currentSamples = [...samples];
|
||||
let bestModel = multipleLinearRegression(currentSamples);
|
||||
if (!bestModel) return null;
|
||||
|
||||
// Determine threshold from data scale (median price × 0.3)
|
||||
const prices = samples.map(s => s.price).sort((a, b) => a - b);
|
||||
const medianPrice = prices[Math.floor(prices.length / 2)];
|
||||
const fixedThreshold = Math.max(medianPrice * 0.3, 0.1);
|
||||
|
||||
for (let i = 0; i < maxIterations; i++) {
|
||||
const predicted = currentSamples.map(
|
||||
s => bestModel!.baseFare + bestModel!.kmRate * s.distance_km + bestModel!.minRate * s.duration_min
|
||||
);
|
||||
const actual = currentSamples.map(s => s.price);
|
||||
|
||||
const inliers = currentSamples.filter((s, idx) => {
|
||||
const residual = actual[idx] - predicted[idx];
|
||||
return residual < fixedThreshold;
|
||||
});
|
||||
|
||||
if (inliers.length < Math.max(5, samples.length * 0.3)) break;
|
||||
if (inliers.length === currentSamples.length) break;
|
||||
|
||||
currentSamples = inliers;
|
||||
const newModel = multipleLinearRegression(currentSamples);
|
||||
if (!newModel) break;
|
||||
bestModel = newModel;
|
||||
}
|
||||
|
||||
return bestModel;
|
||||
}
|
||||
|
||||
/**
|
||||
* Gaussian elimination for solving Ax = B (3x3 system).
|
||||
*/
|
||||
function gaussianElimination(A: number[][], B: number[]): number[] {
|
||||
const n = A.length;
|
||||
const a = A.map(row => [...row]);
|
||||
const b = [...B];
|
||||
|
||||
for (let i = 0; i < n; i++) {
|
||||
let maxEl = Math.abs(a[i][i]);
|
||||
let maxRow = i;
|
||||
for (let k = i + 1; k < n; k++) {
|
||||
if (Math.abs(a[k][i]) > maxEl) {
|
||||
maxEl = Math.abs(a[k][i]);
|
||||
maxRow = k;
|
||||
}
|
||||
}
|
||||
|
||||
[a[maxRow], a[i]] = [a[i], a[maxRow]];
|
||||
[b[maxRow], b[i]] = [b[i], b[maxRow]];
|
||||
|
||||
if (Math.abs(a[i][i]) < 1e-12) continue;
|
||||
|
||||
for (let k = i + 1; k < n; k++) {
|
||||
const c = -a[k][i] / a[i][i];
|
||||
for (let j = i; j < n; j++) {
|
||||
if (i === j) a[k][j] = 0;
|
||||
else a[k][j] += c * a[i][j];
|
||||
}
|
||||
b[k] += c * b[i];
|
||||
}
|
||||
}
|
||||
|
||||
const x = new Array(n).fill(0);
|
||||
for (let i = n - 1; i >= 0; i--) {
|
||||
if (Math.abs(a[i][i]) < 1e-12) continue;
|
||||
x[i] = b[i] / a[i][i];
|
||||
for (let k = i - 1; k >= 0; k--) {
|
||||
b[k] -= a[k][i] * x[i];
|
||||
}
|
||||
}
|
||||
return x;
|
||||
}
|
||||
|
||||
/**
|
||||
* Calculate RMSE between predicted and actual values.
|
||||
*/
|
||||
export function calcRMSE(actual: number[], predicted: number[]): number {
|
||||
const n = Math.min(actual.length, predicted.length);
|
||||
if (n === 0) return Infinity;
|
||||
const sumSq = actual.reduce((sum, a, i) => {
|
||||
if (i >= predicted.length) return sum;
|
||||
return sum + (a - predicted[i]) ** 2;
|
||||
}, 0);
|
||||
return Math.sqrt(sumSq / n);
|
||||
}
|
||||
|
||||
/**
|
||||
* Calculate R² coefficient of determination.
|
||||
*/
|
||||
export function calcRSquared(actual: number[], predicted: number[]): number {
|
||||
const n = Math.min(actual.length, predicted.length);
|
||||
if (n < 2) return 0;
|
||||
const meanActual = mean(actual);
|
||||
const ssTot = actual.reduce((sum, y) => sum + (y - meanActual) ** 2, 0);
|
||||
if (ssTot === 0) return 1;
|
||||
const ssRes = actual.reduce((sum, y, i) => {
|
||||
if (i >= predicted.length) return sum;
|
||||
return sum + (y - predicted[i]) ** 2;
|
||||
}, 0);
|
||||
return 1 - ssRes / ssTot;
|
||||
}
|
||||
|
||||
/**
|
||||
* Median Absolute Deviation outlier detection.
|
||||
* Returns indices of inlier samples.
|
||||
*/
|
||||
export function findInliersMAD(
|
||||
values: number[],
|
||||
threshold: number = 3.5
|
||||
): number[] {
|
||||
const med = median(values);
|
||||
const absDevs = values.map(v => Math.abs(v - med));
|
||||
const mad = median(absDevs);
|
||||
if (mad === 0) return values.map((_, i) => i);
|
||||
|
||||
return values
|
||||
.map((v, i) => ({ v, i, modifiedZ: 0.6745 * Math.abs(v - med) / mad }))
|
||||
.filter(x => x.modifiedZ < threshold)
|
||||
.map(x => x.i);
|
||||
}
|
||||
|
||||
/**
|
||||
* K-Means clustering (for PPK-based tier detection).
|
||||
* Returns cluster assignments (0..k-1) for each sample.
|
||||
*/
|
||||
export function kMeans(
|
||||
values: number[],
|
||||
k: number,
|
||||
maxIterations: number = 100
|
||||
): number[] {
|
||||
if (values.length < k) return values.map(() => 0);
|
||||
|
||||
// Initialize centroids using k-means++
|
||||
let centroids: number[] = [];
|
||||
centroids.push(values[Math.floor(Math.random() * values.length)]);
|
||||
for (let c = 1; c < k; c++) {
|
||||
const dists = values.map(v => Math.min(
|
||||
...centroids.map(cent => Math.abs(v - cent))
|
||||
));
|
||||
const totalDist = dists.reduce((a, b) => a + b, 0);
|
||||
let r = Math.random() * totalDist;
|
||||
for (let i = 0; i < dists.length; i++) {
|
||||
r -= dists[i];
|
||||
if (r <= 0) {
|
||||
centroids.push(values[i]);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const assignments = new Array(values.length).fill(0);
|
||||
|
||||
for (let iter = 0; iter < maxIterations; iter++) {
|
||||
// Assign
|
||||
let changed = false;
|
||||
for (let i = 0; i < values.length; i++) {
|
||||
let minDist = Infinity;
|
||||
let bestCluster = 0;
|
||||
for (let c = 0; c < k; c++) {
|
||||
const dist = Math.abs(values[i] - centroids[c]);
|
||||
if (dist < minDist) {
|
||||
minDist = dist;
|
||||
bestCluster = c;
|
||||
}
|
||||
}
|
||||
if (assignments[i] !== bestCluster) {
|
||||
assignments[i] = bestCluster;
|
||||
changed = true;
|
||||
}
|
||||
}
|
||||
|
||||
if (!changed) break;
|
||||
|
||||
// Update centroids
|
||||
for (let c = 0; c < k; c++) {
|
||||
const clusterVals = values.filter((_, i) => assignments[i] === c);
|
||||
if (clusterVals.length > 0) {
|
||||
centroids[c] = mean(clusterVals);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Sort clusters by centroid value (ascending: economy < standard < premium)
|
||||
const centroidOrder = centroids
|
||||
.map((c, i) => ({ centroid: c, index: i }))
|
||||
.sort((a, b) => a.centroid - b.centroid);
|
||||
|
||||
const labelMap = new Map<number, number>();
|
||||
centroidOrder.forEach((item, newIdx) => labelMap.set(item.index, newIdx));
|
||||
|
||||
return assignments.map(a => labelMap.get(a)!);
|
||||
}
|
||||
|
||||
/**
|
||||
* Find "knee point" in price-vs-distance curve for minimum fare detection.
|
||||
* Uses simple piecewise linear fit.
|
||||
*/
|
||||
export function detectMinimumFare(
|
||||
distances: number[],
|
||||
prices: number[],
|
||||
kmRate: number
|
||||
): number | null {
|
||||
if (distances.length < 5) return null;
|
||||
|
||||
// Sort by distance
|
||||
const pairs = distances.map((d, i) => ({ d, p: prices[i] }))
|
||||
.sort((a, b) => a.d - b.d);
|
||||
|
||||
// Compute expected price without min fare
|
||||
const residuals = pairs.map(({ d, p }) => p - kmRate * d);
|
||||
|
||||
// Find where actual price consistently exceeds predicted
|
||||
// The minimum fare is the max of (price - kmRate*dist) for short rides
|
||||
const shortRides = pairs.filter(({ d }) => d < 10);
|
||||
if (shortRides.length < 3) return null;
|
||||
|
||||
const minFareEstimate = Math.max(
|
||||
...shortRides.map(({ d, p }) => p - kmRate * d)
|
||||
);
|
||||
|
||||
return minFareEstimate > 0 ? Math.round(minFareEstimate * 100) / 100 : null;
|
||||
}
|
||||
Reference in New Issue
Block a user