Update: 2026-07-06 17:51:16

This commit is contained in:
Hamza-Ayed
2026-07-06 17:51:16 +03:00
parent e21e1f4ec2
commit 9d257c5d7d
8 changed files with 292 additions and 241 deletions
@@ -0,0 +1,17 @@
-- Migration 002: Fix surge insights unique key
-- Problem: The old unique key included peak_start_hour and peak_end_hour,
-- which meant a new record was inserted every time the peak window shifted,
-- instead of updating the existing one.
-- Fix: The unique key now only covers (competitor_name, country_code),
-- so ON DUPLICATE KEY UPDATE always updates the existing row correctly.
-- Step 1: Drop the old unique key that included peak hours
ALTER TABLE `competitor_surge_insights`
DROP INDEX `unique_surge`;
-- Step 2: Add the correct unique key — one row per competitor per country
ALTER TABLE `competitor_surge_insights`
ADD UNIQUE KEY `unique_surge` (`competitor_name`, `country_code`);
-- Verify: show the new index
-- SHOW INDEX FROM `competitor_surge_insights`;
@@ -7,9 +7,29 @@ const TIER_LABELS: Array<'economy' | 'standard' | 'premium'> = [
'premium',
];
/**
* City center coordinates per country code.
* Used by classifyZoneType to measure distance from the urban center.
* Add more countries here as new competitors are onboarded.
*/
const CITY_CENTERS: Record<string, { lat: number; lng: number }> = {
JO: { lat: 31.95, lng: 35.90 }, // Amman, Jordan
SY: { lat: 33.51, lng: 36.29 }, // Damascus, Syria
IQ: { lat: 33.34, lng: 44.40 }, // Baghdad, Iraq
SA: { lat: 24.69, lng: 46.72 }, // Riyadh, Saudi Arabia
AE: { lat: 25.20, lng: 55.27 }, // Dubai, UAE
EG: { lat: 30.04, lng: 31.24 }, // Cairo, Egypt
LB: { lat: 33.89, lng: 35.50 }, // Beirut, Lebanon
KW: { lat: 29.37, lng: 47.98 }, // Kuwait City
};
/** Fallback city center when country code is not mapped yet */
const DEFAULT_CITY_CENTER = { lat: 31.95, lng: 35.90 }; // Amman
/**
* Cluster rides into pricing tiers based on price_per_km using K-Means.
* Returns sorted tiers (economy < standard < premium).
* Uses multi-run K-Means++ for stable, deterministic results.
*/
export function clusterTiers(
samples: RideSample[],
@@ -70,11 +90,19 @@ export function assignZone(lat: number, lng: number): string {
}
/**
* Classify zone type based on distance from city center (Amman: 31.95, 35.90).
* Classify zone type based on distance from the city center for a given country.
* Falls back to Amman coordinates if countryCode is not in CITY_CENTERS.
*
* Zone radii (in degrees, ~111km per degree):
* centre < 0.025° ≈ 2.8 km
* mid < 0.050° ≈ 5.6 km
* suburb < 0.100° ≈ 11.1 km
* outskirts ≥ 0.100°
*/
export function classifyZoneType(lat: number, lng: number): string {
const dlat = lat - 31.95;
const dlng = lng - 35.90;
export function classifyZoneType(lat: number, lng: number, countryCode: string = 'JO'): string {
const center = CITY_CENTERS[countryCode] ?? DEFAULT_CITY_CENTER;
const dlat = lat - center.lat;
const dlng = lng - center.lng;
const dist = Math.sqrt(dlat * dlat + dlng * dlng);
if (dist < 0.025) return 'centre';
+25 -10
View File
@@ -4,12 +4,19 @@ import { clusterTiers } from './clustering';
import { analyzeAllTiers } from './regression';
import { detectSurge, aggregateSurgeHours } from './surge';
import { analyzeByZone, analyzeByZoneType } from './zone';
import { median } from 'simple-statistics';
export interface EngineOptions {
competitorName?: string;
countryCode?: string;
cleanOutliers?: boolean;
surgeThreshold?: number;
/**
* Surge threshold as a fraction of the median price (e.g. 0.05 = 5%).
* Default: 0.05 (5% of median ride price).
* The absolute threshold is computed dynamically per dataset so it scales
* correctly across currencies (JOD, SYP, IQD, etc.).
*/
surgeThresholdFraction?: number;
tierCount?: number;
}
@@ -23,7 +30,7 @@ export async function runAnalysis(
): Promise<AnalysisReport> {
const {
cleanOutliers = true,
surgeThreshold = 0.12,
surgeThresholdFraction = 0.05,
tierCount = 3,
} = options;
@@ -36,21 +43,27 @@ export async function runAnalysis(
// Step 1: Remove statistical outliers (MAD on PPK)
const cleanSamples = cleanOutliers ? removeOutliers(samples) : samples;
// Step 2: Group by route and extract base (non-surge) prices
// Step 2: Compute dynamic surge threshold — 5% of median price, clamped to [0.05, 2.0]
// This ensures the threshold scales correctly for high-denomination currencies.
const allPrices = cleanSamples.map(s => s.price);
const medianPrice = median(allPrices);
const surgeThreshold = Math.min(Math.max(medianPrice * surgeThresholdFraction, 0.05), 2.0);
// Step 3: Group by route and extract base (non-surge) prices
const routeGroups = groupByRoute(cleanSamples);
const baseSamples = extractBasePrices(routeGroups, surgeThreshold);
// Step 3: Cluster into pricing tiers by PPK
// Step 4: Cluster into pricing tiers by PPK
const rawTiers = clusterTiers(cleanSamples, tierCount);
// Step 4: Run regression on each tier
// Step 5: Run regression on each tier
const analyzedTiers = analyzeAllTiers(rawTiers);
// Step 5: Detect surge patterns
// Step 6: Detect surge patterns
const surgePatterns = detectSurge(cleanSamples, surgeThreshold);
const surgeHours = aggregateSurgeHours(surgePatterns);
// Step 6: Zone analysis
// Step 7: Zone analysis
const zones = analyzeByZone(cleanSamples);
const zoneTypes = analyzeByZoneType(cleanSamples);
@@ -66,7 +79,7 @@ export async function runAnalysis(
};
// Print summary
printSummary(report, baseSamples, surgeHours, zoneTypes);
printSummary(report, baseSamples, surgeHours, zoneTypes, surgeThreshold);
return report;
}
@@ -75,12 +88,14 @@ function printSummary(
report: AnalysisReport,
baseSamples: RideSample[],
surgeHours: ReturnType<typeof aggregateSurgeHours>,
zoneTypes: ReturnType<typeof analyzeByZoneType>
zoneTypes: ReturnType<typeof analyzeByZoneType>,
surgeThreshold: number
): void {
const sep = '═══════════════════════════════════════════════════════';
console.log(`\n${sep}`);
console.log(` 📊 Pricing Analysis Report — ${report.competitorName} (${report.countryCode})`);
console.log(` ${report.totalSamples} total samples, ${baseSamples.length} base-price samples`);
console.log(` Surge threshold: ${surgeThreshold.toFixed(3)} (dynamic, 5% of median)`);
console.log(` Analyzed at: ${report.analyzedAt}`);
console.log(sep);
@@ -91,7 +106,7 @@ function printSummary(
if (reg) {
const tierIcon = tier.label === 'economy' ? '💰' : tier.label === 'standard' ? '🚗' : '💎';
console.log(` ${tierIcon} ${tier.label.toUpperCase()}:`);
console.log(` Base Fare: ${reg.baseFare.toFixed(3)} ${report.countryCode === 'JO' ? 'JOD' : report.countryCode === 'SY' ? 'SYP' : 'CUR'}`);
console.log(` Base Fare: ${reg.baseFare.toFixed(3)} ${report.countryCode === 'JO' ? 'JOD' : report.countryCode === 'SY' ? 'SYP' : report.countryCode === 'IQ' ? 'IQD' : 'CUR'}`);
console.log(` Per KM: ${reg.kmRate.toFixed(3)}`);
console.log(` Per Min: ${reg.minRate.toFixed(3)}`);
console.log(` Min Fare: ${reg.minFare.toFixed(3)} ${reg.hasMinFare ? '✅ active' : ''}`);
@@ -93,28 +93,3 @@ export function analyzeTier(tier: PricingTier): PricingTier {
export function analyzeAllTiers(tiers: PricingTier[]): PricingTier[] {
return tiers.map(tier => analyzeTier(tier));
}
/**
* Simple distance-only regression for comparison.
* price = kmRate * dist
*/
export function distanceOnlyRegression(
samples: RideSample[]
): { kmRate: number; rmse: number } | null {
if (samples.length < 3) return null;
const distances = samples.map(s => s.distance_km);
const prices = samples.map(s => s.price);
// Simple average of price/km
const ratios = distances.map((d, i) => d > 0 ? prices[i] / d : 0)
.filter(r => r > 0 && isFinite(r));
if (ratios.length < 3) return null;
const kmRate = mean(ratios);
const predicted = distances.map(d => kmRate * d);
const rmse = calcRMSE(prices, predicted);
return { kmRate, rmse };
}
+3 -1
View File
@@ -55,6 +55,7 @@ export function analyzeByZone(samples: RideSample[]): ZoneAnalysis[] {
/**
* Analyze pricing by zone type (centre, mid, suburb, outskirts).
* Passes countryCode to classifyZoneType so the correct city center is used.
*/
export function analyzeByZoneType(
samples: RideSample[]
@@ -62,7 +63,8 @@ export function analyzeByZoneType(
const typeMap = new Map<string, number[]>();
for (const s of samples) {
const zoneType = classifyZoneType(s.startLat, s.startLng);
// Pass countryCode so we use the correct city center (not always Amman)
const zoneType = classifyZoneType(s.startLat, s.startLng, s.countryCode);
if (!typeMap.has(zoneType)) typeMap.set(zoneType, []);
typeMap.get(zoneType)!.push(s.ppk);
}
+78 -16
View File
@@ -52,7 +52,7 @@ async function main(): Promise<void> {
const opts = parseArgs();
const startTime = Date.now();
console.log(`🚀 Siro Pricing Engine v1.0`);
console.log(`🚀 Siro Pricing Engine v1.1`);
console.log(` Mode: ${opts.mode}`);
if (opts.competitor) console.log(` Competitor: ${opts.competitor}`);
if (opts.country) console.log(` Country: ${opts.country}`);
@@ -95,6 +95,67 @@ async function main(): Promise<void> {
}
}
/**
* Compute the peak hours array from surge pattern data.
* Returns the longest contiguous block of hours where avg multiplier > 1.05.
* Used by both formula saving and surge insight saving.
*/
function computePeakHours(surgePatterns: Array<{ surgePrices: Array<{ time: string; multiplier: number }> }>): {
peakHours: number[];
peakStart: number;
peakEnd: number;
} {
// Aggregate all route multipliers per hour of day
const hourMults = new Map<number, number[]>();
for (const sr of surgePatterns) {
for (const sp of sr.surgePrices) {
const h = parseInt(sp.time.split(':')[0]);
if (isNaN(h)) continue;
if (!hourMults.has(h)) hourMults.set(h, []);
hourMults.get(h)!.push(sp.multiplier);
}
}
// Keep only hours where the average multiplier exceeds 1.05
const peakHours: number[] = [];
for (const [h, mults] of hourMults) {
const avg = mults.reduce((a, b) => a + b, 0) / mults.length;
if (avg > 1.05) peakHours.push(h);
}
peakHours.sort((a, b) => a - b);
// Find the longest contiguous block of peak hours
let bestStart = 0, bestEnd = 0, bestLen = 0;
let curStart = -1, curEnd = -1;
for (let i = 0; i < peakHours.length; i++) {
if (curStart < 0) {
curStart = peakHours[i];
curEnd = peakHours[i];
} else if (peakHours[i] === curEnd + 1) {
curEnd = peakHours[i];
} else {
if (curEnd - curStart > bestLen) {
bestLen = curEnd - curStart;
bestStart = curStart;
bestEnd = curEnd;
}
curStart = peakHours[i];
curEnd = peakHours[i];
}
}
if (curEnd - curStart > bestLen) {
bestLen = curEnd - curStart;
bestStart = curStart;
bestEnd = curEnd;
}
const peakStart = bestLen > 0 ? bestStart : 0;
const peakEnd = bestLen > 0 ? bestEnd : 23;
return { peakHours, peakStart, peakEnd };
}
async function processCompetitor(
pool: Pool,
comp: CompetitorEntry,
@@ -126,11 +187,18 @@ async function processCompetitor(
competitorName: comp.competitor_name,
countryCode: comp.country_code,
cleanOutliers: true,
surgeThreshold: 0.12,
// surgeThresholdFraction defaults to 0.05 (5% of median price) — currency-agnostic
tierCount: 3,
});
// Save tier formulas
// --- Compute peak hours once, reuse in both formulas and surge insights ---
const { peakHours, peakStart, peakEnd } = report.surgePatterns.length > 0
? computePeakHours(report.surgePatterns)
: { peakHours: [], peakStart: 0, peakEnd: 23 };
const peakHoursJson = JSON.stringify(peakHours);
// --- Save tier formulas (includes actual peak hours) ---
const formulas = report.tiers
.filter(t => t.regression !== null && t.regression!.sampleCount >= 5)
.map(tier => ({
@@ -145,29 +213,23 @@ async function processCompetitor(
rSquared: tier.regression!.rSquared,
sampleCount: tier.regression!.sampleCount,
surgeMultiplier: 1.0,
peakHours: '[]',
// Now populated with real peak hours instead of always '[]'
peakHours: peakHoursJson,
}));
if (formulas.length > 0) {
await saveFormulas(pool, formulas);
console.log(` ✅ Saved ${formulas.length} tier formulas`);
if (peakHours.length > 0) {
console.log(` Peak hours stored: [${peakHours.join(', ')}]`);
}
}
// Save surge insights — use the average multiplier across all detected routes
// --- Save surge insights ---
if (opts.mode !== 'report' && report.surgePatterns.length > 0) {
const avgMultiplier = report.surgePatterns
.reduce((sum, sr) => sum + sr.maxMultiplier, 0) / report.surgePatterns.length;
// Find peak hour range from aggregate pattern
const allHours = report.surgePatterns.flatMap(sr =>
sr.surgePrices
.filter(sp => sp.multiplier > 1.05)
.map(sp => parseInt(sp.time.split(':')[0]))
);
const peakStart = allHours.length > 0 ? Math.min(...allHours) : 0;
const peakEnd = allHours.length > 0 ? Math.max(...allHours) : 23;
const surgeInsights = [{
competitorName: comp.competitor_name,
countryCode: comp.country_code,
@@ -178,7 +240,7 @@ async function processCompetitor(
}];
await saveSurgeInsights(pool, surgeInsights);
console.log(` ✅ Saved surge insight: avg ${(avgMultiplier).toFixed(3)}x, hours ${peakStart}:00-${peakEnd}:00`);
console.log(` ✅ Saved surge insight: avg ${avgMultiplier.toFixed(3)}x, hours ${peakStart}:00-${peakEnd}:00`);
}
}
+107 -185
View File
@@ -1,13 +1,5 @@
/**
* Matrix and statistical utilities for pricing analysis.
* Pure math — no external dependencies except simple-statistics.
*/
import { median, mean } from 'simple-statistics';
import { median, mean, standardDeviation } from 'simple-statistics';
/**
* Compute Pearson correlation between two arrays.
*/
function pearsonCorr(x: number[], y: number[]): number {
const n = Math.min(x.length, y.length);
if (n < 3) return 0;
@@ -25,64 +17,53 @@ function pearsonCorr(x: number[], y: number[]): number {
return denom === 0 ? 0 : num / denom;
}
/**
* Multiple linear regression via Gaussian elimination with ridge regularization.
* Solves: price = baseFare + kmRate*distance + minRate*duration
*
* Uses L2 ridge (lambda=0.1) when distance≈duration are collinear.
* Falls back to distance-only model if necessary.
*/
export function multipleLinearRegression(
samples: Array<{ distance_km: number; duration_min: number; price: number }>
): { baseFare: number; kmRate: number; minRate: number } | null {
const n = samples.length;
if (n < 3) return null;
// Check collinearity: if distance and duration are highly correlated
const dists = samples.map(s => s.distance_km);
const durs = samples.map(s => s.duration_min);
const prices = samples.map(s => s.price);
const corr = pearsonCorr(dists, durs);
const lambda = Math.abs(corr) > 0.85 ? 0.5 : 0.01; // ridge penalty
const lambda = Math.abs(corr) > 0.85 ? 0.5 : 0.01;
let sumX1 = 0, sumX2 = 0, sumY = 0;
let sumX1Sq = 0, sumX2Sq = 0, sumX1X2 = 0;
let sumX1Y = 0, sumX2Y = 0;
for (const s of samples) {
const x1 = s.distance_km;
const x2 = s.duration_min;
const y = s.price;
const x1 = s.distance_km, x2 = s.duration_min, y = s.price;
sumX1 += x1; sumX2 += x2; sumY += y;
sumX1Sq += x1 * x1; sumX2Sq += x2 * x2; sumX1X2 += x1 * x2;
sumX1Y += x1 * y; sumX2Y += x2 * y;
}
// Ridge: add lambda to diagonal of X^T X (except intercept)
const A = [
[n, sumX1, sumX2],
[sumX1, sumX1Sq + lambda, sumX1X2],
[sumX2, sumX1X2, sumX2Sq + lambda],
[n, sumX1, sumX2],
[sumX1, sumX1Sq + lambda, sumX1X2],
[sumX2, sumX1X2, sumX2Sq + lambda],
];
const B = [sumY, sumX1Y, sumX2Y];
try {
const beta = gaussianElimination(A, B);
const baseFare = Math.max(0, beta[0]);
let baseFare = Math.max(0, beta[0]);
let kmRate = Math.max(0, beta[1]);
let minRate = Math.max(0, beta[2]);
// If minRate is essentially zero after ridge, keep it minimal
if (minRate < 0.001) minRate = 0;
const predFull = samples.map(s => baseFare + kmRate * s.distance_km + minRate * s.duration_min);
const rmseFull = calcRMSE(prices, predFull);
// If both non-intercept terms are zero, try distance-only model
if (kmRate === 0 && minRate === 0) {
const k = sumX1Y / (sumX1Sq + lambda);
if (k > 0) {
kmRate = k;
}
const ratioK = dists.reduce((a, d, i) => d > 0 ? a + prices[i] / d : a, 0) / dists.filter(d => d > 0).length;
if (!isFinite(ratioK)) return { baseFare, kmRate, minRate };
const predDistOnly = dists.map(d => ratioK * d);
const rmseDist = calcRMSE(prices, predDistOnly);
if (rmseDist <= rmseFull * 1.10) {
return { baseFare: 0, kmRate: Math.round(ratioK * 1000) / 1000, minRate: 0 };
}
return { baseFare, kmRate, minRate };
@@ -91,10 +72,6 @@ export function multipleLinearRegression(
}
}
/**
* Robust iterative regression to find floor pricing (exclude surge outliers).
* Uses a fixed JOD/SYP threshold per iteration instead of tightening RMSE.
*/
export function robustMultipleLinearRegression(
samples: Array<{ distance_km: number; duration_min: number; price: number }>,
maxIterations: number = 4
@@ -103,7 +80,6 @@ export function robustMultipleLinearRegression(
let bestModel = multipleLinearRegression(currentSamples);
if (!bestModel) return null;
// Determine threshold from data scale (median price × 0.3)
const prices = samples.map(s => s.price).sort((a, b) => a - b);
const medianPrice = prices[Math.floor(prices.length / 2)];
const fixedThreshold = Math.max(medianPrice * 0.3, 0.1);
@@ -113,214 +89,160 @@ export function robustMultipleLinearRegression(
s => bestModel!.baseFare + bestModel!.kmRate * s.distance_km + bestModel!.minRate * s.duration_min
);
const actual = currentSamples.map(s => s.price);
const inliers = currentSamples.filter((s, idx) => {
const residual = actual[idx] - predicted[idx];
return residual < fixedThreshold;
});
const inliers = currentSamples.filter((_, idx) => actual[idx] - predicted[idx] < fixedThreshold);
if (inliers.length < Math.max(5, samples.length * 0.3)) break;
if (inliers.length === currentSamples.length) break;
currentSamples = inliers;
const newModel = multipleLinearRegression(currentSamples);
if (!newModel) break;
bestModel = newModel;
}
return bestModel;
}
/**
* Gaussian elimination for solving Ax = B (3x3 system).
*/
function gaussianElimination(A: number[][], B: number[]): number[] {
const n = A.length;
const a = A.map(row => [...row]);
const b = [...B];
for (let i = 0; i < n; i++) {
let maxEl = Math.abs(a[i][i]);
let maxRow = i;
let maxEl = Math.abs(a[i][i]), maxRow = i;
for (let k = i + 1; k < n; k++) {
if (Math.abs(a[k][i]) > maxEl) {
maxEl = Math.abs(a[k][i]);
maxRow = k;
}
if (Math.abs(a[k][i]) > maxEl) { maxEl = Math.abs(a[k][i]); maxRow = k; }
}
[a[maxRow], a[i]] = [a[i], a[maxRow]];
[b[maxRow], b[i]] = [b[i], b[maxRow]];
if (Math.abs(a[i][i]) < 1e-12) continue;
for (let k = i + 1; k < n; k++) {
const c = -a[k][i] / a[i][i];
for (let j = i; j < n; j++) {
if (i === j) a[k][j] = 0;
else a[k][j] += c * a[i][j];
if (i === j) a[k][j] = 0; else a[k][j] += c * a[i][j];
}
b[k] += c * b[i];
}
}
const x = new Array(n).fill(0);
for (let i = n - 1; i >= 0; i--) {
if (Math.abs(a[i][i]) < 1e-12) continue;
x[i] = b[i] / a[i][i];
for (let k = i - 1; k >= 0; k--) {
b[k] -= a[k][i] * x[i];
}
for (let k = i - 1; k >= 0; k--) b[k] -= a[k][i] * x[i];
}
return x;
}
/**
* Calculate RMSE between predicted and actual values.
*/
export function calcRMSE(actual: number[], predicted: number[]): number {
const n = Math.min(actual.length, predicted.length);
if (n === 0) return Infinity;
const sumSq = actual.reduce((sum, a, i) => {
if (i >= predicted.length) return sum;
return sum + (a - predicted[i]) ** 2;
}, 0);
return Math.sqrt(sumSq / n);
return Math.sqrt(actual.reduce((sum, a, i) => i < predicted.length ? sum + (a - predicted[i]) ** 2 : sum, 0) / n);
}
/**
* Calculate R² coefficient of determination.
*/
export function calcRSquared(actual: number[], predicted: number[]): number {
const n = Math.min(actual.length, predicted.length);
if (n < 2) return 0;
const meanActual = mean(actual);
const ssTot = actual.reduce((sum, y) => sum + (y - meanActual) ** 2, 0);
if (ssTot === 0) return 1;
const ssRes = actual.reduce((sum, y, i) => {
if (i >= predicted.length) return sum;
return sum + (y - predicted[i]) ** 2;
}, 0);
return 1 - ssRes / ssTot;
return 1 - actual.reduce((sum, y, i) => i < predicted.length ? sum + (y - predicted[i]) ** 2 : sum, 0) / ssTot;
}
/**
* Median Absolute Deviation outlier detection.
* Returns indices of inlier samples.
*/
export function findInliersMAD(
values: number[],
threshold: number = 3.5
): number[] {
export function findInliersMAD(values: number[], threshold: number = 3.5): number[] {
const med = median(values);
const absDevs = values.map(v => Math.abs(v - med));
const mad = median(absDevs);
const mad = median(values.map(v => Math.abs(v - med)));
if (mad === 0) return values.map((_, i) => i);
return values
.map((v, i) => ({ v, i, modifiedZ: 0.6745 * Math.abs(v - med) / mad }))
.filter(x => x.modifiedZ < threshold)
.map(x => x.i);
return values.map((v, i) => ({ v, i, z: 0.6745 * Math.abs(v - med) / mad }))
.filter(x => x.z < threshold).map(x => x.i);
}
/**
* K-Means clustering (for PPK-based tier detection).
* Returns cluster assignments (0..k-1) for each sample.
* Compute K-Means inertia (sum of squared distances from each point to its centroid).
* Lower inertia = better clustering.
*/
function calcInertia(values: number[], assignments: number[], centroids: number[]): number {
return values.reduce((sum, v, i) => sum + (v - centroids[assignments[i]]) ** 2, 0);
}
/**
* Single K-Means run. Returns assignments, centroids, and inertia.
*/
function kMeansOnce(
values: number[],
k: number,
maxIterations: number
): { assignments: number[]; centroids: number[]; inertia: number } {
// K-Means++ seeding for better initialisation
const centroids: number[] = [];
centroids.push(values[Math.floor(Math.random() * values.length)]);
for (let c = 1; c < k; c++) {
const dists = values.map(v => Math.min(...centroids.map(cent => (v - cent) ** 2)));
const total = dists.reduce((a, b) => a + b, 0);
let r = Math.random() * total;
for (let i = 0; i < dists.length; i++) {
r -= dists[i];
if (r <= 0) { centroids.push(values[i]); break; }
}
// Fallback: if loop exits without pushing (floating point edge case)
if (centroids.length < c + 1) centroids.push(values[values.length - 1]);
}
const assignments = new Array(values.length).fill(0);
for (let iter = 0; iter < maxIterations; iter++) {
let changed = false;
for (let i = 0; i < values.length; i++) {
let minDist = Infinity, best = 0;
for (let c = 0; c < k; c++) {
const dist = Math.abs(values[i] - centroids[c]);
if (dist < minDist) { minDist = dist; best = c; }
}
if (assignments[i] !== best) { assignments[i] = best; changed = true; }
}
if (!changed) break;
for (let c = 0; c < k; c++) {
const clusterVals = values.filter((_, i) => assignments[i] === c);
if (clusterVals.length > 0) centroids[c] = mean(clusterVals);
}
}
const inertia = calcInertia(values, assignments, centroids);
return { assignments, centroids, inertia };
}
/**
* K-Means clustering with multiple restarts.
* Runs `runs` times and returns the assignment with the lowest inertia,
* eliminating randomness instability across different executions.
*/
export function kMeans(
values: number[],
k: number,
maxIterations: number = 100
maxIterations: number = 100,
runs: number = 8
): number[] {
if (values.length < k) return values.map(() => 0);
// Initialize centroids using k-means++
let centroids: number[] = [];
centroids.push(values[Math.floor(Math.random() * values.length)]);
for (let c = 1; c < k; c++) {
const dists = values.map(v => Math.min(
...centroids.map(cent => Math.abs(v - cent))
));
const totalDist = dists.reduce((a, b) => a + b, 0);
let r = Math.random() * totalDist;
for (let i = 0; i < dists.length; i++) {
r -= dists[i];
if (r <= 0) {
centroids.push(values[i]);
break;
}
let bestResult: { assignments: number[]; centroids: number[]; inertia: number } | null = null;
for (let run = 0; run < runs; run++) {
const result = kMeansOnce(values, k, maxIterations);
if (bestResult === null || result.inertia < bestResult.inertia) {
bestResult = result;
}
}
const assignments = new Array(values.length).fill(0);
for (let iter = 0; iter < maxIterations; iter++) {
// Assign
let changed = false;
for (let i = 0; i < values.length; i++) {
let minDist = Infinity;
let bestCluster = 0;
for (let c = 0; c < k; c++) {
const dist = Math.abs(values[i] - centroids[c]);
if (dist < minDist) {
minDist = dist;
bestCluster = c;
}
}
if (assignments[i] !== bestCluster) {
assignments[i] = bestCluster;
changed = true;
}
}
if (!changed) break;
// Update centroids
for (let c = 0; c < k; c++) {
const clusterVals = values.filter((_, i) => assignments[i] === c);
if (clusterVals.length > 0) {
centroids[c] = mean(clusterVals);
}
}
}
// Sort clusters by centroid value (ascending: economy < standard < premium)
const centroidOrder = centroids
.map((c, i) => ({ centroid: c, index: i }))
.sort((a, b) => a.centroid - b.centroid);
const labelMap = new Map<number, number>();
centroidOrder.forEach((item, newIdx) => labelMap.set(item.index, newIdx));
return assignments.map(a => labelMap.get(a)!);
// Re-order cluster indices so label 0 = lowest centroid (economy), etc.
const { assignments, centroids } = bestResult!;
const order = centroids.map((c, i) => ({ c, i })).sort((a, b) => a.c - b.c);
const map = new Map(order.map((item, idx) => [item.i, idx]));
return assignments.map(a => map.get(a)!);
}
/**
* Find "knee point" in price-vs-distance curve for minimum fare detection.
* Uses simple piecewise linear fit.
*/
export function detectMinimumFare(
distances: number[],
prices: number[],
kmRate: number
): number | null {
export function detectMinimumFare(distances: number[], prices: number[], kmRate: number): number | null {
if (distances.length < 5) return null;
// Sort by distance
const pairs = distances.map((d, i) => ({ d, p: prices[i] }))
.sort((a, b) => a.d - b.d);
// Compute expected price without min fare
const residuals = pairs.map(({ d, p }) => p - kmRate * d);
// Find where actual price consistently exceeds predicted
// The minimum fare is the max of (price - kmRate*dist) for short rides
const shortRides = pairs.filter(({ d }) => d < 10);
if (shortRides.length < 3) return null;
const minFareEstimate = Math.max(
...shortRides.map(({ d, p }) => p - kmRate * d)
);
return minFareEstimate > 0 ? Math.round(minFareEstimate * 100) / 100 : null;
const pairs = distances.map((d, i) => ({ d, p: prices[i] })).sort((a, b) => a.d - b.d);
const shortCount = Math.max(5, Math.floor(pairs.length * 0.3));
const shortRides = pairs.slice(0, shortCount);
const residuals = shortRides.map(({ d, p }) => p - kmRate * d).sort((a, b) => a - b);
const trimIdx = Math.max(0, Math.floor(residuals.length * 0.1));
const trimmed = residuals.slice(trimIdx, residuals.length - trimIdx);
const estimate = trimmed.length > 0 ? Math.max(...trimmed) : 0;
return estimate > 0 ? Math.round(estimate * 100) / 100 : null;
}
+30
View File
@@ -2146,3 +2146,33 @@ INSERT IGNORE INTO `api_quotas` (`service_name`, `daily_usage`, `quota_limit`, `
('gemini', 0, 100, CURDATE()),
('elevenlabs', 0, 5, CURDATE()),
('creatomate', 0, 2, CURDATE());
-- 3. Competitor Analysis & Surge Pricing Engine Tables
CREATE TABLE IF NOT EXISTS `competitor_secret_formulas` (
`id` INT AUTO_INCREMENT PRIMARY KEY,
`competitor_name` VARCHAR(100) NOT NULL,
`country_code` VARCHAR(10) NOT NULL,
`tier` VARCHAR(20) DEFAULT 'standard',
`base_fare` DECIMAL(8,3) NOT NULL,
`price_per_km` DECIMAL(8,3) NOT NULL,
`price_per_min` DECIMAL(8,3) NOT NULL,
`min_fare` DECIMAL(8,3) DEFAULT 0,
`rmse` DECIMAL(10,4) DEFAULT 0,
`r_squared` DECIMAL(10,4) DEFAULT 0,
`surge_multiplier` DECIMAL(5,3) DEFAULT 1.0,
`sample_size` INT DEFAULT 0,
`last_updated` TIMESTAMP NOT NULL DEFAULT CURRENT_TIMESTAMP ON UPDATE CURRENT_TIMESTAMP,
UNIQUE KEY `idx_comp_country_tier` (`competitor_name`, `country_code`, `tier`)
) ENGINE=InnoDB DEFAULT CHARSET=utf8mb4;
CREATE TABLE IF NOT EXISTS `competitor_surge_insights` (
`id` INT AUTO_INCREMENT PRIMARY KEY,
`competitor_name` VARCHAR(100) NOT NULL,
`country_code` VARCHAR(5) NOT NULL,
`surge_multiplier` DECIMAL(5,3) NOT NULL,
`peak_start_hour` INT NOT NULL,
`peak_end_hour` INT NOT NULL,
`sample_count` INT NOT NULL,
`detected_at` TIMESTAMP DEFAULT CURRENT_TIMESTAMP,
UNIQUE KEY `unique_surge` (`competitor_name`, `country_code`, `peak_start_hour`, `peak_end_hour`)
) ENGINE=InnoDB DEFAULT CHARSET=utf8mb4;