|
| 1 | +import type { Metadata } from "next"; |
| 2 | +import Image from "next/image"; |
| 3 | +import Link from "next/link"; |
| 4 | +import { alpasim2026 as challenge } from "@/data/alpasim2026"; |
| 5 | + |
| 6 | +export const metadata: Metadata = { |
| 7 | + title: `${challenge.title} | OpenDriveLab`, |
| 8 | + description: challenge.description, |
| 9 | + openGraph: { |
| 10 | + title: challenge.title, |
| 11 | + description: challenge.description, |
| 12 | + images: [challenge.image], |
| 13 | + }, |
| 14 | +}; |
| 15 | + |
| 16 | +const navigation = [ |
| 17 | + ["Overview", "overview"], |
| 18 | + ["Tracks", "tracks"], |
| 19 | + ["Evaluation", "evaluation"], |
| 20 | + ["Timeline", "timeline"], |
| 21 | +]; |
| 22 | + |
| 23 | +const resources = [ |
| 24 | + ["Challenge website & leaderboard", challenge.website], |
| 25 | + ["Code & submission instructions", challenge.repository], |
| 26 | + ["Read the announcement", challenge.blog], |
| 27 | +]; |
| 28 | + |
| 29 | +export default function AlpasimChallenge() { |
| 30 | + return ( |
| 31 | + <article className="mx-auto w-full max-w-7xl px-6 pt-36 md:pt-28"> |
| 32 | + <Link href="/events" className="text-sm text-o-gray animated-underline-gray">← All events</Link> |
| 33 | + |
| 34 | + <header className="mt-8 flex flex-col gap-6"> |
| 35 | + <p className="text-sm font-semibold uppercase tracking-widest text-o-blue">Autonomous driving · Challenge 2026</p> |
| 36 | + <h1 className="max-w-5xl text-4xl font-bold leading-tight fg-gradient-blue md:text-6xl">{challenge.title}</h1> |
| 37 | + <p className="max-w-3xl text-lg leading-relaxed text-o-gray">{challenge.description}</p> |
| 38 | + <p className="text-sm leading-relaxed">Organized by HKU OpenDriveLab, NVIDIA ASPIRE Group, and KE:SAI.</p> |
| 39 | + <div className="flex flex-wrap gap-3"> |
| 40 | + <a href={challenge.website} target="_blank" rel="noopener noreferrer" className="rounded-sm bg-o-blue px-6 py-3 font-semibold text-white transition hover:bg-o-dark-blue">Join the challenge ↗</a> |
| 41 | + <a href={challenge.repository} target="_blank" rel="noopener noreferrer" className="rounded-sm border border-o-blue px-6 py-3 font-semibold text-o-blue transition hover:bg-o-blue/5">Code & developer resources ↗</a> |
| 42 | + <a href={challenge.wechat} target="_blank" rel="noopener noreferrer" className="rounded-sm border border-o-blue px-6 py-3 font-semibold text-o-blue transition hover:bg-o-blue/5">WeChat article (中文) ↗</a> |
| 43 | + </div> |
| 44 | + <Image src={challenge.image} alt="AlpaSim E2E Closed Loop Challenge: a closed-loop benchmark for evaluating autonomous driving policies" width={1058} height={468} priority sizes="(max-width: 1280px) 100vw, 1280px" className="mt-4 h-auto w-full rounded-sm" /> |
| 45 | + </header> |
| 46 | + |
| 47 | + <nav aria-label="On this page" className="my-10 flex flex-wrap gap-x-8 gap-y-3 border-y border-foreground/10 py-5 text-sm"> |
| 48 | + {navigation.map(([label, id]) => <a key={id} href={`#${id}`} className="text-o-blue animated-underline">{label}</a>)} |
| 49 | + </nav> |
| 50 | + |
| 51 | + <div className="flex flex-col gap-20 md:gap-28"> |
| 52 | + <section id="overview" className="scroll-mt-28 space-y-6"> |
| 53 | + <h2 className="text-3xl font-bold fg-gradient-blue">From trajectory matching to closed-loop driving</h2> |
| 54 | + <p className="max-w-4xl leading-relaxed">How do we know whether an autonomous driving model is truly getting better? Matching a recorded trajectory is only part of the answer. A useful evaluation must also ask whether a vehicle avoids collisions, stays on the road, completes its task, and produces decisions within the computing limits of a real vehicle.</p> |
| 55 | + <p className="max-w-4xl leading-relaxed">AlpaSim brings these questions into a shared, open evaluation framework. Every predicted trajectory changes the simulated vehicle’s state. The simulator then renders new sensor observations and feeds them back to the policy, allowing evaluation to follow the consequences of successive decisions.</p> |
| 56 | + <ol aria-label="Closed-loop evaluation cycle" className="grid gap-3 sm:grid-cols-2 lg:grid-cols-4"> |
| 57 | + {["Observe the environment", "Predict a trajectory", "Update the vehicle state", "Render new observations"].map((step, index) => ( |
| 58 | + <li key={step} className="rounded-sm border border-o-blue/20 bg-o-blue/5 p-6"> |
| 59 | + <span className="text-sm font-semibold text-o-blue">0{index + 1}</span> |
| 60 | + <p className="mt-3 font-semibold">{step}</p> |
| 61 | + </li> |
| 62 | + ))} |
| 63 | + </ol> |
| 64 | + <p className="max-w-4xl leading-relaxed">This continuous loop tests whether a policy can recover from deviations, whether errors accumulate over time, and how a mistaken decision affects the rest of a drive.</p> |
| 65 | + <details className="rounded-sm border border-foreground/10 p-6"> |
| 66 | + <summary className="cursor-pointer font-semibold text-o-blue">Why move beyond open-loop evaluation and NAVSIM?</summary> |
| 67 | + <div className="mt-6 space-y-5 leading-relaxed"> |
| 68 | + <p>A recorded drive is one valid solution, not the only safe way to navigate a road. Safely slowing down can produce a larger Average Displacement Error (ADE) than an unsafe trajectory that ends closer to the recording. Trajectory error alone can therefore reward the wrong behavior.</p> |
| 69 | + <figure> |
| 70 | + {/* Preserve the source figure’s native proportions. */} |
| 71 | + <img src="/images/alpasim2026/trajectory-error.png" alt="Comparison of safe speed changes and unsafe swerves using ADE and driving compliance" loading="lazy" className="mx-auto h-auto w-full max-w-4xl rounded-sm" /> |
| 72 | + <figcaption className="mt-3 text-sm text-o-gray">A safe speed change can have higher ADE than swerving off the road or into oncoming traffic.</figcaption> |
| 73 | + </figure> |
| 74 | + <p>NAVSIM evaluates safety, comfort, and driving progress through lightweight simulation. Since 2024, it has supported three public challenges; the first attracted 143 teams and 463 submissions. NAVSIM v2 added pseudo-simulation, using 3D Gaussian Splatting to create observations at different positions, headings, and speeds and assess recovery beyond the recorded trajectory.</p> |
| 75 | + <p>AlpaSim takes the next step: a full sensor-level loop in which a policy continuously observes, acts, and receives fresh observations generated from the state it has reached.</p> |
| 76 | + </div> |
| 77 | + </details> |
| 78 | + </section> |
| 79 | + |
| 80 | + <section id="tracks" className="scroll-mt-28 space-y-6"> |
| 81 | + <h2 className="text-3xl font-bold fg-gradient-blue">Two complementary tracks</h2> |
| 82 | + <p className="max-w-4xl leading-relaxed">The challenge pairs evaluation at industry scale with an accessible path for reproducible academic research. A common developer kit supports development and training on both datasets.</p> |
| 83 | + <div className="grid gap-6 md:grid-cols-2"> |
| 84 | + <div className="rounded-sm border border-foreground/10 p-6 md:p-8"> |
| 85 | + <p className="text-sm font-semibold uppercase tracking-wider text-o-blue">Industry-scale evaluation</p> |
| 86 | + <h3 className="my-4 text-2xl font-bold">Physical AI AV Track</h3> |
| 87 | + <p className="leading-relaxed">Built on the NVIDIA Physical AI Autonomous Vehicles Dataset, with approximately 1,700 hours of driving data across diverse regions and environments. NuRec supports closed-loop testing of stability, generalization, and computational efficiency at scale.</p> |
| 88 | + </div> |
| 89 | + <div className="rounded-sm border border-foreground/10 p-6 md:p-8"> |
| 90 | + <p className="text-sm font-semibold uppercase tracking-wider text-o-blue">Reproducible research</p> |
| 91 | + <h3 className="my-4 text-2xl font-bold">nuPlan Track</h3> |
| 92 | + <p className="leading-relaxed">Built on the nuPlan ecosystem, extending the evaluation direction of <a href="https://github.com/OpenDriveLab/WorldEngine" className="text-o-blue animated-underline">WorldEngine</a> with <a href="https://github.com/OpenDriveLab/MTGS/" className="text-o-blue animated-underline">MTGS</a> reconstruction assets. Researchers can adapt NAVSIM-style models for method comparisons, model iteration, and closed-loop behavior analysis.</p> |
| 93 | + </div> |
| 94 | + </div> |
| 95 | + <figure> |
| 96 | + <div className="grid grid-cols-2 gap-3 lg:grid-cols-4"> |
| 97 | + {[1, 2, 3, 4].map(index => ( |
| 98 | + <img key={index} src={`/images/alpasim2026/driving-${index}.gif`} alt={`Closed-loop driving example ${index}, showing a road scene and the corresponding simulation view`} loading="lazy" className="h-auto w-full rounded-sm bg-black" /> |
| 99 | + ))} |
| 100 | + </div> |
| 101 | + <figcaption className="mt-3 text-sm text-o-gray">Closed-loop driving examples from the challenge, with road scenes and corresponding simulation visualizations.</figcaption> |
| 102 | + </figure> |
| 103 | + </section> |
| 104 | + |
| 105 | + <section id="evaluation" className="scroll-mt-28 space-y-8"> |
| 106 | + <h2 className="text-3xl font-bold fg-gradient-blue">Evaluation that accounts for deployment</h2> |
| 107 | + <div className="grid gap-6 sm:grid-cols-3"> |
| 108 | + {[["16 GiB", "GPU memory limit"], ["≤ 0.1 s", "Target model work per Drive call"], ["Docker", "Containerized policy submissions"]].map(([value, label]) => ( |
| 109 | + <div key={value} className="border-l-2 border-o-blue pl-5 py-2"> |
| 110 | + <p className="text-3xl font-bold text-o-blue">{value}</p> |
| 111 | + <p className="mt-2 text-sm text-o-gray">{label}</p> |
| 112 | + </div> |
| 113 | + ))} |
| 114 | + </div> |
| 115 | + <p className="max-w-4xl leading-relaxed">The organizers provide the computing resources and runtime environment for official evaluation. Each team receives the same number of official submission opportunities, while standardized local validation sets support debugging and ablation studies. Observation and control frequencies differ between tracks; all submissions must meet the challenge’s runtime requirements.</p> |
| 116 | + <div className="max-w-4xl space-y-4"> |
| 117 | + <h3 className="text-2xl font-bold">Beyond an average score</h3> |
| 118 | + <p className="leading-relaxed">The challenge is exploring Item Response Theory (IRT) to estimate policy ability and scenario difficulty from patterns of success and failure. Treating policies as test takers and scenarios as questions helps reveal differences in difficulty and discrimination, and provides uncertainty intervals alongside rankings.</p> |
| 119 | + <p className="leading-relaxed">Physical AI AV and nuPlan are scored independently. For policies with overlapping rank intervals, the challenge further considers the average distance driven per at-fault infraction.</p> |
| 120 | + </div> |
| 121 | + </section> |
| 122 | + |
| 123 | + <section id="timeline" className="scroll-mt-28 space-y-6"> |
| 124 | + <h2 className="text-3xl font-bold fg-gradient-blue">Challenge timeline</h2> |
| 125 | + <ol className="grid gap-6 md:grid-cols-3"> |
| 126 | + {[ |
| 127 | + ["June 15, 2026", "Challenge opens", "Start developing and evaluating your driving policy."], |
| 128 | + ["September 15, 2026", "Planned rules freeze", "Rules and submission formats freeze following maintenance."], |
| 129 | + ["October 31, 2026", "Final submissions", "Public leaderboard closes; final submissions and technical reports are due."], |
| 130 | + ].map(([date, title, detail]) => ( |
| 131 | + <li key={date} className="border-t-2 border-o-blue pt-5"> |
| 132 | + <p className="text-sm font-semibold text-o-blue">{date}</p> |
| 133 | + <h3 className="mt-3 text-xl font-bold">{title}</h3> |
| 134 | + <p className="mt-2 leading-relaxed text-o-gray">{detail}</p> |
| 135 | + </li> |
| 136 | + ))} |
| 137 | + </ol> |
| 138 | + <p className="text-sm text-o-gray">Dates follow the published challenge plan. See the <a href={challenge.website} className="text-o-blue animated-underline">official challenge website</a> for the latest schedule, rules, and submission details.</p> |
| 139 | + </section> |
| 140 | + |
| 141 | + <section className="rounded-sm bg-o-blue/5 p-6 md:p-10"> |
| 142 | + <h2 className="text-3xl font-bold fg-gradient-blue">Make progress measurable</h2> |
| 143 | + <p className="mt-4 max-w-3xl leading-relaxed">Join the community in developing safer, more efficient, and more generalizable end-to-end driving policies—and help make every improvement reliably measurable, reproducible, and verifiable.</p> |
| 144 | + <div className="mt-6 flex flex-wrap gap-x-8 gap-y-4"> |
| 145 | + {resources.map(([label, url]) => <a key={url} href={url} target="_blank" rel="noopener noreferrer" className="font-semibold text-o-blue animated-underline">{label} ↗</a>)} |
| 146 | + </div> |
| 147 | + </section> |
| 148 | + </div> |
| 149 | + </article> |
| 150 | + ); |
| 151 | +} |
0 commit comments