Every language
12 langages, copy-ready. One at a time with syntax highlighting, or all inline.
JSJavaScript
// Promise.all settles by INPUT INDEX, not completion time:
const results = await Promise.all(
items.map((item, i) => fetchItem(item, i)),
);
// results[0] is items[0]'s result even if it resolved last.
// collect-everything instead of fail-fast:
const settled = await Promise.allSettled(items.map(fetchItem));
const ok = settled.flatMap(s => (s.status === 'fulfilled' ? [s.value] : []));Promise.all IS order-preserving — it never shuffles; the map() builds the array in input order and Promise.all fills those slots. It is fail-fast: the first rejection rejects the whole await while the remaining promises keep running uncancelled. allSettled never rejects — filter by status to partition.
TSTypeScript
async function parallelMap<T, R>(
items: readonly T[],
fn: (item: T, index: number) => Promise<R>,
): Promise<R[]> {
return Promise.all(items.map((item, i) => fn(item, i)));
}The signature pins what the dynamic version hides: R[] means results in input order, and one rejected promise rejects the whole call. For collect-everything, return allSettled and partition — Promise<Result<R, unknown>[]> keeps both halves typed.
GoGo
import "golang.org/x/sync/errgroup"
func parallelMap[T, R any](items []T, fn func(T) (R, error)) ([]R, error) {
results := make([]R, len(items)) // pre-sized: no append, no shared writer
var g errgroup.Group
g.SetLimit(8) // bounded — 10k items must not mean 10k goroutines
for i, item := range items {
g.Go(func() error {
res, err := fn(item)
if err != nil {
return err
}
results[i] = res // safe: each goroutine writes its OWN slot
return nil
})
}
if err := g.Wait(); err != nil {
return nil, err
}
return results, nil
}g.Wait() returns the FIRST error — fail-fast, not collect. errgroup.Group does not cancel anyone; use g, ctx := errgroup.WithContext(ctx) and pass ctx into fn if the rest should stop. results[i] = res is race-free (distinct slots); a shared append is the data race. Go 1.22+ loop vars are per-iteration — pre-1.22, capture i := i inside the loop.
RsRust
use rayon::prelude::*;
let results: Vec<u16> = urls
.par_iter() // work-stealing parallel iterator
.map(|url| fetch_status(url))
.collect(); // Vec in INPUT order
// async twin: futures::future::join_all(futs).awaitrayon's collect() rebuilds input order even though tasks migrate between threads — that is the whole point of a parallel iterator. The stdlib-only answer is std::thread::scope over chunked slices (each thread maps its chunk, then concatenate chunks in order). join_all is ordered but does NOT short-circuit — try_join_all is its fail-fast sibling.
PHPPHP
// trap: array_map is SEQUENTIAL — PHP core has no parallel map.
// CPU-bound fallback (CLI + unix only): fork workers over slices,
// serialize each slice back, merge BY INDEX (keys preserved):
function parallel_map(array $items, callable $fn, int $workers = 4): array {
$chunks = array_chunk($items, (int) ceil(count($items) / $workers), true);
$pairs = [];
foreach ($chunks as $chunk) {
[$parent, $child] = stream_socket_pair(STREAM_PF_UNIX, STREAM_SOCK_STREAM, 0);
$pairs[] = [$parent, $child];
if (pcntl_fork() === 0) { // worker process
fclose($parent);
fwrite($child, serialize(array_map($fn, $chunk)));
exit(0);
}
fclose($child);
}
$results = [];
foreach ($pairs as [$parent]) {
$results += unserialize(stream_get_contents($parent)); // union keeps keys
fclose($parent);
}
pcntl_waitpid(0, $status); // reap workers
ksort($results);
return $results;
}Honest ecosystem gap: no built-in parallel map — forking (CLI-only, no Windows) or an event loop (amphp/parallel, ReactPHP) is the whole toolbox, and array_map itself never leaves one core. The index discipline lives in array_chunk(..., true) + ksort: keys ARE the original indexes, so shuffle-proof order falls out.
PyPython
from concurrent.futures import ThreadPoolExecutor
with ThreadPoolExecutor(max_workers=8) as pool:
lazy = pool.map(fetch, items) # LAZY and ORDERED
results = list(lazy) # list() forces the work
# asyncio twin: await asyncio.gather(*[fetch(i) for i in items])executor.map preserves input order and is lazy — nothing runs until iterated, so list() (or summing/consuming) is the forcing step. Exceptions are fail-fast-ish: the first one raises at the point its index is REACHED during iteration, not when the task fails. gather(return_exceptions=True) is the async collect-everything twin.
C#C#
using System;
using System.Linq;
using System.Threading.Tasks;
static async Task<R[]> ParallelMapAsync<T, R>(
IEnumerable<T> items, Func<T, Task<R>> fn) =>
await Task.WhenAll(items.Select(fn)); // R[] in SOURCE order
// collect-everything: wrap each, then partition —
static async Task<(R[] ok, Exception[] failed)> Partition<T, R>(
IEnumerable<T> items, Func<T, Task<R>> fn)
{
var wrapped = await Task.WhenAll(items.Select(async item =>
{
try { return (value: await fn(item), error: null as Exception); }
catch (Exception e) { return (default!, e); }
}));
return (wrapped.Where(w => w.error is null).Select(w => w.value!).ToArray(),
wrapped.Where(w => w.error is not null).Select(w => w.error!).ToArray());
}Task.WhenAll yields R[] in source order no matter the completion sequence; the await is fail-fast (first fault throws — the surviving tasks keep running, uncancelled). Parallel.ForEach + thread-local aggregation is the CPU-bound cousin, and PLINQ preserves order ONLY with .AsOrdered() — without it the stream may hand you reordered output.
JvJava
import java.util.concurrent.*;
import java.util.ArrayList;
import java.util.List;
static <R> List<R> parallelMap(List<Callable<R>> tasks) throws InterruptedException {
try (ExecutorService pool = Executors.newFixedThreadPool(8)) {
List<Future<R>> futures = pool.invokeAll(tasks); // blocks until ALL finish
List<R> results = new ArrayList<>(futures.size());
for (Future<R> f : futures) { // futures are in SUBMISSION order
results.add(f.get()); // wraps the task's exception
}
return results;
}
}invokeAll blocks and returns futures in SUBMISSION order — iterating them in that order gives input order for free. f.get() rethrows the first failure AT ITS INDEX — fail-fast. Parallel streams are a different tool: forEach order is nondeterministic and only collect() on an ordered stream preserves it. CompletableFuture.allOf is the non-blocking form.
SwSwift
func parallelMap<T, R: Sendable>(
_ items: [T], _ transform: @Sendable (T) async throws -> R
) async throws -> [R] {
var results = [R?](repeating: nil, count: items.count)
try await withThrowingTaskGroup(of: (Int, R).self) { group in
for (i, item) in items.enumerated() {
group.addTask { (i, try await transform(item)) } // CARRY the index
}
for try await (i, value) in group { // yields in COMPLETION order
results[i] = value // index restores input order
}
}
return results.map { $0! }
}addTask has NO ordering guarantee — the group yields children as they finish, so the (index, result) tuple is the whole trick: write into a pre-sized array at the carried index. withThrowingTaskGroup is fail-fast (first throw cancels the remaining children); the non-throwing TaskGroup plus Result-wrapped children is the collect-everything shape.
KtKotlin
suspend fun <T, R> parallelMap(
items: List<T>,
fn: suspend (T) -> R,
): List<R> = coroutineScope {
items.map { item ->
async { fn(item) } // launched in input order
}.awaitAll() // results in LAUNCH order, not completion
}async + awaitAll() is the ordered answer: awaitAll returns results in launch order however the coroutines finish. Structured concurrency makes it fail-fast — one uncaught exception cancels the whole scope. Collect-everything: wrap each body in runCatching { fn(item) } and partition the Results yourself (a SupervisorJob alone does not aggregate).
RbRuby
def parallel_map(items, &fn)
results = Array.new(items.size) # pre-sized slots
threads = items.each_with_index.map do |item, i|
Thread.new { results[i] = fn.call(item) } # index-write: own slot
end
threads.each(&:join)
results
endarr[i] = is the thread-safe shape — no two threads touch the same slot, so no lock is needed. The race is results << value (or each_with_object): shared push makes completion order the array order and is not atomic on JRuby/TruffleRuby. Concurrent::Promises.zip(*futures) is the higher-level ordered form; cap thread count for large N.
ZigZig
const std = @import("std");
// fixed pool: workers pull indexes from one atomic counter
fn parallelMap(
gpa: std.mem.Allocator,
comptime workers: usize,
items: []const u32,
out: []u32,
) !void {
var next = std.atomic.Value(usize).init(0);
const Worker = struct {
fn run(items: []const u32, out: []u32, next: *std.atomic.Value(usize)) void {
while (true) {
const i = next.fetchAdd(1, .monotonic); // claim one index
if (i >= items.len) return;
out[i] = transform(items[i]); // write OWN slot
}
}
};
const threads = try gpa.alloc(std.Thread, workers - 1);
defer gpa.free(threads);
for (threads) |*t|
t.* = try std.Thread.spawn(.{}, Worker.run, .{ items, out, &next });
Worker.run(items, out, &next); // the main thread is worker N
for (threads) |t| t.join(); // every worker has drained the counter
}The work queue is a single integer: fetchAdd claims an index atomically, so no two workers see the same i and out[i] = needs no lock — distinct slots. This fixed-pool shape scales to thousands of items; the spawn-per-item spelling (std.Thread.spawn inside the item loop) is fine for dozens, not thousands. Atomics are std.atomic.Value since Zig 0.12.