[{"data":1,"prerenderedAt":676},["ShallowReactive",2],{"navigation":3,"\u002Fresources\u002Fbenchmarks":174,"\u002Fresources\u002Fbenchmarks-surround":671},[4,8,101,165,170],{"title":5,"path":6,"stem":7},"Getting started","\u002Fgetting-started","1.getting-started",{"title":9,"path":10,"stem":11,"children":12},"Algorithms","\u002Falgorithms","2.algorithms",[13,15,58],{"title":9,"path":10,"stem":14},"2.algorithms\u002Findex",{"title":16,"path":17,"stem":18,"children":19},"Assignment","\u002Falgorithms\u002Fassignment","2.algorithms\u002F1.assignment\u002Findex",[20,21,25,29,32,36,40],{"title":16,"path":17,"stem":18},{"title":22,"path":23,"stem":24},"Quickstart","\u002Falgorithms\u002Fassignment\u002Fquickstart","2.algorithms\u002F1.assignment\u002F1.quickstart",{"title":26,"path":27,"stem":28},"Tracking","\u002Falgorithms\u002Fassignment\u002Ftracking","2.algorithms\u002F1.assignment\u002F2.tracking",{"title":9,"path":30,"stem":31},"\u002Falgorithms\u002Fassignment\u002Falgorithms","2.algorithms\u002F1.assignment\u002F3.algorithms",{"title":33,"path":34,"stem":35},"Reference","\u002Falgorithms\u002Fassignment\u002Freference","2.algorithms\u002F1.assignment\u002F4.reference",{"title":37,"path":38,"stem":39},"Choosing","\u002Falgorithms\u002Fassignment\u002Fchoosing","2.algorithms\u002F1.assignment\u002F5.choosing",{"title":41,"path":42,"stem":43,"children":44},"Tutorials","\u002Falgorithms\u002Fassignment\u002Ftutorials","2.algorithms\u002F1.assignment\u002F6.tutorials\u002Findex",[45,46,50,54],{"title":41,"path":42,"stem":43},{"title":47,"path":48,"stem":49},"Fundamentals","\u002Falgorithms\u002Fassignment\u002Ftutorials\u002Ffundamentals","2.algorithms\u002F1.assignment\u002F6.tutorials\u002F1.fundamentals",{"title":51,"path":52,"stem":53},"Backends","\u002Falgorithms\u002Fassignment\u002Ftutorials\u002Fbackends","2.algorithms\u002F1.assignment\u002F6.tutorials\u002F2.backends",{"title":55,"path":56,"stem":57},"Object tracking","\u002Falgorithms\u002Fassignment\u002Ftutorials\u002Ftracking","2.algorithms\u002F1.assignment\u002F6.tutorials\u002F3.tracking",{"title":59,"path":60,"stem":61,"children":62},"Transport","\u002Falgorithms\u002Ftransport","2.algorithms\u002F2.transport\u002Findex",[63,64,67,71,74,77,80,84],{"title":59,"path":60,"stem":61},{"title":22,"path":65,"stem":66},"\u002Falgorithms\u002Ftransport\u002Fquickstart","2.algorithms\u002F2.transport\u002F1.quickstart",{"title":68,"path":69,"stem":70},"Point-cloud tutorial","\u002Falgorithms\u002Ftransport\u002Fpoint-clouds","2.algorithms\u002F2.transport\u002F2.point-clouds",{"title":9,"path":72,"stem":73},"\u002Falgorithms\u002Ftransport\u002Falgorithms","2.algorithms\u002F2.transport\u002F3.algorithms",{"title":33,"path":75,"stem":76},"\u002Falgorithms\u002Ftransport\u002Freference","2.algorithms\u002F2.transport\u002F4.reference",{"title":37,"path":78,"stem":79},"\u002Falgorithms\u002Ftransport\u002Fchoosing","2.algorithms\u002F2.transport\u002F5.choosing",{"title":81,"path":82,"stem":83},"Building","\u002Falgorithms\u002Ftransport\u002Fbuilding","2.algorithms\u002F2.transport\u002F6.building",{"title":41,"path":85,"stem":86,"children":87},"\u002Falgorithms\u002Ftransport\u002Ftutorials","2.algorithms\u002F2.transport\u002F7.tutorials\u002Findex",[88,89,93,97],{"title":41,"path":85,"stem":86},{"title":90,"path":91,"stem":92},"Optimal transport","\u002Falgorithms\u002Ftransport\u002Ftutorials\u002Foptimal-transport","2.algorithms\u002F2.transport\u002F7.tutorials\u002F1.optimal-transport",{"title":94,"path":95,"stem":96},"Sinkhorn","\u002Falgorithms\u002Ftransport\u002Ftutorials\u002Fsinkhorn","2.algorithms\u002F2.transport\u002F7.tutorials\u002F2.sinkhorn",{"title":98,"path":99,"stem":100},"Point clouds","\u002Falgorithms\u002Ftransport\u002Ftutorials\u002Fpoint-clouds","2.algorithms\u002F2.transport\u002F7.tutorials\u002F3.point-clouds",{"title":102,"path":103,"stem":104,"children":105},"Resources","\u002Fresources","3.resources",[106,108,147,151,155],{"title":102,"path":103,"stem":107},"3.resources\u002Findex",{"title":41,"path":109,"stem":110,"children":111},"\u002Fresources\u002Ftutorials","3.resources\u002F1.tutorials\u002Findex",[112,113,131],{"title":41,"path":109,"stem":110},{"title":16,"path":114,"stem":115,"children":116,"page":130},"\u002Fresources\u002Ftutorials\u002Fassignment","3.resources\u002F1.tutorials\u002Fassignment",[117,122,126],{"title":118,"path":119,"stem":120,"icon":121},"Tutorial 1 — The Assignment Problem","\u002Fresources\u002Ftutorials\u002Fassignment\u002F01_the_assignment_problem","3.resources\u002F1.tutorials\u002Fassignment\u002F01_the_assignment_problem","i-lucide-notebook",{"title":123,"path":124,"stem":125,"icon":121},"Tutorial 2 — Backends and Batching","\u002Fresources\u002Ftutorials\u002Fassignment\u002F02_backends_and_batching","3.resources\u002F1.tutorials\u002Fassignment\u002F02_backends_and_batching",{"title":127,"path":128,"stem":129,"icon":121},"Tutorial 3 — Object Tracking with the Assignment Problem","\u002Fresources\u002Ftutorials\u002Fassignment\u002F03_object_tracking","3.resources\u002F1.tutorials\u002Fassignment\u002F03_object_tracking",false,{"title":59,"path":132,"stem":133,"children":134,"page":130},"\u002Fresources\u002Ftutorials\u002Ftransport","3.resources\u002F1.tutorials\u002Ftransport",[135,139,143],{"title":136,"path":137,"stem":138,"icon":121},"Tutorial 1 — What Is Optimal Transport?","\u002Fresources\u002Ftutorials\u002Ftransport\u002F01_optimal_transport","3.resources\u002F1.tutorials\u002Ftransport\u002F01_optimal_transport",{"title":140,"path":141,"stem":142,"icon":121},"Tutorial 2 — The Sinkhorn Algorithm","\u002Fresources\u002Ftutorials\u002Ftransport\u002F02_sinkhorn_algorithm","3.resources\u002F1.tutorials\u002Ftransport\u002F02_sinkhorn_algorithm",{"title":144,"path":145,"stem":146,"icon":121},"Tutorial 3 — Point-Cloud OT and Shape Learning","\u002Fresources\u002Ftutorials\u002Ftransport\u002F03_point_clouds","3.resources\u002F1.tutorials\u002Ftransport\u002F03_point_clouds",{"title":148,"path":149,"stem":150},"Assignment applications","\u002Fresources\u002Fassignment-applications","3.resources\u002F2.assignment-applications",{"title":152,"path":153,"stem":154},"Transport applications","\u002Fresources\u002Ftransport-applications","3.resources\u002F3.transport-applications",{"title":156,"path":157,"stem":158,"children":159},"Benchmarks","\u002Fresources\u002Fbenchmarks","3.resources\u002F4.benchmarks\u002Findex",[160,161],{"title":156,"path":157,"stem":158},{"title":162,"path":163,"stem":164},"Contributing benchmarks","\u002Fresources\u002Fbenchmarks\u002Fcontributing","3.resources\u002F4.benchmarks\u002Fcontributing",{"title":166,"path":167,"stem":168,"icon":169},"API Reference","\u002Fapi","4.api","i-lucide-package",{"title":171,"path":172,"stem":173},"References","\u002Freferences","5.references",{"id":175,"title":156,"api":176,"body":177,"description":666,"extension":667,"links":176,"meta":668,"navigation":176,"path":157,"seo":669,"stem":158,"__hash__":670},"docs\u002F3.resources\u002F4.benchmarks\u002Findex.md",null,{"type":178,"value":179,"toc":648},"minimark",[180,184,207,210,217,222,225,229,232,310,314,317,382,389,393,492,496,502,510,514,518,522,525,529,534,537,541,546,549,553,559,564,567,571,575,578,582,591,596,600,604,607,624,634,638],[181,182,183],"p",{},"This page aggregates pytest-benchmark runs contributed by users. The\nsuite covers three files:",[185,186,187,195,201],"ul",{},[188,189,190,194],"li",{},[191,192,193],"code",{},"tests\u002Fbenchmark_single.py"," — single-problem assignment ops",[188,196,197,200],{},[191,198,199],{},"tests\u002Fbenchmark_batched.py"," — batched assignment ops",[188,202,203,206],{},[191,204,205],{},"tests\u002Fbenchmark_transport.py"," — transport ops (matrix and samples faces)",[181,208,209],{},"Numbers differ across hardware. Rankings within a machine are stable;\nabsolute timings are hardware-specific.",[181,211,212,213,216],{},"To submit your own run, see ",[214,215,162],"a",{"href":163},".",[218,219,221],"h2",{"id":220},"overview","Overview",[223,224],"bench-explorer",{},[218,226,228],{"id":227},"assignment-methodology","Assignment methodology",[181,230,231],{},"The assignment suite parametrizes over:",[185,233,234,263,269,275,281],{},[188,235,236,240,241,244,245,244,248,244,251,254,255,258,259,262],{},[237,238,239],"strong",{},"Cost distributions",": ",[191,242,243],{},"uniform",", ",[191,246,247],{},"gamma",[191,249,250],{},"iou",[191,252,253],{},"gated_sparse",",\n",[191,256,257],{},"integer_tied",". See ",[191,260,261],{},"tests\u002F_costgen.py"," for the generators.",[188,264,265,268],{},[237,266,267],{},"Problem sizes",": N ∈ {16, 64, 256, 1024} for single-problem ops;\nN ∈ {16, 64, 128} for batched CPU; N ∈ {8, 32, 64} for batched CUDA\n(the tiled kernel caps at K = 64 (K is the padded square problem size; the CUDA kernel uses shared memory sized for at most 64 rows and 64 columns)).",[188,270,271,274],{},[237,272,273],{},"Batch sizes",": B ∈ {16, 64} for batched ops.",[188,276,277,280],{},[237,278,279],{},"Dtypes",": float32 and float64.",[188,282,283,286,287,244,290,244,293,296,297,300,301,244,304,307,308,216],{},[237,284,285],{},"Devices",": CPU for the Jonker-Volgenant (JV) family (",[191,288,289],{},"jonker_scalar",[191,291,292],{},"jonker_dense",[191,294,295],{},"jonker_compact",", and the CPU backend of ",[191,298,299],{},"jonker_dense_batch","); CUDA for ",[191,302,303],{},"munkres",[191,305,306],{},"lawler",", and the CUDA backend of ",[191,309,299],{},[218,311,313],{"id":312},"transport-methodology","Transport methodology",[181,315,316],{},"The transport suite parametrizes over:",[185,318,319,337,345,351,357,371,376],{},[188,320,321,323,324,244,327,254,330,244,333,336],{},[237,322,51],{}," (matrix face): ",[191,325,326],{},"log_sinkhorn",[191,328,329],{},"sinkhorn_divergence",[191,331,332],{},"unbalanced_sinkhorn",[191,334,335],{},"exact_emd",". Each backend receives a uniform\nrandom cost matrix; the transport algorithm does not branch on cost\ndistribution, so one distribution is sufficient.",[188,338,339,341,342,344],{},[237,340,267],{}," (matrix face): N ∈ {64, 256, 1024}. ",[191,343,335],{}," is\nskipped above N = 128 (network-simplex worst case is O((N+M)³ log(N+M))).",[188,346,347,350],{},[237,348,349],{},"Point-cloud size"," (samples face): N ∈ {256, 1024, 4096} points.",[188,352,353,356],{},[237,354,355],{},"Feature dimension"," (samples face): D ∈ {3, 64} — 3D shape coordinates\nand feature-space clouds.",[188,358,359,362,363,366,367,370],{},[237,360,361],{},"Modes"," (samples face): ",[191,364,365],{},"samples_loss"," (standard Sinkhorn, one solver call) and\n",[191,368,369],{},"samples_loss_debias"," (Sinkhorn divergence, which removes entropic bias by combining three Sinkhorn solves).",[188,372,373,375],{},[237,374,279],{}," (matrix face): float32 and float64. The samples face operates\nin float32 (the Triton kernel precision).",[188,377,378,381],{},[191,379,380],{},"n_iter"," is fixed at 100 for all Sinkhorn backends.",[181,383,384,385,388],{},"All timed cases run through ",[191,386,387],{},"pytest-benchmark"," (warm-up + multiple rounds,\nmedian reported). CUDA cases synchronize the stream inside the timed region.",[218,390,392],{"id":391},"distributions-tested","Distributions tested",[394,395,396,412],"table",{},[397,398,399],"thead",{},[400,401,402,406,409],"tr",{},[403,404,405],"th",{},"Name",[403,407,408],{},"Description",[403,410,411],{},"Why it matters",[413,414,415,430,445,460,476],"tbody",{},[400,416,417,422,427],{},[418,419,420],"td",{},[191,421,243],{},[418,423,424],{},[191,425,426],{},"cost ~ U(0, 1)",[418,428,429],{},"Algorithmic baseline",[400,431,432,436,442],{},[418,433,434],{},[191,435,247],{},[418,437,438,441],{},[191,439,440],{},"cost ~ Gamma(2, 0.5)"," (long tail to ~5)",[418,443,444],{},"Score-like ML pipeline costs",[400,446,447,451,457],{},[418,448,449],{},[191,450,250],{},[418,452,453,456],{},[191,454,455],{},"1 - IoU(box_i, box_j)"," for random axis-aligned bounding boxes (AABBs)",[418,458,459],{},"Realistic for assigning detected objects to tracked objects across frames",[400,461,462,466,473],{},[418,463,464],{},[191,465,253],{},[418,467,468,469,472],{},"~70 % ",[191,470,471],{},"+inf"," (forbidden pairs), with at least one feasible perfect matching guaranteed",[418,474,475],{},"Models tracker outputs after a distance-gating step rejects implausible pairings",[400,477,478,482,489],{},[418,479,480],{},[191,481,257],{},[418,483,484,485,488],{},"Random integers in ",[191,486,487],{},"{0, ..., 7}"," cast to float",[418,490,491],{},"Stresses tie-breaking; quantized cost workloads",[218,493,495],{"id":494},"single-problem-cpu-latency","Single-problem CPU latency",[181,497,498,499,501],{},"Median across ",[191,500,243],{}," cost on float32. Use the machine picker above\nto restrict the dataset to one machine.",[503,504],"bench-chart",{"filter":505,"group":506,"series":507,"title":508,"x":509},"{\"dtype\":\"f32\",\"dist\":\"uniform\"}","single-cpu","op","single-cpu, f32, uniform","n",[511,512],"bench-table",{"col":507,"filter":505,"group":506,"row":509,"title":513},"single-cpu, f32, uniform (median)",[218,515,517],{"id":516},"single-problem-cuda-latency","Single-problem CUDA latency",[503,519],{"filter":505,"group":520,"series":507,"title":521,"x":509},"single-cuda","single-cuda, f32, uniform",[511,523],{"col":507,"filter":505,"group":520,"row":509,"title":524},"single-cuda, f32, uniform (median)",[218,526,528],{"id":527},"batched-cpu-latency","Batched CPU latency",[503,530],{"filter":531,"group":532,"series":507,"title":533,"x":509},"{\"b\":16,\"dtype\":\"f32\",\"dist\":\"uniform\"}","batch-cpu","batch-cpu, B=16, f32, uniform",[511,535],{"col":507,"filter":531,"group":532,"row":509,"title":536},"batch-cpu, B=16, f32, uniform (median)",[218,538,540],{"id":539},"batched-cuda-latency","Batched CUDA latency",[503,542],{"filter":505,"group":543,"series":544,"title":545,"x":509},"batch-cuda","b","batch-cuda jonker_dense_batch, f32, uniform",[511,547],{"col":544,"filter":505,"group":543,"row":509,"title":548},"batch-cuda, f32, uniform (median, B columns)",[218,550,552],{"id":551},"transport-matrix-face-cpu-latency","Transport matrix-face CPU latency",[181,554,555,556,558],{},"Median across float32, uniform cost. Compare Sinkhorn backends across\nproblem sizes; ",[191,557,335],{}," only appears at N ≤ 128.",[503,560],{"filter":561,"group":562,"series":507,"title":563,"x":509},"{\"dtype\":\"f32\"}","transport-matrix-cpu","transport matrix CPU, f32",[511,565],{"col":507,"filter":561,"group":562,"row":509,"title":566},"transport matrix CPU, f32 (median)",[218,568,570],{"id":569},"transport-matrix-face-cuda-latency","Transport matrix-face CUDA latency",[503,572],{"filter":561,"group":573,"series":507,"title":574,"x":509},"transport-matrix-cuda","transport matrix CUDA, f32",[511,576],{"col":507,"filter":561,"group":573,"row":509,"title":577},"transport matrix CUDA, f32 (median)",[218,579,581],{"id":580},"transport-samples-face-cuda-latency","Transport samples-face CUDA latency",[181,583,584,585,587,588,590],{},"Point-cloud optimal transport (OT) via the Triton streaming kernel. D = 3 (3D shapes) and\nD = 64 (feature-space clouds). ",[191,586,369],{}," runs three forward\npasses; the ratio to ",[191,589,365],{}," is roughly 3× for large N but\nslightly more at small N due to fixed overhead.",[503,592],{"filter":593,"group":594,"series":507,"title":595,"x":509},"{\"dim\":3}","transport-samples-cuda","transport samples CUDA, D=3",[503,597],{"filter":598,"group":594,"series":507,"title":599,"x":509},"{\"dim\":64}","transport samples CUDA, D=64",[218,601,603],{"id":602},"assignment-caveats","Assignment caveats",[181,605,606],{},"The assignment sweep covers common Linear Assignment Problem (LAP) cost regimes but not:",[185,608,609,612,615,621],{},[188,610,611],{},"Mahalanobis-distance costs (covariance-weighted distances produced by Kalman-filter trackers, common in SORT-style tracking).",[188,613,614],{},"Cosine-similarity costs in high-dimensional re-identification (re-ID) embeddings.",[188,616,617,618,620],{},"Truly sparse problems (≥ 95 % ",[191,619,471],{},").",[188,622,623],{},"Very small problems (N = 2 to 8), where launch overhead may flip the ranking.",[181,625,626,627,629,630,633],{},"Extend ",[191,628,261],{}," and re-run for out-of-distribution workloads. See\n",[214,631,632],{"href":38},"Choosing the right op"," for the decision tree.",[218,635,637],{"id":636},"transport-caveats","Transport caveats",[181,639,640,641,643,644,647],{},"The transport sweep uses uniform random cost matrices. Sinkhorn convergence\nspeed depends on cost structure: smooth, low-variance costs converge faster\n(fewer effective iterations) than costs with a very wide range of values or near-uniform rows\u002Fcolumns, which require more iterations to reach a good transport plan.\nThe 100-iteration fixed ",[191,642,380],{}," is conservative for typical use. See\n",[214,645,646],{"href":78},"Choosing the right backend"," for guidance.",{"title":649,"searchDepth":650,"depth":650,"links":651},"",3,[652,654,655,656,657,658,659,660,661,662,663,664,665],{"id":220,"depth":653,"text":221},2,{"id":227,"depth":653,"text":228},{"id":312,"depth":653,"text":313},{"id":391,"depth":653,"text":392},{"id":494,"depth":653,"text":495},{"id":516,"depth":653,"text":517},{"id":527,"depth":653,"text":528},{"id":539,"depth":653,"text":540},{"id":551,"depth":653,"text":552},{"id":569,"depth":653,"text":570},{"id":580,"depth":653,"text":581},{"id":602,"depth":653,"text":603},{"id":636,"depth":653,"text":637},"Per-op benchmark sweep across problem sizes, dtypes, and devices — covering assignment (single-problem, batched) and transport (matrix-face Sinkhorn\u002FEMD, samples-face Triton) ops.","md",{},{"title":156,"description":666},"6d5EVVNIzz3JC7Jky5euLMQOseWyxoBTY2bz8cuCT5M",[672,674],{"title":152,"path":153,"stem":154,"description":673,"children":-1},"Historical and modern applications of optimal transport — from Monge's earth-moving problem through Wasserstein GANs, single-cell genomics, domain adaptation, and geometric deep learning.",{"title":162,"path":163,"stem":164,"description":675,"children":-1},"Run the torchmatch benchmark suite on your hardware and submit the result via PR.",1785218159644]