[{"data":1,"prerenderedAt":1522},["ShallowReactive",2],{"navigation":3,"\u002Falgorithms\u002Fassignment\u002Ftutorials\u002Fbackends":174,"\u002Falgorithms\u002Fassignment\u002Ftutorials\u002Fbackends-surround":1517},[4,8,101,165,170],{"title":5,"path":6,"stem":7},"Getting started","\u002Fgetting-started","1.getting-started",{"title":9,"path":10,"stem":11,"children":12},"Algorithms","\u002Falgorithms","2.algorithms",[13,15,58],{"title":9,"path":10,"stem":14},"2.algorithms\u002Findex",{"title":16,"path":17,"stem":18,"children":19},"Assignment","\u002Falgorithms\u002Fassignment","2.algorithms\u002F1.assignment\u002Findex",[20,21,25,29,32,36,40],{"title":16,"path":17,"stem":18},{"title":22,"path":23,"stem":24},"Quickstart","\u002Falgorithms\u002Fassignment\u002Fquickstart","2.algorithms\u002F1.assignment\u002F1.quickstart",{"title":26,"path":27,"stem":28},"Tracking","\u002Falgorithms\u002Fassignment\u002Ftracking","2.algorithms\u002F1.assignment\u002F2.tracking",{"title":9,"path":30,"stem":31},"\u002Falgorithms\u002Fassignment\u002Falgorithms","2.algorithms\u002F1.assignment\u002F3.algorithms",{"title":33,"path":34,"stem":35},"Reference","\u002Falgorithms\u002Fassignment\u002Freference","2.algorithms\u002F1.assignment\u002F4.reference",{"title":37,"path":38,"stem":39},"Choosing","\u002Falgorithms\u002Fassignment\u002Fchoosing","2.algorithms\u002F1.assignment\u002F5.choosing",{"title":41,"path":42,"stem":43,"children":44},"Tutorials","\u002Falgorithms\u002Fassignment\u002Ftutorials","2.algorithms\u002F1.assignment\u002F6.tutorials\u002Findex",[45,46,50,54],{"title":41,"path":42,"stem":43},{"title":47,"path":48,"stem":49},"Fundamentals","\u002Falgorithms\u002Fassignment\u002Ftutorials\u002Ffundamentals","2.algorithms\u002F1.assignment\u002F6.tutorials\u002F1.fundamentals",{"title":51,"path":52,"stem":53},"Backends","\u002Falgorithms\u002Fassignment\u002Ftutorials\u002Fbackends","2.algorithms\u002F1.assignment\u002F6.tutorials\u002F2.backends",{"title":55,"path":56,"stem":57},"Object tracking","\u002Falgorithms\u002Fassignment\u002Ftutorials\u002Ftracking","2.algorithms\u002F1.assignment\u002F6.tutorials\u002F3.tracking",{"title":59,"path":60,"stem":61,"children":62},"Transport","\u002Falgorithms\u002Ftransport","2.algorithms\u002F2.transport\u002Findex",[63,64,67,71,74,77,80,84],{"title":59,"path":60,"stem":61},{"title":22,"path":65,"stem":66},"\u002Falgorithms\u002Ftransport\u002Fquickstart","2.algorithms\u002F2.transport\u002F1.quickstart",{"title":68,"path":69,"stem":70},"Point-cloud tutorial","\u002Falgorithms\u002Ftransport\u002Fpoint-clouds","2.algorithms\u002F2.transport\u002F2.point-clouds",{"title":9,"path":72,"stem":73},"\u002Falgorithms\u002Ftransport\u002Falgorithms","2.algorithms\u002F2.transport\u002F3.algorithms",{"title":33,"path":75,"stem":76},"\u002Falgorithms\u002Ftransport\u002Freference","2.algorithms\u002F2.transport\u002F4.reference",{"title":37,"path":78,"stem":79},"\u002Falgorithms\u002Ftransport\u002Fchoosing","2.algorithms\u002F2.transport\u002F5.choosing",{"title":81,"path":82,"stem":83},"Building","\u002Falgorithms\u002Ftransport\u002Fbuilding","2.algorithms\u002F2.transport\u002F6.building",{"title":41,"path":85,"stem":86,"children":87},"\u002Falgorithms\u002Ftransport\u002Ftutorials","2.algorithms\u002F2.transport\u002F7.tutorials\u002Findex",[88,89,93,97],{"title":41,"path":85,"stem":86},{"title":90,"path":91,"stem":92},"Optimal transport","\u002Falgorithms\u002Ftransport\u002Ftutorials\u002Foptimal-transport","2.algorithms\u002F2.transport\u002F7.tutorials\u002F1.optimal-transport",{"title":94,"path":95,"stem":96},"Sinkhorn","\u002Falgorithms\u002Ftransport\u002Ftutorials\u002Fsinkhorn","2.algorithms\u002F2.transport\u002F7.tutorials\u002F2.sinkhorn",{"title":98,"path":99,"stem":100},"Point clouds","\u002Falgorithms\u002Ftransport\u002Ftutorials\u002Fpoint-clouds","2.algorithms\u002F2.transport\u002F7.tutorials\u002F3.point-clouds",{"title":102,"path":103,"stem":104,"children":105},"Resources","\u002Fresources","3.resources",[106,108,147,151,155],{"title":102,"path":103,"stem":107},"3.resources\u002Findex",{"title":41,"path":109,"stem":110,"children":111},"\u002Fresources\u002Ftutorials","3.resources\u002F1.tutorials\u002Findex",[112,113,131],{"title":41,"path":109,"stem":110},{"title":16,"path":114,"stem":115,"children":116,"page":130},"\u002Fresources\u002Ftutorials\u002Fassignment","3.resources\u002F1.tutorials\u002Fassignment",[117,122,126],{"title":118,"path":119,"stem":120,"icon":121},"Tutorial 1 — The Assignment Problem","\u002Fresources\u002Ftutorials\u002Fassignment\u002F01_the_assignment_problem","3.resources\u002F1.tutorials\u002Fassignment\u002F01_the_assignment_problem","i-lucide-notebook",{"title":123,"path":124,"stem":125,"icon":121},"Tutorial 2 — Backends and Batching","\u002Fresources\u002Ftutorials\u002Fassignment\u002F02_backends_and_batching","3.resources\u002F1.tutorials\u002Fassignment\u002F02_backends_and_batching",{"title":127,"path":128,"stem":129,"icon":121},"Tutorial 3 — Object Tracking with the Assignment Problem","\u002Fresources\u002Ftutorials\u002Fassignment\u002F03_object_tracking","3.resources\u002F1.tutorials\u002Fassignment\u002F03_object_tracking",false,{"title":59,"path":132,"stem":133,"children":134,"page":130},"\u002Fresources\u002Ftutorials\u002Ftransport","3.resources\u002F1.tutorials\u002Ftransport",[135,139,143],{"title":136,"path":137,"stem":138,"icon":121},"Tutorial 1 — What Is Optimal Transport?","\u002Fresources\u002Ftutorials\u002Ftransport\u002F01_optimal_transport","3.resources\u002F1.tutorials\u002Ftransport\u002F01_optimal_transport",{"title":140,"path":141,"stem":142,"icon":121},"Tutorial 2 — The Sinkhorn Algorithm","\u002Fresources\u002Ftutorials\u002Ftransport\u002F02_sinkhorn_algorithm","3.resources\u002F1.tutorials\u002Ftransport\u002F02_sinkhorn_algorithm",{"title":144,"path":145,"stem":146,"icon":121},"Tutorial 3 — Point-Cloud OT and Shape Learning","\u002Fresources\u002Ftutorials\u002Ftransport\u002F03_point_clouds","3.resources\u002F1.tutorials\u002Ftransport\u002F03_point_clouds",{"title":148,"path":149,"stem":150},"Assignment applications","\u002Fresources\u002Fassignment-applications","3.resources\u002F2.assignment-applications",{"title":152,"path":153,"stem":154},"Transport applications","\u002Fresources\u002Ftransport-applications","3.resources\u002F3.transport-applications",{"title":156,"path":157,"stem":158,"children":159},"Benchmarks","\u002Fresources\u002Fbenchmarks","3.resources\u002F4.benchmarks\u002Findex",[160,161],{"title":156,"path":157,"stem":158},{"title":162,"path":163,"stem":164},"Contributing benchmarks","\u002Fresources\u002Fbenchmarks\u002Fcontributing","3.resources\u002F4.benchmarks\u002Fcontributing",{"title":166,"path":167,"stem":168,"icon":169},"API Reference","\u002Fapi","4.api","i-lucide-package",{"title":171,"path":172,"stem":173},"References","\u002Freferences","5.references",{"id":175,"title":176,"api":177,"body":178,"description":1511,"extension":1512,"links":177,"meta":1513,"navigation":1514,"path":52,"seo":1515,"stem":53,"__hash__":1516},"docs\u002F2.algorithms\u002F1.assignment\u002F6.tutorials\u002F2.backends.md","Backends and Batching",null,{"type":179,"value":180,"toc":1500},"minimark",[181,186,202,209,213,216,260,570,574,577,609,612,616,701,705,712,826,829,833,839,962,972,1063,1070,1156,1160,1187,1449,1456,1460,1479,1483,1496],[182,183,185],"h2",{"id":184},"what-auto-does","What AUTO does",[187,188,189,193,194,197,198,201],"p",{},[190,191,192],"code",{},"torchmatch.assignment.solve(cost)"," defaults to ",[190,195,196],{},"backend=Backend.AUTO",". At call time, it inspects the device, shape, and whether the matrix is square, then selects the fastest available op and dispatches. The resolution happens in Python at call time, not at graph-capture time, so the chosen op traces cleanly under ",[190,199,200],{},"torch.compile",".",[187,203,204,205,208],{},"You do not need to think about backends for everyday use. This tutorial covers them so you understand what ",[190,206,207],{},"solve"," is selecting, and to show when pinning a specific backend is worthwhile.",[182,210,212],{"id":211},"the-three-cpu-backends","The three CPU backends",[187,214,215],{},"All three CPU backends implement the Jonker-Volgenant (JV) successive shortest-path algorithm. They differ in their inner loop:",[217,218,219,233,245],"ul",{},[220,221,222,228,229,232],"li",{},[223,224,225],"strong",{},[190,226,227],{},"jonker_scalar"," is the sequential reference implementation, with no SIMD. It is the most portable: it runs correctly on any CPU, including those without AVX2. AUTO selects it when the problem is tiny (",[190,230,231],{},"N * M \u003C= 64","), where the overhead of setting up SIMD outweighs the benefit.",[220,234,235,240,241,244],{},[223,236,237],{},[190,238,239],{},"jonker_dense"," uses AVX2 vector instructions over a flat contiguous memory layout. It handles rectangular matrices natively (the underlying algorithm from Crouse 2016 supports ",[190,242,243],{},"n != m"," without padding). AUTO picks this as the default for any CPU problem above the scalar threshold.",[220,246,247,252,253,256,257,259],{},[223,248,249],{},[190,250,251],{},"jonker_compact"," also uses AVX2 but with a tighter address pattern suited to square problems. On square matrices at ",[190,254,255],{},"N \u003C= 256"," with smoothly distributed costs (uniform random, gamma-distributed scores, L2 distances), it runs 20 to 30 percent faster than ",[190,258,239],{},". On IoU costs or gated-sparse matrices it falls behind. It is square-only; calling it on a rectangular matrix pads internally.",[261,262,267],"pre",{"className":263,"code":264,"language":265,"meta":266,"style":266},"language-python shiki shiki-themes material-theme-lighter github-light github-dark","import torch\nimport torchmatch\n\ntorch.manual_seed(0)\ncost = torch.rand(64, 64, dtype=torch.float64)\n\nfor name in (\"jonker_scalar\", \"jonker_dense\", \"jonker_compact\"):\n    op = getattr(torchmatch.assignment.ops, name)\n    out = op(cost)\n    total = cost[torch.arange(64), out].sum().item()\n    print(f\"{name:20s} total={total:.4f}\")\n# All three produce the same optimal total; tied costs may yield different\n# but equally optimal assignments.\n","python","",[190,268,269,282,290,297,320,367,372,416,450,468,514,557,564],{"__ignoreMap":266},[270,271,274,278],"span",{"class":272,"line":273},"line",1,[270,275,277],{"class":276},"sVHd0","import",[270,279,281],{"class":280},"su5hD"," torch\n",[270,283,285,287],{"class":272,"line":284},2,[270,286,277],{"class":276},[270,288,289],{"class":280}," torchmatch\n",[270,291,293],{"class":272,"line":292},3,[270,294,296],{"emptyLinePlaceholder":295},true,"\n",[270,298,300,303,306,310,313,317],{"class":272,"line":299},4,[270,301,302],{"class":280},"torch",[270,304,201],{"class":305},"sP7_E",[270,307,309],{"class":308},"slqww","manual_seed",[270,311,312],{"class":305},"(",[270,314,316],{"class":315},"srdBf","0",[270,318,319],{"class":305},")\n",[270,321,323,326,330,333,335,338,340,343,346,349,351,355,357,359,361,365],{"class":272,"line":322},5,[270,324,325],{"class":280},"cost ",[270,327,329],{"class":328},"smGrS","=",[270,331,332],{"class":280}," torch",[270,334,201],{"class":305},[270,336,337],{"class":308},"rand",[270,339,312],{"class":305},[270,341,342],{"class":315},"64",[270,344,345],{"class":305},",",[270,347,348],{"class":315}," 64",[270,350,345],{"class":305},[270,352,354],{"class":353},"s99_P"," dtype",[270,356,329],{"class":328},[270,358,302],{"class":308},[270,360,201],{"class":305},[270,362,364],{"class":363},"skxfh","float64",[270,366,319],{"class":305},[270,368,370],{"class":272,"line":369},6,[270,371,296],{"emptyLinePlaceholder":295},[270,373,375,378,381,384,387,391,394,396,398,401,403,405,407,409,411,413],{"class":272,"line":374},7,[270,376,377],{"class":276},"for",[270,379,380],{"class":280}," name ",[270,382,383],{"class":276},"in",[270,385,386],{"class":305}," (",[270,388,390],{"class":389},"sjJ54","\"",[270,392,227],{"class":393},"s_sjI",[270,395,390],{"class":389},[270,397,345],{"class":305},[270,399,400],{"class":389}," \"",[270,402,239],{"class":393},[270,404,390],{"class":389},[270,406,345],{"class":305},[270,408,400],{"class":389},[270,410,251],{"class":393},[270,412,390],{"class":389},[270,414,415],{"class":305},"):\n",[270,417,419,422,424,428,430,433,435,438,440,443,445,448],{"class":272,"line":418},8,[270,420,421],{"class":280},"    op ",[270,423,329],{"class":328},[270,425,427],{"class":426},"sptTA"," getattr",[270,429,312],{"class":305},[270,431,432],{"class":308},"torchmatch",[270,434,201],{"class":305},[270,436,437],{"class":363},"assignment",[270,439,201],{"class":305},[270,441,442],{"class":363},"ops",[270,444,345],{"class":305},[270,446,447],{"class":308}," name",[270,449,319],{"class":305},[270,451,453,456,458,461,463,466],{"class":272,"line":452},9,[270,454,455],{"class":280},"    out ",[270,457,329],{"class":328},[270,459,460],{"class":308}," op",[270,462,312],{"class":305},[270,464,465],{"class":308},"cost",[270,467,319],{"class":305},[270,469,471,474,476,479,482,484,486,489,491,493,496,499,502,505,508,511],{"class":272,"line":470},10,[270,472,473],{"class":280},"    total ",[270,475,329],{"class":328},[270,477,478],{"class":280}," cost",[270,480,481],{"class":305},"[",[270,483,302],{"class":280},[270,485,201],{"class":305},[270,487,488],{"class":308},"arange",[270,490,312],{"class":305},[270,492,342],{"class":315},[270,494,495],{"class":305},"),",[270,497,498],{"class":280}," out",[270,500,501],{"class":305},"].",[270,503,504],{"class":308},"sum",[270,506,507],{"class":305},"().",[270,509,510],{"class":308},"item",[270,512,513],{"class":305},"()\n",[270,515,517,520,522,526,528,531,534,537,540,543,545,548,551,553,555],{"class":272,"line":516},11,[270,518,519],{"class":426},"    print",[270,521,312],{"class":305},[270,523,525],{"class":524},"sbsja","f",[270,527,390],{"class":393},[270,529,530],{"class":315},"{",[270,532,533],{"class":308},"name",[270,535,536],{"class":524},":20s",[270,538,539],{"class":315},"}",[270,541,542],{"class":393}," total=",[270,544,530],{"class":315},[270,546,547],{"class":308},"total",[270,549,550],{"class":524},":.4f",[270,552,539],{"class":315},[270,554,390],{"class":393},[270,556,319],{"class":305},[270,558,560],{"class":272,"line":559},12,[270,561,563],{"class":562},"sutJx","# All three produce the same optimal total; tied costs may yield different\n",[270,565,567],{"class":272,"line":566},13,[270,568,569],{"class":562},"# but equally optimal assignments.\n",[182,571,573],{"id":572},"the-cuda-backends","The CUDA backends",[187,575,576],{},"Two CUDA backends implement a different algorithm family: the Hungarian primal-dual method with starred and primed zeros.",[217,578,579,591],{},[220,580,581,586,587,590],{},[223,582,583],{},[190,584,585],{},"munkres"," follows Munkres' 1957 single-path procedure. It is fastest for small problems (",[190,588,589],{},"N \u003C 32",") and for integer-tied cost matrices, where many cells share the same value and augmenting paths are short. It is also the only CUDA op that beats the CPU JV variants, specifically on integer-tied costs at large N.",[220,592,593,598,599,601,602,604,605,608],{},[223,594,595],{},[190,596,597],{},"lawler"," uses Lawler's 1976 tree-augmentation variant, which finds all vertex-disjoint augmenting paths in a single breadth-first pass instead of one at a time. This exposes more GPU parallelism and makes ",[190,600,597],{}," faster than ",[190,603,585],{}," for large, fully-populated cost matrices (",[190,606,607],{},"N >= 512",").",[187,610,611],{},"Both require the cost tensor on a CUDA device. Neither supports CUDA graph capture, because they synchronize with the CPU between iterations; for graph-safe batched solving, see the next section.",[182,613,615],{"id":614},"quick-decision-table","Quick decision table",[617,618,619,632],"table",{},[620,621,622],"thead",{},[623,624,625,629],"tr",{},[626,627,628],"th",{},"Situation",[626,630,631],{},"Best op",[633,634,635,646,658,667,679,690],"tbody",{},[623,636,637,641],{},[638,639,640],"td",{},"CPU, any shape, default choice",[638,642,643,645],{},[190,644,239],{}," (AUTO picks this)",[623,647,648,654],{},[638,649,650,651,653],{},"CPU, square ",[190,652,255],{},", smooth costs",[638,655,656],{},[190,657,251],{},[623,659,660,663],{},[638,661,662],{},"CPU, no AVX2 instruction set",[638,664,665],{},[190,666,227],{},[623,668,669,675],{},[638,670,671,672,674],{},"CUDA, ",[190,673,589],{}," or integer-tied costs",[638,676,677],{},[190,678,585],{},[623,680,681,686],{},[638,682,671,683,685],{},[190,684,607],{},", dense costs",[638,687,688],{},[190,689,597],{},[623,691,692,696],{},[638,693,671,694,685],{},[190,695,255],{},[638,697,698,700],{},[190,699,585],{}," leads on GPU, but CPU JV is 10 to 100x faster",[182,702,704],{"id":703},"pinning-a-backend-directly","Pinning a backend directly",[187,706,707,708,711],{},"To skip AUTO and call a specific op, use ",[190,709,710],{},"torchmatch.assignment.ops",":",[261,713,715],{"className":263,"code":714,"language":265,"meta":266,"style":266},"cost = torch.rand(128, 128)\nrow_to_col = torchmatch.assignment.ops.jonker_dense(cost)\n\ncost_gpu = cost.to(\"cuda\")\nrow_to_col_gpu = torchmatch.assignment.ops.lawler(cost_gpu)\n",[190,716,717,741,769,773,798],{"__ignoreMap":266},[270,718,719,721,723,725,727,729,731,734,736,739],{"class":272,"line":273},[270,720,325],{"class":280},[270,722,329],{"class":328},[270,724,332],{"class":280},[270,726,201],{"class":305},[270,728,337],{"class":308},[270,730,312],{"class":305},[270,732,733],{"class":315},"128",[270,735,345],{"class":305},[270,737,738],{"class":315}," 128",[270,740,319],{"class":305},[270,742,743,746,748,751,753,755,757,759,761,763,765,767],{"class":272,"line":284},[270,744,745],{"class":280},"row_to_col ",[270,747,329],{"class":328},[270,749,750],{"class":280}," torchmatch",[270,752,201],{"class":305},[270,754,437],{"class":363},[270,756,201],{"class":305},[270,758,442],{"class":363},[270,760,201],{"class":305},[270,762,239],{"class":308},[270,764,312],{"class":305},[270,766,465],{"class":308},[270,768,319],{"class":305},[270,770,771],{"class":272,"line":292},[270,772,296],{"emptyLinePlaceholder":295},[270,774,775,778,780,782,784,787,789,791,794,796],{"class":272,"line":299},[270,776,777],{"class":280},"cost_gpu ",[270,779,329],{"class":328},[270,781,478],{"class":280},[270,783,201],{"class":305},[270,785,786],{"class":308},"to",[270,788,312],{"class":305},[270,790,390],{"class":389},[270,792,793],{"class":393},"cuda",[270,795,390],{"class":389},[270,797,319],{"class":305},[270,799,800,803,805,807,809,811,813,815,817,819,821,824],{"class":272,"line":322},[270,801,802],{"class":280},"row_to_col_gpu ",[270,804,329],{"class":328},[270,806,750],{"class":280},[270,808,201],{"class":305},[270,810,437],{"class":363},[270,812,201],{"class":305},[270,814,442],{"class":363},[270,816,201],{"class":305},[270,818,597],{"class":308},[270,820,312],{"class":305},[270,822,823],{"class":308},"cost_gpu",[270,825,319],{"class":305},[187,827,828],{},"Pinning is useful for benchmarking and for making the backend choice visible in the source.",[182,830,832],{"id":831},"batched-solving-with-a-3-d-tensor","Batched solving with a 3-D tensor",[187,834,835,836,711],{},"Many workloads solve many independent assignment problems in parallel (one per frame in a video, one per anchor in a detector head). Pass a 3-D tensor of shape ",[190,837,838],{},"(B, N, M)",[261,840,842],{"className":263,"code":841,"language":265,"meta":266,"style":266},"torch.manual_seed(42)\ncosts = torch.rand(32, 48, 48)       # 32 independent 48x48 problems\nassignments = torchmatch.assignment.solve(costs)\nprint(assignments.shape)             # torch.Size([32, 48])\nprint((assignments >= 0).all())      # True: all rows matched (square)\n",[190,843,844,859,892,916,936],{"__ignoreMap":266},[270,845,846,848,850,852,854,857],{"class":272,"line":273},[270,847,302],{"class":280},[270,849,201],{"class":305},[270,851,309],{"class":308},[270,853,312],{"class":305},[270,855,856],{"class":315},"42",[270,858,319],{"class":305},[270,860,861,864,866,868,870,872,874,877,879,882,884,886,889],{"class":272,"line":284},[270,862,863],{"class":280},"costs ",[270,865,329],{"class":328},[270,867,332],{"class":280},[270,869,201],{"class":305},[270,871,337],{"class":308},[270,873,312],{"class":305},[270,875,876],{"class":315},"32",[270,878,345],{"class":305},[270,880,881],{"class":315}," 48",[270,883,345],{"class":305},[270,885,881],{"class":315},[270,887,888],{"class":305},")",[270,890,891],{"class":562},"       # 32 independent 48x48 problems\n",[270,893,894,897,899,901,903,905,907,909,911,914],{"class":272,"line":292},[270,895,896],{"class":280},"assignments ",[270,898,329],{"class":328},[270,900,750],{"class":280},[270,902,201],{"class":305},[270,904,437],{"class":363},[270,906,201],{"class":305},[270,908,207],{"class":308},[270,910,312],{"class":305},[270,912,913],{"class":308},"costs",[270,915,319],{"class":305},[270,917,918,921,923,926,928,931,933],{"class":272,"line":299},[270,919,920],{"class":426},"print",[270,922,312],{"class":305},[270,924,925],{"class":308},"assignments",[270,927,201],{"class":305},[270,929,930],{"class":363},"shape",[270,932,888],{"class":305},[270,934,935],{"class":562},"             # torch.Size([32, 48])\n",[270,937,938,940,943,945,948,951,953,956,959],{"class":272,"line":322},[270,939,920],{"class":426},[270,941,942],{"class":305},"((",[270,944,896],{"class":308},[270,946,947],{"class":328},">=",[270,949,950],{"class":315}," 0",[270,952,608],{"class":305},[270,954,955],{"class":308},"all",[270,957,958],{"class":305},"())",[270,960,961],{"class":562},"      # True: all rows matched (square)\n",[187,963,964,965,968,969,201],{},"On CPU, the dispatcher uses PyTorch's ",[190,966,967],{},"parallel_for"," to distribute problems across threads. On CUDA, a tiled shared-memory kernel solves one problem per CUDA block; this is restricted to square inputs with ",[190,970,971],{},"K \u003C= 64",[261,973,975],{"className":263,"code":974,"language":265,"meta":266,"style":266},"costs_gpu = torch.rand(64, 32, 32, device=\"cuda\")\nassignments_gpu = torchmatch.assignment.solve(costs_gpu)  # uses jonker_dense_batch CUDA\nprint(assignments_gpu.shape)   # torch.Size([64, 32])\n",[190,976,977,1018,1045],{"__ignoreMap":266},[270,978,979,982,984,986,988,990,992,994,996,999,1001,1003,1005,1008,1010,1012,1014,1016],{"class":272,"line":273},[270,980,981],{"class":280},"costs_gpu ",[270,983,329],{"class":328},[270,985,332],{"class":280},[270,987,201],{"class":305},[270,989,337],{"class":308},[270,991,312],{"class":305},[270,993,342],{"class":315},[270,995,345],{"class":305},[270,997,998],{"class":315}," 32",[270,1000,345],{"class":305},[270,1002,998],{"class":315},[270,1004,345],{"class":305},[270,1006,1007],{"class":353}," device",[270,1009,329],{"class":328},[270,1011,390],{"class":389},[270,1013,793],{"class":393},[270,1015,390],{"class":389},[270,1017,319],{"class":305},[270,1019,1020,1023,1025,1027,1029,1031,1033,1035,1037,1040,1042],{"class":272,"line":284},[270,1021,1022],{"class":280},"assignments_gpu ",[270,1024,329],{"class":328},[270,1026,750],{"class":280},[270,1028,201],{"class":305},[270,1030,437],{"class":363},[270,1032,201],{"class":305},[270,1034,207],{"class":308},[270,1036,312],{"class":305},[270,1038,1039],{"class":308},"costs_gpu",[270,1041,888],{"class":305},[270,1043,1044],{"class":562},"  # uses jonker_dense_batch CUDA\n",[270,1046,1047,1049,1051,1054,1056,1058,1060],{"class":272,"line":292},[270,1048,920],{"class":426},[270,1050,312],{"class":305},[270,1052,1053],{"class":308},"assignments_gpu",[270,1055,201],{"class":305},[270,1057,930],{"class":363},[270,1059,888],{"class":305},[270,1061,1062],{"class":562},"   # torch.Size([64, 32])\n",[187,1064,1065,1066,1069],{},"For ",[190,1067,1068],{},"K > 64"," on CUDA, move the tensor to CPU for the solve and transfer the result back:",[261,1071,1073],{"className":263,"code":1072,"language":265,"meta":266,"style":266},"costs_large = torch.rand(16, 128, 128, device=\"cuda\")\nassignments = torchmatch.assignment.solve(costs_large.cpu()).to(\"cuda\")\n",[190,1074,1075,1115],{"__ignoreMap":266},[270,1076,1077,1080,1082,1084,1086,1088,1090,1093,1095,1097,1099,1101,1103,1105,1107,1109,1111,1113],{"class":272,"line":273},[270,1078,1079],{"class":280},"costs_large ",[270,1081,329],{"class":328},[270,1083,332],{"class":280},[270,1085,201],{"class":305},[270,1087,337],{"class":308},[270,1089,312],{"class":305},[270,1091,1092],{"class":315},"16",[270,1094,345],{"class":305},[270,1096,738],{"class":315},[270,1098,345],{"class":305},[270,1100,738],{"class":315},[270,1102,345],{"class":305},[270,1104,1007],{"class":353},[270,1106,329],{"class":328},[270,1108,390],{"class":389},[270,1110,793],{"class":393},[270,1112,390],{"class":389},[270,1114,319],{"class":305},[270,1116,1117,1119,1121,1123,1125,1127,1129,1131,1133,1136,1138,1141,1144,1146,1148,1150,1152,1154],{"class":272,"line":284},[270,1118,896],{"class":280},[270,1120,329],{"class":328},[270,1122,750],{"class":280},[270,1124,201],{"class":305},[270,1126,437],{"class":363},[270,1128,201],{"class":305},[270,1130,207],{"class":308},[270,1132,312],{"class":305},[270,1134,1135],{"class":308},"costs_large",[270,1137,201],{"class":305},[270,1139,1140],{"class":308},"cpu",[270,1142,1143],{"class":305},"()).",[270,1145,786],{"class":308},[270,1147,312],{"class":305},[270,1149,390],{"class":389},[270,1151,793],{"class":393},[270,1153,390],{"class":389},[270,1155,319],{"class":305},[182,1157,1159],{"id":1158},"unpacked-output-matched-pairs-and-unmatched-sets","Unpacked output: matched pairs and unmatched sets",[187,1161,1162,1163,1166,1167,1170,1171,1174,1175,1178,1179,1182,1183,1186],{},"The default batch output is shape ",[190,1164,1165],{},"(B, N)",": entry ",[190,1168,1169],{},"[b, i]"," is the column assigned to row ",[190,1172,1173],{},"i"," in problem ",[190,1176,1177],{},"b",", or ",[190,1180,1181],{},"-1"," if unmatched. Pass ",[190,1184,1185],{},"unpack=True"," to get matched pairs and unmatched sets in one call without a Python loop:",[261,1188,1190],{"className":263,"code":1189,"language":265,"meta":266,"style":266},"costs = torch.rand(16, 40, 40)\nmatches, unmatched_rows, unmatched_cols, n_matched = torchmatch.assignment.solve(\n    costs, unpack=True,\n)\n# matches[b, k]         : (row, col) pair of the k-th match in problem b\n# unmatched_rows[b, k]  : index of the k-th unmatched row in problem b\n# unmatched_cols[b, k]  : index of the k-th unmatched column in problem b\n# n_matched[b]          : actual number of matches in problem b (rest are padding -1)\n\n# Use a specific batch element:\nb = 0\nnm = int(n_matched[b].item())\nreal_matches = matches[b, :nm]           # shape (nm, 2)\nlost_rows    = unmatched_rows[b, :costs.size(1) - nm]\nnew_cols     = unmatched_cols[b, :costs.size(2) - nm]\n",[190,1191,1192,1219,1254,1273,1277,1282,1287,1292,1297,1301,1306,1316,1343,1371,1412],{"__ignoreMap":266},[270,1193,1194,1196,1198,1200,1202,1204,1206,1208,1210,1213,1215,1217],{"class":272,"line":273},[270,1195,863],{"class":280},[270,1197,329],{"class":328},[270,1199,332],{"class":280},[270,1201,201],{"class":305},[270,1203,337],{"class":308},[270,1205,312],{"class":305},[270,1207,1092],{"class":315},[270,1209,345],{"class":305},[270,1211,1212],{"class":315}," 40",[270,1214,345],{"class":305},[270,1216,1212],{"class":315},[270,1218,319],{"class":305},[270,1220,1221,1224,1226,1229,1231,1234,1236,1239,1241,1243,1245,1247,1249,1251],{"class":272,"line":284},[270,1222,1223],{"class":280},"matches",[270,1225,345],{"class":305},[270,1227,1228],{"class":280}," unmatched_rows",[270,1230,345],{"class":305},[270,1232,1233],{"class":280}," unmatched_cols",[270,1235,345],{"class":305},[270,1237,1238],{"class":280}," n_matched ",[270,1240,329],{"class":328},[270,1242,750],{"class":280},[270,1244,201],{"class":305},[270,1246,437],{"class":363},[270,1248,201],{"class":305},[270,1250,207],{"class":308},[270,1252,1253],{"class":305},"(\n",[270,1255,1256,1259,1261,1264,1266,1270],{"class":272,"line":292},[270,1257,1258],{"class":308},"    costs",[270,1260,345],{"class":305},[270,1262,1263],{"class":353}," unpack",[270,1265,329],{"class":328},[270,1267,1269],{"class":1268},"s39Yj","True",[270,1271,1272],{"class":305},",\n",[270,1274,1275],{"class":272,"line":299},[270,1276,319],{"class":305},[270,1278,1279],{"class":272,"line":322},[270,1280,1281],{"class":562},"# matches[b, k]         : (row, col) pair of the k-th match in problem b\n",[270,1283,1284],{"class":272,"line":369},[270,1285,1286],{"class":562},"# unmatched_rows[b, k]  : index of the k-th unmatched row in problem b\n",[270,1288,1289],{"class":272,"line":374},[270,1290,1291],{"class":562},"# unmatched_cols[b, k]  : index of the k-th unmatched column in problem b\n",[270,1293,1294],{"class":272,"line":418},[270,1295,1296],{"class":562},"# n_matched[b]          : actual number of matches in problem b (rest are padding -1)\n",[270,1298,1299],{"class":272,"line":452},[270,1300,296],{"emptyLinePlaceholder":295},[270,1302,1303],{"class":272,"line":470},[270,1304,1305],{"class":562},"# Use a specific batch element:\n",[270,1307,1308,1311,1313],{"class":272,"line":516},[270,1309,1310],{"class":280},"b ",[270,1312,329],{"class":328},[270,1314,1315],{"class":315}," 0\n",[270,1317,1318,1321,1323,1327,1329,1332,1334,1336,1338,1340],{"class":272,"line":559},[270,1319,1320],{"class":280},"nm ",[270,1322,329],{"class":328},[270,1324,1326],{"class":1325},"sZMiF"," int",[270,1328,312],{"class":305},[270,1330,1331],{"class":308},"n_matched",[270,1333,481],{"class":305},[270,1335,1177],{"class":308},[270,1337,501],{"class":305},[270,1339,510],{"class":308},[270,1341,1342],{"class":305},"())\n",[270,1344,1345,1348,1350,1353,1355,1357,1359,1362,1365,1368],{"class":272,"line":566},[270,1346,1347],{"class":280},"real_matches ",[270,1349,329],{"class":328},[270,1351,1352],{"class":280}," matches",[270,1354,481],{"class":305},[270,1356,1177],{"class":280},[270,1358,345],{"class":305},[270,1360,1361],{"class":305}," :",[270,1363,1364],{"class":280},"nm",[270,1366,1367],{"class":305},"]",[270,1369,1370],{"class":562},"           # shape (nm, 2)\n",[270,1372,1374,1377,1379,1381,1383,1385,1387,1389,1391,1393,1396,1398,1401,1403,1406,1409],{"class":272,"line":1373},14,[270,1375,1376],{"class":280},"lost_rows    ",[270,1378,329],{"class":328},[270,1380,1228],{"class":280},[270,1382,481],{"class":305},[270,1384,1177],{"class":280},[270,1386,345],{"class":305},[270,1388,1361],{"class":305},[270,1390,913],{"class":280},[270,1392,201],{"class":305},[270,1394,1395],{"class":308},"size",[270,1397,312],{"class":305},[270,1399,1400],{"class":315},"1",[270,1402,888],{"class":305},[270,1404,1405],{"class":328}," -",[270,1407,1408],{"class":280}," nm",[270,1410,1411],{"class":305},"]\n",[270,1413,1415,1418,1420,1422,1424,1426,1428,1430,1432,1434,1436,1438,1441,1443,1445,1447],{"class":272,"line":1414},15,[270,1416,1417],{"class":280},"new_cols     ",[270,1419,329],{"class":328},[270,1421,1233],{"class":280},[270,1423,481],{"class":305},[270,1425,1177],{"class":280},[270,1427,345],{"class":305},[270,1429,1361],{"class":305},[270,1431,913],{"class":280},[270,1433,201],{"class":305},[270,1435,1395],{"class":308},[270,1437,312],{"class":305},[270,1439,1440],{"class":315},"2",[270,1442,888],{"class":305},[270,1444,1405],{"class":328},[270,1446,1408],{"class":280},[270,1448,1411],{"class":305},[187,1450,1451,1452,1455],{},"The ",[190,1453,1454],{},"_unpacked"," variants cost about 5 percent more than the packed variants. Use them whenever you would otherwise iterate over the batch in Python to compute the unmatched sets.",[182,1457,1459],{"id":1458},"gpu-vs-cpu-the-practical-guideline","GPU vs CPU: the practical guideline",[187,1461,1462,1463,1465,1466,1468,1469,1471,1472,1474,1475,1478],{},"For small to medium problems (",[190,1464,255],{},"), CPU is almost always faster than CUDA for assignment solving. The JV inner loop runs sequentially per problem, and a modern AVX2 core finishes it before the GPU has moved the first iteration flag to host memory. CUDA wins on integer-tied costs (where ",[190,1467,585],{}," often beats even ",[190,1470,239],{}," on CPU) and for batched square problems with ",[190,1473,971],{}," using the tiled kernel (the only CUDA-graph-safe option). The ",[1476,1477,37],"a",{"href":38}," page has the full benchmark-backed decision tree.",[182,1480,1482],{"id":1481},"see-also","See also",[217,1484,1485,1491],{},[220,1486,1487,1490],{},[1476,1488,1489],{"href":38},"Choosing the right op",": the decision tree behind AUTO, with benchmark numbers.",[220,1492,1493,1495],{},[1476,1494,33],{"href":34},": exact input\u002Foutput shapes and constraints for every op.",[1497,1498,1499],"style",{},"html pre.shiki code .sVHd0, html code.shiki .sVHd0{--shiki-light:#39ADB5;--shiki-light-font-style:italic;--shiki-default:#D73A49;--shiki-default-font-style:inherit;--shiki-dark:#F97583;--shiki-dark-font-style:inherit}html pre.shiki code .su5hD, html code.shiki .su5hD{--shiki-light:#90A4AE;--shiki-default:#24292E;--shiki-dark:#E1E4E8}html pre.shiki code .sP7_E, html code.shiki .sP7_E{--shiki-light:#39ADB5;--shiki-default:#24292E;--shiki-dark:#E1E4E8}html pre.shiki code .slqww, html code.shiki .slqww{--shiki-light:#6182B8;--shiki-default:#24292E;--shiki-dark:#E1E4E8}html pre.shiki code .srdBf, html code.shiki .srdBf{--shiki-light:#F76D47;--shiki-default:#005CC5;--shiki-dark:#79B8FF}html pre.shiki code .smGrS, html code.shiki .smGrS{--shiki-light:#39ADB5;--shiki-default:#D73A49;--shiki-dark:#F97583}html pre.shiki code .s99_P, html code.shiki .s99_P{--shiki-light:#90A4AE;--shiki-light-font-style:italic;--shiki-default:#E36209;--shiki-default-font-style:inherit;--shiki-dark:#FFAB70;--shiki-dark-font-style:inherit}html pre.shiki code .skxfh, html code.shiki .skxfh{--shiki-light:#E53935;--shiki-default:#24292E;--shiki-dark:#E1E4E8}html pre.shiki code .sjJ54, html code.shiki .sjJ54{--shiki-light:#39ADB5;--shiki-default:#032F62;--shiki-dark:#9ECBFF}html pre.shiki code .s_sjI, html code.shiki .s_sjI{--shiki-light:#91B859;--shiki-default:#032F62;--shiki-dark:#9ECBFF}html pre.shiki code .sptTA, html code.shiki .sptTA{--shiki-light:#6182B8;--shiki-default:#005CC5;--shiki-dark:#79B8FF}html pre.shiki code .sbsja, html code.shiki .sbsja{--shiki-light:#9C3EDA;--shiki-default:#D73A49;--shiki-dark:#F97583}html pre.shiki code .sutJx, html code.shiki .sutJx{--shiki-light:#90A4AE;--shiki-light-font-style:italic;--shiki-default:#6A737D;--shiki-default-font-style:inherit;--shiki-dark:#6A737D;--shiki-dark-font-style:inherit}html .light .shiki span {color: var(--shiki-light);background: var(--shiki-light-bg);font-style: var(--shiki-light-font-style);font-weight: var(--shiki-light-font-weight);text-decoration: var(--shiki-light-text-decoration);}html.light .shiki span {color: var(--shiki-light);background: var(--shiki-light-bg);font-style: var(--shiki-light-font-style);font-weight: var(--shiki-light-font-weight);text-decoration: var(--shiki-light-text-decoration);}html .default .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .dark .shiki span {color: var(--shiki-dark);background: var(--shiki-dark-bg);font-style: var(--shiki-dark-font-style);font-weight: var(--shiki-dark-font-weight);text-decoration: var(--shiki-dark-text-decoration);}html.dark .shiki span {color: var(--shiki-dark);background: var(--shiki-dark-bg);font-style: var(--shiki-dark-font-style);font-weight: var(--shiki-dark-font-weight);text-decoration: var(--shiki-dark-text-decoration);}html pre.shiki code .s39Yj, html code.shiki .s39Yj{--shiki-light:#39ADB5;--shiki-default:#005CC5;--shiki-dark:#79B8FF}html pre.shiki code .sZMiF, html code.shiki .sZMiF{--shiki-light:#E2931D;--shiki-default:#005CC5;--shiki-dark:#79B8FF}",{"title":266,"searchDepth":292,"depth":292,"links":1501},[1502,1503,1504,1505,1506,1507,1508,1509,1510],{"id":184,"depth":284,"text":185},{"id":211,"depth":284,"text":212},{"id":572,"depth":284,"text":573},{"id":614,"depth":284,"text":615},{"id":703,"depth":284,"text":704},{"id":831,"depth":284,"text":832},{"id":1158,"depth":284,"text":1159},{"id":1458,"depth":284,"text":1459},{"id":1481,"depth":284,"text":1482},"Choosing between jonker_scalar, jonker_dense, jonker_compact, munkres, and lawler, and how to solve thousands of problems at once.","md",{},{"title":51},{"title":176,"description":1511},"_QPiy2E1h6FCsp-UF_xV0FEGza4kNna-zigo2bD-ADA",[1518,1520],{"title":47,"path":48,"stem":49,"description":1519,"children":-1},"What the linear assignment problem is, why brute force fails, and how to solve your first problem with torchmatch.assignment.solve.",{"title":55,"path":56,"stem":57,"description":1521,"children":-1},"Building a SORT-style multi-object tracker step by step using torchmatch.assignment.solve and IoU cost matrices.",1785218169554]