[{"data":1,"prerenderedAt":600},["ShallowReactive",2],{"navigation":3,"\u002Falgorithms\u002Ftransport\u002Fbuilding":174,"\u002Falgorithms\u002Ftransport\u002Fbuilding-surround":595},[4,8,101,165,170],{"title":5,"path":6,"stem":7},"Getting started","\u002Fgetting-started","1.getting-started",{"title":9,"path":10,"stem":11,"children":12},"Algorithms","\u002Falgorithms","2.algorithms",[13,15,58],{"title":9,"path":10,"stem":14},"2.algorithms\u002Findex",{"title":16,"path":17,"stem":18,"children":19},"Assignment","\u002Falgorithms\u002Fassignment","2.algorithms\u002F1.assignment\u002Findex",[20,21,25,29,32,36,40],{"title":16,"path":17,"stem":18},{"title":22,"path":23,"stem":24},"Quickstart","\u002Falgorithms\u002Fassignment\u002Fquickstart","2.algorithms\u002F1.assignment\u002F1.quickstart",{"title":26,"path":27,"stem":28},"Tracking","\u002Falgorithms\u002Fassignment\u002Ftracking","2.algorithms\u002F1.assignment\u002F2.tracking",{"title":9,"path":30,"stem":31},"\u002Falgorithms\u002Fassignment\u002Falgorithms","2.algorithms\u002F1.assignment\u002F3.algorithms",{"title":33,"path":34,"stem":35},"Reference","\u002Falgorithms\u002Fassignment\u002Freference","2.algorithms\u002F1.assignment\u002F4.reference",{"title":37,"path":38,"stem":39},"Choosing","\u002Falgorithms\u002Fassignment\u002Fchoosing","2.algorithms\u002F1.assignment\u002F5.choosing",{"title":41,"path":42,"stem":43,"children":44},"Tutorials","\u002Falgorithms\u002Fassignment\u002Ftutorials","2.algorithms\u002F1.assignment\u002F6.tutorials\u002Findex",[45,46,50,54],{"title":41,"path":42,"stem":43},{"title":47,"path":48,"stem":49},"Fundamentals","\u002Falgorithms\u002Fassignment\u002Ftutorials\u002Ffundamentals","2.algorithms\u002F1.assignment\u002F6.tutorials\u002F1.fundamentals",{"title":51,"path":52,"stem":53},"Backends","\u002Falgorithms\u002Fassignment\u002Ftutorials\u002Fbackends","2.algorithms\u002F1.assignment\u002F6.tutorials\u002F2.backends",{"title":55,"path":56,"stem":57},"Object tracking","\u002Falgorithms\u002Fassignment\u002Ftutorials\u002Ftracking","2.algorithms\u002F1.assignment\u002F6.tutorials\u002F3.tracking",{"title":59,"path":60,"stem":61,"children":62},"Transport","\u002Falgorithms\u002Ftransport","2.algorithms\u002F2.transport\u002Findex",[63,64,67,71,74,77,80,84],{"title":59,"path":60,"stem":61},{"title":22,"path":65,"stem":66},"\u002Falgorithms\u002Ftransport\u002Fquickstart","2.algorithms\u002F2.transport\u002F1.quickstart",{"title":68,"path":69,"stem":70},"Point-cloud tutorial","\u002Falgorithms\u002Ftransport\u002Fpoint-clouds","2.algorithms\u002F2.transport\u002F2.point-clouds",{"title":9,"path":72,"stem":73},"\u002Falgorithms\u002Ftransport\u002Falgorithms","2.algorithms\u002F2.transport\u002F3.algorithms",{"title":33,"path":75,"stem":76},"\u002Falgorithms\u002Ftransport\u002Freference","2.algorithms\u002F2.transport\u002F4.reference",{"title":37,"path":78,"stem":79},"\u002Falgorithms\u002Ftransport\u002Fchoosing","2.algorithms\u002F2.transport\u002F5.choosing",{"title":81,"path":82,"stem":83},"Building","\u002Falgorithms\u002Ftransport\u002Fbuilding","2.algorithms\u002F2.transport\u002F6.building",{"title":41,"path":85,"stem":86,"children":87},"\u002Falgorithms\u002Ftransport\u002Ftutorials","2.algorithms\u002F2.transport\u002F7.tutorials\u002Findex",[88,89,93,97],{"title":41,"path":85,"stem":86},{"title":90,"path":91,"stem":92},"Optimal transport","\u002Falgorithms\u002Ftransport\u002Ftutorials\u002Foptimal-transport","2.algorithms\u002F2.transport\u002F7.tutorials\u002F1.optimal-transport",{"title":94,"path":95,"stem":96},"Sinkhorn","\u002Falgorithms\u002Ftransport\u002Ftutorials\u002Fsinkhorn","2.algorithms\u002F2.transport\u002F7.tutorials\u002F2.sinkhorn",{"title":98,"path":99,"stem":100},"Point clouds","\u002Falgorithms\u002Ftransport\u002Ftutorials\u002Fpoint-clouds","2.algorithms\u002F2.transport\u002F7.tutorials\u002F3.point-clouds",{"title":102,"path":103,"stem":104,"children":105},"Resources","\u002Fresources","3.resources",[106,108,147,151,155],{"title":102,"path":103,"stem":107},"3.resources\u002Findex",{"title":41,"path":109,"stem":110,"children":111},"\u002Fresources\u002Ftutorials","3.resources\u002F1.tutorials\u002Findex",[112,113,131],{"title":41,"path":109,"stem":110},{"title":16,"path":114,"stem":115,"children":116,"page":130},"\u002Fresources\u002Ftutorials\u002Fassignment","3.resources\u002F1.tutorials\u002Fassignment",[117,122,126],{"title":118,"path":119,"stem":120,"icon":121},"Tutorial 1 — The Assignment Problem","\u002Fresources\u002Ftutorials\u002Fassignment\u002F01_the_assignment_problem","3.resources\u002F1.tutorials\u002Fassignment\u002F01_the_assignment_problem","i-lucide-notebook",{"title":123,"path":124,"stem":125,"icon":121},"Tutorial 2 — Backends and Batching","\u002Fresources\u002Ftutorials\u002Fassignment\u002F02_backends_and_batching","3.resources\u002F1.tutorials\u002Fassignment\u002F02_backends_and_batching",{"title":127,"path":128,"stem":129,"icon":121},"Tutorial 3 — Object Tracking with the Assignment Problem","\u002Fresources\u002Ftutorials\u002Fassignment\u002F03_object_tracking","3.resources\u002F1.tutorials\u002Fassignment\u002F03_object_tracking",false,{"title":59,"path":132,"stem":133,"children":134,"page":130},"\u002Fresources\u002Ftutorials\u002Ftransport","3.resources\u002F1.tutorials\u002Ftransport",[135,139,143],{"title":136,"path":137,"stem":138,"icon":121},"Tutorial 1 — What Is Optimal Transport?","\u002Fresources\u002Ftutorials\u002Ftransport\u002F01_optimal_transport","3.resources\u002F1.tutorials\u002Ftransport\u002F01_optimal_transport",{"title":140,"path":141,"stem":142,"icon":121},"Tutorial 2 — The Sinkhorn Algorithm","\u002Fresources\u002Ftutorials\u002Ftransport\u002F02_sinkhorn_algorithm","3.resources\u002F1.tutorials\u002Ftransport\u002F02_sinkhorn_algorithm",{"title":144,"path":145,"stem":146,"icon":121},"Tutorial 3 — Point-Cloud OT and Shape Learning","\u002Fresources\u002Ftutorials\u002Ftransport\u002F03_point_clouds","3.resources\u002F1.tutorials\u002Ftransport\u002F03_point_clouds",{"title":148,"path":149,"stem":150},"Assignment applications","\u002Fresources\u002Fassignment-applications","3.resources\u002F2.assignment-applications",{"title":152,"path":153,"stem":154},"Transport applications","\u002Fresources\u002Ftransport-applications","3.resources\u002F3.transport-applications",{"title":156,"path":157,"stem":158,"children":159},"Benchmarks","\u002Fresources\u002Fbenchmarks","3.resources\u002F4.benchmarks\u002Findex",[160,161],{"title":156,"path":157,"stem":158},{"title":162,"path":163,"stem":164},"Contributing benchmarks","\u002Fresources\u002Fbenchmarks\u002Fcontributing","3.resources\u002F4.benchmarks\u002Fcontributing",{"title":166,"path":167,"stem":168,"icon":169},"API Reference","\u002Fapi","4.api","i-lucide-package",{"title":171,"path":172,"stem":173},"References","\u002Freferences","5.references",{"id":175,"title":81,"api":176,"body":177,"description":589,"extension":590,"links":176,"meta":591,"navigation":592,"path":82,"seo":593,"stem":83,"__hash__":594},"docs\u002F2.algorithms\u002F2.transport\u002F6.building.md",null,{"type":178,"value":179,"toc":581},"minimark",[180,185,189,230,241,245,255,376,380,455,469,473,510,514,529,533,541,577],[181,182,184],"h2",{"id":183},"build-vs-runtime","Build vs runtime",[186,187,188],"p",{},"torchmatch supports two runtime paths:",[190,191,192,209],"ul",{},[193,194,195,199,200,204,205,208],"li",{},[196,197,198],"strong",{},"Prebuilt",": a ",[201,202,203],"code",{},".so"," compiled for the right Python version and PyTorch build (ABI) ships in\nthe wheel and loads via ",[201,206,207],{},"torch.ops.load_library"," at first call. No\ncompiler needed at install time.",[193,210,211,214,215,217,218,221,222,225,226,229],{},[196,212,213],{},"JIT",": when no matching prebuilt ",[201,216,203],{}," is present (source-distribution install (sdist),\nABI mismatch, or explicit override via ",[201,219,220],{},"TORCHMATCH_FORCE_JIT=1","), the loader compiles the\nC++\u002FCUDA sources via ",[201,223,224],{},"torch.utils.cpp_extension.load",". The result\ncaches in ",[201,227,228],{},"$TORCH_EXTENSIONS_DIR",".",[186,231,232,233,236,237,240],{},"Both paths register the same ",[201,234,235],{},"torch.ops.assignment.*"," and ",[201,238,239],{},"torch.ops.transport.*"," ops; the choice is\ntransparent to callers.",[181,242,244],{"id":243},"building-wheels","Building wheels",[186,246,247,248,251,252,229],{},"The build system pairs ",[201,249,250],{},"setuptools"," with\n",[201,253,254],{},"torch.utils.cpp_extension.BuildExtension",[256,257,262],"pre",{"className":258,"code":259,"language":260,"meta":261,"style":261},"language-bash shiki shiki-themes material-theme-lighter github-light github-dark","# default: CPU extension always; CUDA extension when a toolchain is found\n# (torch.utils.cpp_extension.CUDA_HOME is not None)\npip wheel . -w dist\u002F\n\n# CPU-only wheel\nTORCHMATCH_SKIP_CUDA=1 pip wheel . -w dist\u002F\n\n# CUDA GPU architecture targets (e.g., `8.6` for Ampere) passed to the `nvcc` compiler\nTORCH_CUDA_ARCH_LIST=\"8.0;8.6;8.9;9.0\" pip wheel . -w dist\u002F\n","bash","",[201,263,264,273,279,300,307,313,338,343,349],{"__ignoreMap":261},[265,266,269],"span",{"class":267,"line":268},"line",1,[265,270,272],{"class":271},"sutJx","# default: CPU extension always; CUDA extension when a toolchain is found\n",[265,274,276],{"class":267,"line":275},2,[265,277,278],{"class":271},"# (torch.utils.cpp_extension.CUDA_HOME is not None)\n",[265,280,282,286,290,293,297],{"class":267,"line":281},3,[265,283,285],{"class":284},"sbgvK","pip",[265,287,289],{"class":288},"s_sjI"," wheel",[265,291,292],{"class":288}," .",[265,294,296],{"class":295},"stzsN"," -w",[265,298,299],{"class":288}," dist\u002F\n",[265,301,303],{"class":267,"line":302},4,[265,304,306],{"emptyLinePlaceholder":305},true,"\n",[265,308,310],{"class":267,"line":309},5,[265,311,312],{"class":271},"# CPU-only wheel\n",[265,314,316,320,324,327,330,332,334,336],{"class":267,"line":315},6,[265,317,319],{"class":318},"su5hD","TORCHMATCH_SKIP_CUDA",[265,321,323],{"class":322},"smGrS","=",[265,325,326],{"class":288},"1",[265,328,329],{"class":284}," pip",[265,331,289],{"class":288},[265,333,292],{"class":288},[265,335,296],{"class":295},[265,337,299],{"class":288},[265,339,341],{"class":267,"line":340},7,[265,342,306],{"emptyLinePlaceholder":305},[265,344,346],{"class":267,"line":345},8,[265,347,348],{"class":271},"# CUDA GPU architecture targets (e.g., `8.6` for Ampere) passed to the `nvcc` compiler\n",[265,350,352,355,357,361,364,366,368,370,372,374],{"class":267,"line":351},9,[265,353,354],{"class":318},"TORCH_CUDA_ARCH_LIST",[265,356,323],{"class":322},[265,358,360],{"class":359},"sjJ54","\"",[265,362,363],{"class":288},"8.0;8.6;8.9;9.0",[265,365,360],{"class":359},[265,367,329],{"class":284},[265,369,289],{"class":288},[265,371,292],{"class":288},[265,373,296],{"class":295},[265,375,299],{"class":288},[181,377,379],{"id":378},"build-time-environment-variables","Build-time environment variables",[381,382,383,396],"table",{},[384,385,386],"thead",{},[387,388,389,393],"tr",{},[390,391,392],"th",{},"Variable",[390,394,395],{},"Effect",[397,398,399,410,420,438],"tbody",{},[387,400,401,407],{},[402,403,404],"td",{},[201,405,406],{},"TORCHMATCH_SKIP_CUDA=1",[402,408,409],{},"Skip the CUDA extensions; produce a CPU-only wheel",[387,411,412,417],{},[402,413,414],{},[201,415,416],{},"TORCHMATCH_SKIP_CPU=1",[402,418,419],{},"Skip the CPU extensions (rarely useful)",[387,421,422,427],{},[402,423,424],{},[201,425,426],{},"TORCHMATCH_SKIP_TRANSPORT=1",[402,428,429,430,433,434,437],{},"Skip both transport extensions (CPU and CUDA); produces an assignment-only wheel. The ",[201,431,432],{},"torchmatch.transport"," namespace still imports cleanly because the Sinkhorn-family ops are registered in pure Python; only ",[201,435,436],{},"EXACT_EMD"," becomes unavailable.",[387,439,440,444],{},[402,441,442],{},[201,443,354],{},[402,445,446,447,450,451,454],{},"Comma-separated CUDA GPU architecture targets (e.g., ",[201,448,449],{},"8.6"," for Ampere) passed to the ",[201,452,453],{},"nvcc"," compiler",[186,456,457,458,460,461,464,465,468],{},"When the NVIDIA CUDA compiler (",[201,459,453],{},") is absent — meaning ",[201,462,463],{},"torch.utils.cpp_extension.CUDA_HOME"," is ",[201,466,467],{},"None"," — the CUDA extensions are skipped silently, so building from an sdist on a\nCPU-only host still produces a usable CPU wheel.",[181,470,472],{"id":471},"runtime-overrides","Runtime overrides",[381,474,475,483],{},[384,476,477],{},[387,478,479,481],{},[390,480,392],{},[390,482,395],{},[397,484,485,497],{},[387,486,487,491],{},[402,488,489],{},[201,490,220],{},[402,492,493,494,496],{},"Skip the prebuilt ",[201,495,203],{}," and recompile from source. Useful for diagnosing ABI mismatches or developing C++ changes.",[387,498,499,504],{},[402,500,501],{},[201,502,503],{},"TORCH_EXTENSIONS_DIR",[402,505,506,507,229],{},"(PyTorch-builtin) Where the JIT cache lives. Defaults to ",[201,508,509],{},"~\u002F.cache\u002Ftorch_extensions",[181,511,513],{"id":512},"cpu-simd","CPU SIMD",[186,515,516,517,520,521,524,525,528],{},"The CPU extension builds with ",[201,518,519],{},"-O3 -std=c++17 -mavx2 -mfma"," by\ndefault, enabling AVX2 and FMA vectorization for best performance on modern x86-64 CPUs.\nOn older x86 CPUs without AVX2, pip falls back to a source-distribution (sdist) install; the JIT compiler then queries ",[201,522,523],{},"torch.cpu._is_avx2_supported()"," at build time and omits the AVX2\u002FFMA flags automatically.\nThe ",[201,526,527],{},"jonker_scalar"," op works without SIMD.",[181,530,532],{"id":531},"source-layout","Source layout",[256,534,539],{"className":535,"code":537,"language":538},[536],"language-text","sources\u002Ftorchmatch\u002F\n├── __init__.py           # eager-loads the active sub-packages\n├── _loader.py            # shared prebuilt + JIT plumbing\n├── assignment\u002F           # active sub-package\n│   ├── __init__.py       # public API: solve, Backend, ops, load_cpu\u002Fload_cuda\n│   ├── _solve.py         # dispatcher and Backend enum\n│   ├── _greedy.py        # pure-PyTorch Kurtzberg 1962 heuristic\n│   ├── ops.py            # direct handles to torch.ops.assignment.*\n│   ├── _cpu.py \u002F _cuda.py  # extension loaders\n│   ├── cpu\u002F              # CPU sources (jonker_*.{h,cpp}, ops.cpp)\n│   └── cuda\u002F             # CUDA sources (munkres.cu, hybrid.cu, lawler.cu, jonker_tiled.cuh, ...)\n└── transport\u002F            # optimal transport (OT) sub-package\n    ├── __init__.py       # eager-loads matrix; samples is lazy\n    ├── matrix\u002F           # cost-matrix face\n    │   ├── __init__.py   # public API: solve, Backend, ops, load_cpu\u002Fload_cuda\n    │   ├── _solve.py     # dispatcher and Backend enum\n    │   ├── _log_sinkhorn.py \u002F _sinkhorn_divergence.py \u002F _unbalanced_sinkhorn.py\n    │   ├── _exact_emd.py # Python wrapper around the EMD C++ extension\n    │   ├── _validate.py \u002F _schedule.py  # input validation, epsilon annealing schedule\n    │   ├── ops.py        # direct handles to torch.ops.transport.*\n    │   ├── _cpu.py \u002F _cuda.py  # extension loaders\n    │   ├── cpu\u002F          # CPU sources (exact_emd_op.cpp, ops.cpp, exact\u002F network simplex sources)\n    │   └── cuda\u002F         # CUDA sources (ops.cpp)\n    └── samples\u002F          # point-cloud face (CUDA-only, Triton kernels)\n        ├── __init__.py   # public API: loss\n        ├── _loss.py      # entry point\n        ├── _autograd.py  # custom_op + register_autograd\n        ├── _solvers.py \u002F _c_transform.py \u002F _cg.py \u002F _hvp.py \u002F _implicit_grad.py\n        └── kernels\u002F      # Triton kernels (flashstyle_sqeuclid, ...)\n","text",[201,540,537],{"__ignoreMap":261},[186,542,543,546,547,549,550,236,553,556,557,560,561,564,565,236,568,571,572,574,575,229],{},[201,544,545],{},"setup.py"," declares up to four ",[201,548,250],{}," extensions:\n",[201,551,552],{},"torchmatch._assignment_cpu_impl",[201,554,555],{},"torchmatch._assignment_cuda_impl","\n(via ",[201,558,559],{},"CppExtension"," \u002F ",[201,562,563],{},"CUDAExtension","), plus the matching\n",[201,566,567],{},"_transport_cpu_impl",[201,569,570],{},"_transport_cuda_impl"," translation units.\nCUDA extensions are included only when a CUDA toolchain is found and\n",[201,573,319],{}," is unset; transport extensions are skipped when\n",[201,576,426],{},[578,579,580],"style",{},"html pre.shiki code .sutJx, html code.shiki .sutJx{--shiki-light:#90A4AE;--shiki-light-font-style:italic;--shiki-default:#6A737D;--shiki-default-font-style:inherit;--shiki-dark:#6A737D;--shiki-dark-font-style:inherit}html pre.shiki code .sbgvK, html code.shiki .sbgvK{--shiki-light:#E2931D;--shiki-default:#6F42C1;--shiki-dark:#B392F0}html pre.shiki code .s_sjI, html code.shiki .s_sjI{--shiki-light:#91B859;--shiki-default:#032F62;--shiki-dark:#9ECBFF}html pre.shiki code .stzsN, html code.shiki .stzsN{--shiki-light:#91B859;--shiki-default:#005CC5;--shiki-dark:#79B8FF}html pre.shiki code .su5hD, html code.shiki .su5hD{--shiki-light:#90A4AE;--shiki-default:#24292E;--shiki-dark:#E1E4E8}html pre.shiki code .smGrS, html code.shiki .smGrS{--shiki-light:#39ADB5;--shiki-default:#D73A49;--shiki-dark:#F97583}html pre.shiki code .sjJ54, html code.shiki .sjJ54{--shiki-light:#39ADB5;--shiki-default:#032F62;--shiki-dark:#9ECBFF}html .light .shiki span {color: var(--shiki-light);background: var(--shiki-light-bg);font-style: var(--shiki-light-font-style);font-weight: var(--shiki-light-font-weight);text-decoration: var(--shiki-light-text-decoration);}html.light .shiki span {color: var(--shiki-light);background: var(--shiki-light-bg);font-style: var(--shiki-light-font-style);font-weight: var(--shiki-light-font-weight);text-decoration: var(--shiki-light-text-decoration);}html .default .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .dark .shiki span {color: var(--shiki-dark);background: var(--shiki-dark-bg);font-style: var(--shiki-dark-font-style);font-weight: var(--shiki-dark-font-weight);text-decoration: var(--shiki-dark-text-decoration);}html.dark .shiki span {color: var(--shiki-dark);background: var(--shiki-dark-bg);font-style: var(--shiki-dark-font-style);font-weight: var(--shiki-dark-font-weight);text-decoration: var(--shiki-dark-text-decoration);}",{"title":261,"searchDepth":281,"depth":281,"links":582},[583,584,585,586,587,588],{"id":183,"depth":275,"text":184},{"id":243,"depth":275,"text":244},{"id":378,"depth":275,"text":379},{"id":471,"depth":275,"text":472},{"id":512,"depth":275,"text":513},{"id":531,"depth":275,"text":532},"Wheel vs JIT runtime paths, build-time and runtime environment variables, CPU SIMD flags, and source layout.","md",{},{"title":81},{"title":81,"description":589},"dYqbnfB2ZCOuiNDUhiGzJVu-td_ett_6C250sh_lvjA",[596,598],{"title":37,"path":78,"stem":79,"description":597,"children":-1},"A decision guide for transport.matrix and transport.samples backends, organised by cost type, differentiability requirements, and problem scale.",{"title":41,"path":85,"stem":86,"description":599,"children":-1},"Hands-on Jupyter notebooks covering optimal transport — from earth-mover intuition through Sinkhorn and point-cloud Wasserstein losses.",1785218164353]