看起来你正在尝试做类似的事情:
#include <stdio.h>
double hullSpeed(double lgth)
{
double result;
__asm__(
"fldl %1\n\t" //st(0)=>st(1), st(0)=lgth . FLDL means load double float
"fsqrt\n\t" //st(0) = square root st(0)
"fmulp\n\t" //Multiplies st(0) and st(1) (1.34). Result in st(0)
: "=&t" (result) : "m" (lgth), "0" (1.34));
return result;
}
int main()
{
printf ("%f\n", hullSpeed(64.0));
}
我使用的模板可以简化,但为了演示目的就足够了。我们使用"=&t" 约束,因为我们在st(0) 中返回浮点堆栈顶部的结果,并且我们使用& 来表示早期的clobber(我们将使用顶部浮点堆栈在 1.34 中传递)。我们通过约束"m" (lgth) 将lgth 的地址与内存引用一起传递,"0"(1.34) 约束表示我们将在与参数0 相同的寄存器中传递1.34,在这种情况下是浮点数的顶部堆。这些是我们的汇编器将覆盖但不显示为输入或输出约束的寄存器(或内存)。
使用内联汇编器学习汇编语言是一种非常难学的方法。 x86 特有的机器约束可以在 x86 系列 下的here 中找到。关于约束修饰符的信息可以在here找到,关于GCC扩展汇编模板的信息可以在here找到。
我只是给你一个起点,因为 GCC 的内联汇编器的使用可能相当复杂,而且任何答案对于 Stackoverflow 的答案都可能过于宽泛。您使用带有 x87 浮点的内联汇编程序这一事实使它变得更加复杂。
一旦你掌握了约束和修饰符,另一种机制可以让编译器产生更好的汇编代码:
__asm__(
"fsqrt\n\t" // st(0) = square root st(0)
"fmulp\n\t" // Multiplies st(0) and st(1) (1.34). Result in st(0)
: "=t"(result) : "0"(lgth), "u" (1.34) : "st(1)");
提示:约束 "u" 在 x87 浮点寄存器 st(1) 中放置一个值。汇编器模板约束有效地将lgth 放在st(0) 中,将1.34 放在st(1) 中。 st(1) 在内联汇编完成后无效,因此我们将其列为 clobber。我们使用约束将我们的值放在浮点堆栈上。这样做的效果是减少了我们必须在汇编代码本身内完成的工作。
如果您正在开发 64 位应用程序,我强烈建议您至少使用 SSE/SSE2 进行基本浮点计算。上面的代码应该适用于 32 位和 64 位。在 64 位代码中,x87 浮点指令通常不如 SSE/SSE2 高效,但它们可以工作。
使用内联汇编和 x87 进行舍入
如果您尝试基于 x87 上的 4 种舍入模式之一进行舍入,则可以使用如下代码:
#include <stdint.h>
#include <stdio.h>
#define RND_CTL_BIT_SHIFT 10
typedef enum {
ROUND_NEAREST_EVEN = 0 << RND_CTL_BIT_SHIFT,
ROUND_MINUS_INF = 1 << RND_CTL_BIT_SHIFT,
ROUND_PLUS_INF = 2 << RND_CTL_BIT_SHIFT,
ROUND_TOWARD_ZERO = 3 << RND_CTL_BIT_SHIFT
} RoundingMode;
double roundd (double n, RoundingMode mode)
{
uint16_t cw; /* Storage for the current x87 control register */
uint16_t newcw; /* Storage for the new value of the control register */
uint16_t dummyreg; /* Temporary dummy register used in the template */
__asm__ __volatile__ (
"fstcw %w[cw] \n\t" /* Read current x87 control register into cw*/
"fwait \n\t" /* Do an fwait after an fstcw instruction */
"mov %w[cw],%w[treg] \n\t" /* ax = value in cw variable*/
"and $0xf3ff,%w[treg] \n\t" /* Set rounding mode bits 10 and 11 of control
register to zero*/
"or %w[rmode],%w[treg] \n\t" /* Set the rounding mode bits */
"mov %w[treg],%w[newcw]\n\t" /* newcw = value for new control reg value*/
"fldcw %w[newcw] \n\t" /* Set control register to newcw */
"frndint \n\t" /* st(0) = round(st(0)) */
"fldcw %w[cw] \n\t" /* restore control reg to orig value in cw*/
: [cw]"=m"(cw),
[newcw]"=m"(newcw),
[treg]"=&r"(dummyreg), /* Register constraint with dummy variable
allows compiler to choose available register */
[n]"+t"(n) /* +t constraint passes `n` through
top of FPU stack (st0) for both input&output*/
: [rmode]"rmi"((uint16_t)mode)); /* "g" constraint same as "rmi" */
return n;
}
double hullSpeed(double lgth)
{
double result;
__asm__(
"fsqrt\n\t" // st(0) = square root st(0)
"fmulp\n\t" // Multiplies st(0) and st(1) (1.34). Result in st(0)
: "=t"(result) : "0"(lgth), "u" (1.34) : "st(1)");
return result;
}
int main()
{
double dbHullSpeed = hullSpeed(64.0);
printf ("%f, %f\n", dbHullSpeed, roundd(dbHullSpeed, ROUND_NEAREST_EVEN));
printf ("%f, %f\n", dbHullSpeed, roundd(dbHullSpeed, ROUND_MINUS_INF));
printf ("%f, %f\n", dbHullSpeed, roundd(dbHullSpeed, ROUND_PLUS_INF));
printf ("%f, %f\n", dbHullSpeed, roundd(dbHullSpeed, ROUND_TOWARD_ZERO));
return 0;
}
正如您在 cmets 中指出的那样,此 Stackoverflow answer 中有等效代码,但它使用了多个 __asm__ 语句,您很好奇如何编码单个 __asm__ 语句。
四舍五入模式(0,1,2,3)可以在Intel Architecture Document中找到:
舍入模式 RC 字段
00B 四舍五入的结果最接近无限精确的结果。如果两个值同样接近,则结果是偶数值(即最低有效位为零的那个)。默认向下舍入(朝向 -∞)
01B 舍入结果最接近但不大于无限精确结果。向上舍入(朝向 +∞)
10B 舍入结果最接近但不小于无限精确结果。向零舍入(截断)
11B 舍入后的结果在绝对值上最接近但不大于无限精确结果。
在第 8.1.5 节(在第 8.1.5.3 节中具体描述的舍入模式)中有字段的描述。这 4 种舍入模式在 4.8.4 节的图 4-8 中定义。