from https://github.com/ARM-software/optimized-routines, commit 04884bd04eac4b251da4026900010ea7d8850edc Assume __FP_FAST_FMA implies __builtin_fma is inlined as a single instruction. code size change: +4588 bytes (+2540 bytes with fma). benchmark on x86_64 before, after, speedup: -Os: log rthruput: 12.61 ns/call 7.95 ns/call 1.59x log latency: 41.64 ns/call 23.38 ns/call 1.78x -O3: log rthruput: 12.51 ns/call 7.75 ns/call 1.61x log latency: 41.82 ns/call 23.55 ns/call 1.78x
29 lines
552 B
C
29 lines
552 B
C
/*
|
|
* Copyright (c) 2018, Arm Limited.
|
|
* SPDX-License-Identifier: MIT
|
|
*/
|
|
#ifndef _LOG_DATA_H
|
|
#define _LOG_DATA_H
|
|
|
|
#include <features.h>
|
|
|
|
#define LOG_TABLE_BITS 7
|
|
#define LOG_POLY_ORDER 6
|
|
#define LOG_POLY1_ORDER 12
|
|
extern hidden const struct log_data {
|
|
double ln2hi;
|
|
double ln2lo;
|
|
double poly[LOG_POLY_ORDER - 1]; /* First coefficient is 1. */
|
|
double poly1[LOG_POLY1_ORDER - 1];
|
|
struct {
|
|
double invc, logc;
|
|
} tab[1 << LOG_TABLE_BITS];
|
|
#if !__FP_FAST_FMA
|
|
struct {
|
|
double chi, clo;
|
|
} tab2[1 << LOG_TABLE_BITS];
|
|
#endif
|
|
} __log_data;
|
|
|
|
#endif
|